Compare commits
14
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
23779e1ab9 | ||
|
|
772df0ac8d | ||
|
|
beae687cdd | ||
|
|
2b09c63c37 | ||
|
|
1b5dedc8f9 | ||
|
|
7996b3a776 | ||
|
|
96f6ca5665 | ||
|
|
76fca89cfb | ||
|
|
3ea5570959 | ||
|
|
5794e38af8 | ||
|
|
0b5a765923 | ||
|
|
4c4ddf3928 | ||
|
|
ba3fc58806 | ||
|
|
5fd3861229 |
@@ -162,7 +162,7 @@ jobs:
|
||||
|
||||
- name: get hypre
|
||||
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os != 'windows-latest'
|
||||
uses: mfem/github-actions/build-hypre@v2.4
|
||||
uses: mfem/github-actions/build-hypre@v2.2
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
@@ -171,7 +171,7 @@ jobs:
|
||||
|
||||
- name: get hypre (Windows)
|
||||
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os == 'windows-latest'
|
||||
uses: mfem/github-actions/build-hypre@v2.4
|
||||
uses: mfem/github-actions/build-hypre@v2.2
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
@@ -190,7 +190,7 @@ jobs:
|
||||
|
||||
- name: install metis
|
||||
if: matrix.mpi == 'par' && matrix.os != 'windows-latest' && steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.4
|
||||
uses: mfem/github-actions/build-metis@v2.2
|
||||
with:
|
||||
archive: ${{ env.METIS_ARCHIVE }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
@@ -217,7 +217,7 @@ jobs:
|
||||
|
||||
# MFEM build and test
|
||||
- name: build
|
||||
uses: mfem/github-actions/build-mfem@v2.4
|
||||
uses: mfem/github-actions/build-mfem@v2.3
|
||||
env:
|
||||
VCPKG_DEFAULT_BINARY_CACHE: ${{ github.workspace }}/vcpkg_cache
|
||||
with:
|
||||
@@ -263,15 +263,13 @@ jobs:
|
||||
if: matrix.build-system == 'cmake' && matrix.target == 'opt' && matrix.os != 'ubuntu-latest'
|
||||
run: |
|
||||
CTEST_CONFIG="Release"
|
||||
cd ${{ env.MFEM_TOP_DIR }}/build && \
|
||||
ctest --output-on-failure -C ${CTEST_CONFIG} || \
|
||||
ctest --rerun-failed --output-on-failure -C ${CTEST_CONFIG}
|
||||
cd ${{ env.MFEM_TOP_DIR }}/build && ctest --output-on-failure -C ${CTEST_CONFIG}
|
||||
shell: bash
|
||||
|
||||
# Code coverage (process and upload reports)
|
||||
- name: codecov
|
||||
if: matrix.codecov == 'YES'
|
||||
uses: mfem/github-actions/upload-coverage@v2.4
|
||||
uses: mfem/github-actions/upload-coverage@v2.2
|
||||
with:
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}
|
||||
project_dir: ${{ env.MFEM_TOP_DIR }}
|
||||
|
||||
@@ -57,7 +57,7 @@ jobs:
|
||||
|
||||
- name: Get Hypre
|
||||
if: steps.hypre-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-hypre@v2.4
|
||||
uses: mfem/github-actions/build-hypre@v2.2
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
@@ -72,14 +72,14 @@ jobs:
|
||||
|
||||
- name: Install Metis
|
||||
if: steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.4
|
||||
uses: mfem/github-actions/build-metis@v2.2
|
||||
with:
|
||||
archive: ${{ env.METIS_ARCHIVE }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
|
||||
# MFEM build and test
|
||||
- name: build-mfem
|
||||
uses: mfem/github-actions/build-mfem@v2.4
|
||||
uses: mfem/github-actions/build-mfem@v2.2
|
||||
with:
|
||||
os: ${{ runner.os }}
|
||||
target: opt
|
||||
|
||||
@@ -1,70 +0,0 @@
|
||||
# Copyright (c) 2010-2023, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
name: "Sanitizer"
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
- next
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
Serial:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- name: Cancel Previous Runs
|
||||
uses: styfle/cancel-workflow-action@0.11.0
|
||||
with:
|
||||
access_token: ${{ github.token }}
|
||||
|
||||
- name: MFEM Checkout
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
path: mfem
|
||||
|
||||
- name: MFEM Build
|
||||
uses: mfem/github-actions/build-mfem@v2.4
|
||||
with:
|
||||
os: ${{ runner.os }}
|
||||
target: opt
|
||||
mpi: seq
|
||||
hypre-dir: unused-hypre-dir
|
||||
metis-dir: unused-metis-dir
|
||||
mfem-dir: mfem
|
||||
build-system: make
|
||||
library-only: false
|
||||
config-options:
|
||||
CXX="clang++-14"
|
||||
CXXFLAGS="-g -O1 -std=c++11
|
||||
-fsanitize=address
|
||||
-fno-omit-frame-pointer
|
||||
-fsanitize-address-use-after-scope"
|
||||
|
||||
- name: MFEM Info
|
||||
working-directory: mfem
|
||||
run: make info
|
||||
|
||||
- name: MFEM Sanitize
|
||||
working-directory: mfem
|
||||
run:
|
||||
ASAN_OPTIONS="detect_leaks=1,
|
||||
strict_init_order=1,
|
||||
strict_string_checks=1,
|
||||
check_initialization_order=1,
|
||||
detect_stack_use_after_return=1"
|
||||
make test
|
||||
-27
@@ -29,7 +29,6 @@ CMakeFiles/
|
||||
config/_config.hpp
|
||||
config/config.mk
|
||||
config/sample-runs-build.log
|
||||
config/user.cmake
|
||||
config/user.mk
|
||||
doc/CodeDocumentation.conf
|
||||
doc/CodeDocumentation.html
|
||||
@@ -113,12 +112,6 @@ examples/ex25p-*.*
|
||||
examples/ex28_*
|
||||
examples/ex28p_*
|
||||
examples/flux.*
|
||||
examples/dsol.*
|
||||
examples/cond.*
|
||||
examples/cond_j.*
|
||||
examples/cond_mesh.*
|
||||
examples/port_mesh.*
|
||||
examples/port_mode.*
|
||||
|
||||
examples/amgx/ex1
|
||||
examples/amgx/ex1p
|
||||
@@ -217,11 +210,9 @@ miniapps/meshing/trimmer
|
||||
miniapps/meshing/reflector
|
||||
miniapps/meshing/mesh-optimizer
|
||||
miniapps/meshing/pmesh-optimizer
|
||||
miniapps/meshing/pmesh-fitting
|
||||
miniapps/meshing/minimal-surface
|
||||
miniapps/meshing/pminimal-surface
|
||||
miniapps/meshing/polar-nc
|
||||
miniapps/meshing/mesh-quality
|
||||
miniapps/meshing/mobius-strip.mesh
|
||||
miniapps/meshing/klein-bottle.mesh
|
||||
miniapps/meshing/toroid-*.mesh
|
||||
@@ -324,28 +315,12 @@ miniapps/solvers/ParaView
|
||||
miniapps/solvers/mesh.*
|
||||
miniapps/solvers/sol.*
|
||||
|
||||
miniapps/hdiv-linear-solver/darcy
|
||||
miniapps/hdiv-linear-solver/grad_div
|
||||
|
||||
miniapps/parelag/MultilevelHcurlHdivSolver
|
||||
miniapps/parelag/*.mesh
|
||||
|
||||
miniapps/multidomain/multidomain
|
||||
miniapps/hooke/hooke
|
||||
|
||||
miniapps/dpg/diffusion
|
||||
miniapps/dpg/pdiffusion
|
||||
miniapps/dpg/convection-diffusion
|
||||
miniapps/dpg/pconvection-diffusion
|
||||
miniapps/dpg/acoustics
|
||||
miniapps/dpg/pacoustics
|
||||
miniapps/dpg/maxwell
|
||||
miniapps/dpg/pmaxwell
|
||||
miniapps/dpg/ParaView
|
||||
|
||||
miniapps/spde/generate_random_field
|
||||
miniapps/spde/ParaView
|
||||
|
||||
# Unit test binary and outputs
|
||||
tests/unit/output_meshes
|
||||
tests/unit/unit_tests
|
||||
@@ -358,8 +333,6 @@ tests/unit/tmop_pa_tests_*
|
||||
tests/unit/ptmop_pa_tests_*
|
||||
tests/unit/ceed_tests
|
||||
tests/unit/debug_device_tests
|
||||
tests/unit/parallel_in_serial.mesh
|
||||
tests/unit/parallel_in_serial.gf
|
||||
|
||||
# Benchmark binaries
|
||||
tests/benchmarks/bench_ceed
|
||||
|
||||
@@ -22,10 +22,12 @@
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
|
||||
# every 5 seconds; we may want to add a counter for the number of
|
||||
# every 5 seconds; simply using no timeout, i.e. 'flock 9', causes the
|
||||
# command to hang indefinitely sometimes, so we use the timeout & retry
|
||||
# as a workaround; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
while ! flock -w 5 9; do
|
||||
true
|
||||
done
|
||||
echo "Acquired lock on '$PWD/autotest.lock'"
|
||||
date
|
||||
@@ -55,10 +57,12 @@
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
|
||||
# every 5 seconds; we may want to add a counter for the number of
|
||||
# every 5 seconds; simply using no timeout, i.e. 'flock 9', causes the
|
||||
# command to hang indefinitely sometimes, so we use the timeout & retry
|
||||
# as a workaround; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
while ! flock -w 5 9; do
|
||||
true
|
||||
done
|
||||
echo "Acquired lock on '$PWD/autotest.lock'"
|
||||
date
|
||||
|
||||
@@ -47,10 +47,12 @@ setup_baseline:
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
|
||||
# every 5 seconds; we may want to add a counter for the number of
|
||||
# every 5 seconds; simply using no timeout, i.e. 'flock 9', causes the
|
||||
# command to hang indefinitely sometimes, so we use the timeout & retry
|
||||
# as a workaround; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
while ! flock -w 5 9; do
|
||||
true
|
||||
done
|
||||
echo "Acquired lock on '$PWD/autotest.lock'"
|
||||
date
|
||||
|
||||
@@ -35,11 +35,13 @@ setup:
|
||||
(
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/mfem-data.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (mfem-data.lock) repeating the
|
||||
# try every 5 seconds; we may want to add a counter for the number of
|
||||
# try to get an exclusive lock on fd 9 (mfem-data.lock) repeating the try
|
||||
# every 5 seconds; simply using no timeout, i.e. 'flock 9', causes the
|
||||
# command to hang indefinitely sometimes, so we use the timeout & retry
|
||||
# as a workaround; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
while ! flock -w 5 9; do
|
||||
true
|
||||
done
|
||||
echo "Acquired lock on '$PWD/mfem-data.lock'"
|
||||
date
|
||||
@@ -67,10 +69,12 @@ setup:
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
|
||||
# every 5 seconds; we may want to add a counter for the number of
|
||||
# every 5 seconds; simply using no timeout, i.e. 'flock 9', causes the
|
||||
# command to hang indefinitely sometimes, so we use the timeout & retry
|
||||
# as a workaround; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
while ! flock -w 5 9; do
|
||||
true
|
||||
done
|
||||
echo "Acquired lock on '$PWD/autotest.lock'"
|
||||
date
|
||||
|
||||
@@ -14,14 +14,14 @@ stages:
|
||||
- build_and_test
|
||||
- report
|
||||
|
||||
opt_mpi_cuda_xl_16_1_1_12:
|
||||
opt_mpi_cuda_xl_16_1_1_8:
|
||||
variables:
|
||||
SPEC: "%xl@16.1.1.12 +mpi +cuda cuda_arch=70"
|
||||
SPEC: "%xl@16.1.1.8 +mpi +cuda cuda_arch=70"
|
||||
extends: .build_and_test_on_lassen
|
||||
|
||||
opt_mpi_cuda_hypre_cuda_xl:
|
||||
variables:
|
||||
SPEC: "%xl@16.1.1.12 +mpi +cuda cuda_arch=70 ^hypre+cuda~shared cuda_arch=70"
|
||||
SPEC: "%xl@16.1.1.8 +mpi +cuda cuda_arch=70 ^hypre+cuda~shared cuda_arch=70"
|
||||
extends: .build_and_test_on_lassen
|
||||
|
||||
# Jobs report
|
||||
|
||||
@@ -51,8 +51,6 @@ cleanup:
|
||||
script:
|
||||
- echo "BUILD_ROOT=${BUILD_ROOT}"
|
||||
- rm -rf "${BUILD_ROOT}" || true
|
||||
- echo "CI_PROJECT_DIR=${CI_PROJECT_DIR}"
|
||||
- make -C "${CI_PROJECT_DIR}" distclean
|
||||
|
||||
report_baseline:
|
||||
extends: [.on_quartz]
|
||||
@@ -68,10 +66,12 @@ report_baseline:
|
||||
date
|
||||
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
|
||||
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
|
||||
# every 5 seconds; we may want to add a counter for the number of
|
||||
# every 5 seconds; simply using no timeout, i.e. 'flock 9', causes the
|
||||
# command to hang indefinitely sometimes, so we use the timeout & retry
|
||||
# as a workaround; we may want to add a counter for the number of
|
||||
# retries to interrupt a potential infinite loop
|
||||
while ! flock -n 9; do
|
||||
sleep 5
|
||||
while ! flock -w 5 9; do
|
||||
true
|
||||
done
|
||||
echo "Acquired lock on '$PWD/autotest.lock'"
|
||||
date
|
||||
@@ -82,14 +82,12 @@ report_baseline:
|
||||
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-${BASELINE_TEST}-${CI_COMMIT_REF_SLUG}"
|
||||
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir ${rundir})
|
||||
cp ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/* ${rundir}
|
||||
# We create an autotest-email.html file, because that's how we signal that there was a diff (temporary).
|
||||
if [[ -f ${rundir}/${BASELINE_TEST}.err ]]; then
|
||||
cp ${rundir}/${BASELINE_TEST}.err ${rundir}/autotest-email.html
|
||||
fi
|
||||
printf "%s\n" "" "Pipeline URL:" "$CI_PIPELINE_URL" \
|
||||
>> ${rundir}/pipeline.txt
|
||||
# We create an autotest-email.html file, because that's how we signal
|
||||
# that there was an error / diff (temporary).
|
||||
if [[ -f ${rundir}/${BASELINE_TEST}.err ]] || \
|
||||
[[ -f ${rundir}/${BASELINE_TEST}-${SYS_TYPE}.diff ]]; then
|
||||
cp ${rundir}/pipeline.txt ${rundir}/autotest-email.html
|
||||
fi
|
||||
msg="GitLab CI log for ${BASELINE_TEST} on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
|
||||
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
|
||||
git pull && \
|
||||
|
||||
@@ -27,39 +27,44 @@ allocate_resource:
|
||||
timeout: 6h
|
||||
|
||||
# GitLab jobs for the Quartz machine at LLNL
|
||||
debug_ser_gcc_10:
|
||||
debug_ser_gcc_4_9_3:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +debug~mpi"
|
||||
SPEC: "%gcc@4.9.3 +debug~mpi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
debug_par_gcc_10:
|
||||
debug_ser_gcc_6_1_0:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +debug+mpi"
|
||||
SPEC: "%gcc@6.1.0 +debug~mpi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_ser_gcc_10:
|
||||
debug_par_gcc_6_1_0:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 ~mpi"
|
||||
SPEC: "%gcc@6.1.0 +debug+mpi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_10:
|
||||
opt_ser_gcc_6_1_0:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1"
|
||||
SPEC: "%gcc@6.1.0 ~mpi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_10_sundials:
|
||||
opt_par_gcc_6_1_0:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +sundials"
|
||||
SPEC: "%gcc@6.1.0"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_10_petsc:
|
||||
opt_par_gcc_6_1_0_sundials:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +petsc ^petsc+mumps~superlu-dist"
|
||||
SPEC: "%gcc@6.1.0 +sundials"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_10_pumi:
|
||||
opt_par_gcc_6_1_0_petsc:
|
||||
variables:
|
||||
SPEC: "%gcc@10.3.1 +pumi"
|
||||
SPEC: "%gcc@6.1.0 +petsc ^petsc+mumps~superlu-dist"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_6_1_0_pumi:
|
||||
variables:
|
||||
SPEC: "%gcc@6.1.0 +pumi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
# Release
|
||||
|
||||
+28
-15
@@ -42,34 +42,47 @@ fi
|
||||
# post
|
||||
mkdir ${artifacts_path}
|
||||
|
||||
status=0
|
||||
if [[ -f ${BASELINE_TEST}.out ]]; then
|
||||
cp ${BASELINE_TEST}.out ${artifacts_path}
|
||||
fi
|
||||
if [[ -s ${glob_err} ]]; then
|
||||
echo "ERROR during ${BASELINE_TEST} execution"
|
||||
echo "Here is the ${glob_err} file content"
|
||||
if [[ -s ${glob_err} ]]
|
||||
then
|
||||
echo "ERROR during ${BASELINE_TEST} execution";
|
||||
echo "Here is the ${glob_err} file content";
|
||||
cat ${glob_err}
|
||||
cp ${glob_err} ${artifacts_path}/${glob_err}
|
||||
status=1
|
||||
fi
|
||||
if [[ -f ${base_patch} ]]; then
|
||||
exit 1;
|
||||
elif [[ ! -f ${base_patch} && ! -f ${base_out} ]]
|
||||
then
|
||||
echo "Something went WRONG in ${BASELINE_TEST}:";
|
||||
echo "Either ${base_patch} or ${base_out} should exists";
|
||||
exit 1;
|
||||
elif [[ -f ${base_patch} ]]
|
||||
then
|
||||
echo "${BASELINE_TEST}: Differences found, patch generated"
|
||||
cp ${base_patch} ${artifacts_path}/${base_patch}
|
||||
elif [[ -f ${base_out} ]]; then
|
||||
elif [[ -f ${base_out} ]]
|
||||
then
|
||||
echo "${BASELINE_TEST}: Differences found, replacement file generated"
|
||||
cp ${base_out} ${artifacts_path}/${base_out}
|
||||
fi
|
||||
|
||||
if [[ -f ${BASELINE_TEST}.out ]]; then
|
||||
cp ${BASELINE_TEST}.out ${artifacts_path}
|
||||
fi
|
||||
|
||||
# base_diff won't even exist if there is no difference.
|
||||
if [[ -f ${base_diff} ]]; then
|
||||
if [[ -f ${base_diff} ]]
|
||||
then
|
||||
echo "${BASELINE_TEST}: Relevant differences (filtered diff) ..."
|
||||
cat ${base_diff}
|
||||
cp ${base_diff} ${artifacts_path}/${base_diff}
|
||||
status=1
|
||||
# We create a .err file, because that's how we signal that there was a diff.
|
||||
cp ${base_diff} ${artifacts_path}/gitlab-${BASELINE_TEST}-${MACHINE_NAME}.err
|
||||
fi
|
||||
if [[ $status -eq 0 ]]; then
|
||||
|
||||
if [[ ! -s ${base_diff} ]]
|
||||
then
|
||||
echo "${BASELINE_TEST}: PASSED"
|
||||
true
|
||||
else
|
||||
echo "${BASELINE_TEST}: FAILED"
|
||||
false
|
||||
fi
|
||||
exit $status
|
||||
|
||||
@@ -11,86 +11,9 @@
|
||||
Version 4.5.3 (development)
|
||||
===========================
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added a new example code, Example 36/36p, to demonstrate the solution of
|
||||
the obstacle problem with a new finite element method.
|
||||
|
||||
- Added a new miniapp, Mesh Quality, for evaluating mesh quality using size,
|
||||
skewness, and aspect-ratio computed from the Jacobian of the transformation.
|
||||
|
||||
- Added a new miniapp for interface and boundary fitting to implicit domains
|
||||
defined using level-set functions. See miniapps/meshing/pmesh-fitting.cpp
|
||||
|
||||
- Added new Discontinuous Petrov-Galerkin (DPG) miniapp which includes serial
|
||||
and parallel examples for diffusion, convection-diffusion, acoustics and
|
||||
Maxwell equations. The miniapp includes new classes such as (Par)DPGWeakForm,
|
||||
(Par)ComplexDPGWeakForm and (Complex)BlockStaticCondensation. Three new
|
||||
integrators are added in support of DPG systems: TraceIntegrator,
|
||||
NormalTraceIntegrator and TangentTraceIntegrator.
|
||||
|
||||
- Added new SubMesh examples demonstrating source terms and boundary conditions
|
||||
transferred from SubMesh objects.
|
||||
|
||||
- Added a new H(div) solvers miniapp in miniapps/hdiv-linear-solver,
|
||||
demonstrating the use of a matrix-free saddle-point solver methodology,
|
||||
suitable for high-order discretizations and for GPU acceleration. Examples
|
||||
illustrating the solution of Darcy and grad-div problems are included.
|
||||
|
||||
- Added a random refinement option to the mesh-explorer miniapp to assist users
|
||||
in experimenting with nonconforming meshes.
|
||||
|
||||
- Moved the distance solver methods from miniapps/shifted to miniapps/common.
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
- Added new methods in the Mesh class to set and get attributes on NURBS patches
|
||||
and patch boundaries.
|
||||
|
||||
- Added HIP support to the SUNDIALS interface.
|
||||
|
||||
- TMOP improvement: added asymptotically-balanced compound metrics 90, 94, 328,
|
||||
338. Added the tmop-metric-magnitude tool for tracking how metrics change
|
||||
under geometric perturbations.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Face restriction operators for Nedelec and Raviart-Thomas finite element
|
||||
spaces are now supported through the ConformingFaceRestriction class.
|
||||
|
||||
- SubMesh and ParSubMesh have been extended to support the transfer of
|
||||
Nedelec and Raviart-Thomas finite element spaces.
|
||||
|
||||
- VectorFEBoundaryFluxLFIntegrator is now supported on device/GPU.
|
||||
|
||||
- Added support for p-refined meshes in FindPointsGSLIB.
|
||||
|
||||
Linear and nonlinear solvers
|
||||
----------------------------
|
||||
- Updated interface to MUMPS direct solver to support multiple right-hand
|
||||
sides, block low-rank compression, builds using 64-bit integers, and other
|
||||
improvements.
|
||||
|
||||
- Added an interface to the MKL Pardiso sparse direct solver developed by Intel.
|
||||
This interface provides a serial (OpenMP shared memory) version of Pardiso for
|
||||
use with SparseMatrix. This complements the existing parallel (MPI distributed
|
||||
memory) version already available through the CPardiso MFEM integration.
|
||||
|
||||
Integrations, testing and documentation
|
||||
---------------------------------------
|
||||
- Added an address sanitizer GitHub action for a serial build/test on Ubuntu,
|
||||
based on Clang/LLVM (https://clang.llvm.org/docs/AddressSanitizer.html).
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Improved lambda body debugging with the addition of mfem::forall functions.
|
||||
These functions can take the place of the MFEM_FORALL macros, which have been
|
||||
preserved for backwards compatibility.
|
||||
|
||||
- Reorganized files for bilinear form, linear form, and nonlinear form integrators
|
||||
in the fem/integ/ subdirectory.
|
||||
|
||||
|
||||
Version 4.5.2, released on March 23, 2023
|
||||
=========================================
|
||||
|
||||
|
||||
+3
-13
@@ -82,7 +82,7 @@ if (MFEM_USE_CONDUIT OR
|
||||
# * find_package(PETSc REQUIRED)
|
||||
set(XSDK_ENABLE_C ON)
|
||||
endif()
|
||||
if (MFEM_USE_STRUMPACK OR MFEM_USE_MUMPS)
|
||||
if (MFEM_USE_STRUMPACK)
|
||||
# Just needed to find the MPI_Fortran libraries to link with
|
||||
set(XSDK_ENABLE_Fortran ON)
|
||||
endif()
|
||||
@@ -317,9 +317,6 @@ if (MFEM_USE_SUNDIALS)
|
||||
if (MFEM_USE_CUDA)
|
||||
list(APPEND SUNDIALS_COMPONENTS NVector_Cuda)
|
||||
endif()
|
||||
if (MFEM_USE_HIP)
|
||||
list(APPEND SUNDIALS_COMPONENTS NVector_Hip)
|
||||
endif()
|
||||
find_package(SUNDIALS REQUIRED ${SUNDIALS_COMPONENTS})
|
||||
endif()
|
||||
|
||||
@@ -336,7 +333,6 @@ endif()
|
||||
if (MFEM_USE_MUMPS)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(MUMPS REQUIRED mumps_common pord)
|
||||
set(MFEM_MUMPS_VERSION ${MUMPS_VERSION})
|
||||
else()
|
||||
message(FATAL_ERROR " *** MUMPS requires that MPI be enabled.")
|
||||
endif()
|
||||
@@ -470,18 +466,12 @@ if (MFEM_USE_ADIOS2)
|
||||
find_package(ADIOS2 REQUIRED)
|
||||
endif()
|
||||
|
||||
# MKL CPardiso
|
||||
if (MFEM_USE_MKL_CPARDISO)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(MKL_CPARDISO REQUIRED MKL_SEQUENTIAL MKL_LP64 MKL_MPI_WRAPPER)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# MKL Pardiso
|
||||
if (MFEM_USE_MKL_PARDISO)
|
||||
find_package(MKL_PARDISO REQUIRED MKL_SEQUENTIAL MKL_LP64)
|
||||
endif()
|
||||
|
||||
# PARELAG
|
||||
if (MFEM_USE_PARELAG)
|
||||
find_package(PARELAG REQUIRED)
|
||||
@@ -531,8 +521,8 @@ find_package(Threads REQUIRED)
|
||||
set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
|
||||
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB
|
||||
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
|
||||
ADIOS2 CUBLAS CUSPARSE MKL_CPARDISO MKL_PARDISO AMGX CALIPER CODIPACK
|
||||
BENCHMARK PARELAG MPI_CXX HIP HIPSPARSE MOONOLITH BLITZ ALGOIM ENZYME)
|
||||
ADIOS2 CUBLAS CUSPARSE MKL_CPARDISO AMGX CALIPER CODIPACK BENCHMARK PARELAG
|
||||
MPI_CXX HIP HIPSPARSE MOONOLITH BLITZ ALGOIM ENZYME)
|
||||
|
||||
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
|
||||
set(TPL_LIBRARIES "")
|
||||
|
||||
+1
-3
@@ -121,7 +121,6 @@ The MFEM source code has the following structure:
|
||||
├── fem
|
||||
│ ├── ceed
|
||||
│ ├── fe
|
||||
│ ├── integ
|
||||
│ ├── lor
|
||||
│ ├── moonolith
|
||||
│ ├── qinterp
|
||||
@@ -137,7 +136,6 @@ The MFEM source code has the following structure:
|
||||
│ ├── common
|
||||
│ ├── electromagnetics
|
||||
│ ├── gslib
|
||||
│ ├── hdiv-linear-solver
|
||||
│ ├── hooke
|
||||
│ ├── meshing
|
||||
│ ├── mtop
|
||||
@@ -211,7 +209,7 @@ device/host memory manager.
|
||||
- The main device-relevant classes and sources are:
|
||||
+ [`Device`](https://docs.mfem.org/html/device_8hpp.html)
|
||||
+ [`MemoryManager`](https://docs.mfem.org/html/mem_manager_8hpp.html)
|
||||
+ the [`mfem::forall`](https://docs.mfem.org/html/forall_8hpp.html) function
|
||||
+ the [`MFEM_FORALL`](https://docs.mfem.org/html/forall_8hpp.html) macro
|
||||
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
|
||||
|
||||
#### Utilities, building and documentation
|
||||
|
||||
@@ -628,13 +628,9 @@ The specific libraries and their options are:
|
||||
both MPI and hypre.
|
||||
If MFEM_USE_CUDA is enabled, we expect that SUNDIALS is built with support
|
||||
for CUDA.
|
||||
If MFEM_USE_HIP is enabled, we expect that SUNDIALS is built with support
|
||||
for HIP.
|
||||
URL: http://computing.llnl.gov/projects/sundials/sundials-software
|
||||
URL: http://computation.llnl.gov/projects/sundials/sundials-software
|
||||
Options: SUNDIALS_OPT, SUNDIALS_LIB.
|
||||
Versions: SUNDIALS >= 5.0.0,
|
||||
SUNDIALS >= 5.4.0 for CUDA support, and
|
||||
SUNDIALS >= 5.7.0 for HIP support.
|
||||
Versions: SUNDIALS >= 5.0.0, SUNDIALS >= 5.4.0 for CUDA support.
|
||||
|
||||
- SuiteSparse (optional), used when MFEM_USE_SUITESPARSE = YES.
|
||||
URL: http://faculty.cse.tamu.edu/davis/suitesparse.html
|
||||
|
||||
@@ -55,8 +55,6 @@ set(MFEM_USE_SIMD @MFEM_USE_SIMD@)
|
||||
set(MFEM_USE_ADIOS2 @MFEM_USE_ADIOS2@)
|
||||
set(MFEM_USE_MOONOLITH @MFEM_USE_MOONOLITH@)
|
||||
set(MFEM_USE_CODIPACK @MFEM_USE_CODIPACK@)
|
||||
set(MFEM_USE_MKL_CPARDISO @MFEM_USE_MKL_CPARDISO@)
|
||||
set(MFEM_USE_MKL_PARDISO @MFEM_USE_MKL_PARDISO@)
|
||||
set(MFEM_USE_ADFORWARD @MFEM_USE_ADFORWARD@)
|
||||
set(MFEM_USE_CALIPER @MFEM_USE_CALIPER@)
|
||||
set(MFEM_USE_ALGOIM @MFEM_USE_ALGOIM@)
|
||||
|
||||
+58
-66
@@ -80,102 +80,97 @@
|
||||
// Internal MFEM option: enable group/batch allocation for some small objects.
|
||||
#cmakedefine MFEM_USE_MEMALLOC
|
||||
|
||||
// Which library functions to use in class StopWatch for measuring time.
|
||||
// For a list of the available options, see INSTALL.
|
||||
// If not defined, an option is selected automatically.
|
||||
#cmakedefine MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@
|
||||
|
||||
// Enable MFEM functionality based on the SUNDIALS libraries.
|
||||
#cmakedefine MFEM_USE_SUNDIALS
|
||||
|
||||
// Enable MFEM functionality based on the SuiteSparse library.
|
||||
#cmakedefine MFEM_USE_SUITESPARSE
|
||||
|
||||
// Enable MFEM functionality based on the SuperLU_DIST library.
|
||||
#cmakedefine MFEM_USE_SUPERLU
|
||||
#cmakedefine MFEM_USE_SUPERLU5
|
||||
|
||||
// Enable MFEM functionality based on the MUMPS library.
|
||||
#cmakedefine MFEM_USE_MUMPS
|
||||
#cmakedefine MFEM_MUMPS_VERSION @MFEM_MUMPS_VERSION@
|
||||
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
#cmakedefine MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable functionality based on the Ginkgo library.
|
||||
// Enable functionality based on the Ginkgo library
|
||||
#cmakedefine MFEM_USE_GINKGO
|
||||
|
||||
// Enable MFEM functionality based on the AmgX library.
|
||||
// Enable MFEM functionality based on the AmgX library
|
||||
#cmakedefine MFEM_USE_AMGX
|
||||
|
||||
// Enable secure socket streams based on the GNUTLS library.
|
||||
// Enable MFEM functionality based on the GnuTLS library
|
||||
#cmakedefine MFEM_USE_GNUTLS
|
||||
|
||||
// Enable Sidre support.
|
||||
#cmakedefine MFEM_USE_SIDRE
|
||||
|
||||
// Enable the use of SIMD in the high performance templated classes.
|
||||
#cmakedefine MFEM_USE_SIMD
|
||||
|
||||
// Enable FMS support.
|
||||
#cmakedefine MFEM_USE_FMS
|
||||
|
||||
// Enable Conduit support.
|
||||
#cmakedefine MFEM_USE_CONDUIT
|
||||
|
||||
// Enable functionality based on the NetCDF library (reading CUBIT files).
|
||||
#cmakedefine MFEM_USE_NETCDF
|
||||
|
||||
// Enable functionality based on the PETSc library.
|
||||
#cmakedefine MFEM_USE_PETSC
|
||||
|
||||
// Enable functionality based on the SLEPc library.
|
||||
#cmakedefine MFEM_USE_SLEPC
|
||||
|
||||
// Enable functionality based on the MPFR library.
|
||||
#cmakedefine MFEM_USE_MPFR
|
||||
|
||||
// Enable MFEM functionality based on the PUMI library.
|
||||
#cmakedefine MFEM_USE_PUMI
|
||||
|
||||
// Enable Moonolith-based general interpolation between finite element spaces.
|
||||
#cmakedefine MFEM_USE_MOONOLITH
|
||||
|
||||
// Enable MFEM functionality based on the HIOP library.
|
||||
#cmakedefine MFEM_USE_HIOP
|
||||
|
||||
// Enable MFEM functionality based on the GSLIB library.
|
||||
// Enable MFEM functionality based on the GSLIB library
|
||||
#cmakedefine MFEM_USE_GSLIB
|
||||
|
||||
// Build the NVIDIA GPU/CUDA-enabled version of the MFEM library.
|
||||
// Enable MFEM functionality based on the NetCDF library
|
||||
#cmakedefine MFEM_USE_NETCDF
|
||||
|
||||
// Enable MFEM functionality based on the PETSc library
|
||||
#cmakedefine MFEM_USE_PETSC
|
||||
|
||||
// Enable MFEM functionality based on the SLEPc library
|
||||
#cmakedefine MFEM_USE_SLEPC
|
||||
|
||||
// Enable MFEM functionality based on the Sidre library
|
||||
#cmakedefine MFEM_USE_SIDRE
|
||||
|
||||
// Enable the use of SIMD in the high performance templated classes
|
||||
#cmakedefine MFEM_USE_SIMD
|
||||
|
||||
// Enable MFEM functionality based on the FMS library
|
||||
#cmakedefine MFEM_USE_FMS
|
||||
|
||||
// Enable MFEM functionality based on Conduit
|
||||
#cmakedefine MFEM_USE_CONDUIT
|
||||
|
||||
// Enable MFEM functionality based on the PUMI library
|
||||
#cmakedefine MFEM_USE_PUMI
|
||||
|
||||
// Enable MFEM functionality based on the Moonolith library
|
||||
#cmakedefine MFEM_USE_MOONOLITH
|
||||
|
||||
// Enable MFEM functionality based on the HiOp library
|
||||
#cmakedefine MFEM_USE_HIOP
|
||||
|
||||
// Build the GPU/CUDA-enabled version of the MFEM library.
|
||||
// Requires a CUDA compiler (nvcc).
|
||||
#cmakedefine MFEM_USE_CUDA
|
||||
|
||||
// Build the AMD GPU/HIP-enabled version of the MFEM library.
|
||||
// Build the HIP-enabled version of the MFEM library.
|
||||
// Requires a HIP compiler (hipcc).
|
||||
#cmakedefine MFEM_USE_HIP
|
||||
|
||||
// Enable functionality based on the RAJA library.
|
||||
// Enable MFEM functionality based on the RAJA library
|
||||
#cmakedefine MFEM_USE_RAJA
|
||||
|
||||
// Enable functionality based on the OCCA library.
|
||||
// Enable MFEM functionality based on the OCCA library
|
||||
#cmakedefine MFEM_USE_OCCA
|
||||
|
||||
// Enable functionality based on the libCEED library.
|
||||
// Enable MFEM functionality based on the libCEED library
|
||||
#cmakedefine MFEM_USE_CEED
|
||||
|
||||
// Enable functionality based on the Caliper library.
|
||||
#cmakedefine MFEM_USE_CALIPER
|
||||
|
||||
// Enable functionality based on the Algoim library.
|
||||
#cmakedefine MFEM_USE_ALGOIM
|
||||
|
||||
// Enable functionality based on the Umpire library.
|
||||
// Enable MFEM functionality based on the Umpire library
|
||||
#cmakedefine MFEM_USE_UMPIRE
|
||||
|
||||
// Enable IO functionality based on the ADIOS2 library.
|
||||
// Enable MFEM functionality based on the ADIOS2 library
|
||||
#cmakedefine MFEM_USE_ADIOS2
|
||||
|
||||
// Enable MFEM functionality based on the Caliper library
|
||||
#cmakedefine MFEM_USE_CALIPER
|
||||
|
||||
// Enable MFEM functionality based on the Algoim library
|
||||
#cmakedefine MFEM_USE_ALGOIM
|
||||
|
||||
// Which library functions to use in class StopWatch for measuring time.
|
||||
// For a list of the available options, see INSTALL.
|
||||
// If not defined, an option is selected automatically.
|
||||
#define MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@
|
||||
|
||||
// Enable MFEM functionality based on the SUNDIALS libraries.
|
||||
#cmakedefine MFEM_USE_SUNDIALS
|
||||
|
||||
// Version of HYPRE used for building MFEM.
|
||||
#cmakedefine MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
|
||||
|
||||
@@ -186,16 +181,13 @@
|
||||
// Enable interface to the MKL CPardiso library.
|
||||
#cmakedefine MFEM_USE_MKL_CPARDISO
|
||||
|
||||
// Enable interface to the MKL Pardiso library.
|
||||
#cmakedefine MFEM_USE_MKL_PARDISO
|
||||
|
||||
// Use forward mode for automatic differentiation.
|
||||
// Use forward mode for automatic differentiation
|
||||
#cmakedefine MFEM_USE_ADFORWARD
|
||||
|
||||
// Enable the use of the CoDiPack library for AD.
|
||||
// Enable the use of the CoDiPack library for AD
|
||||
#cmakedefine MFEM_USE_CODIPACK
|
||||
|
||||
// Enable functionality based on the Google Benchmark library.
|
||||
// Enable MFEM functionality based on the Google Benchmark library.
|
||||
#cmakedefine MFEM_USE_BENCHMARK
|
||||
|
||||
// Enable Enzyme for AD
|
||||
|
||||
@@ -1,27 +0,0 @@
|
||||
# Copyright (c) 2010-2023, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Defines the following variables:
|
||||
# - MKL_PARDISO_FOUND
|
||||
# - MKL_PARDISO_LIBRARIES
|
||||
# - MKL_PARDISO_INCLUDE_DIRS
|
||||
|
||||
if(NOT MKL_LIBRARY_DIR)
|
||||
message(WARNING "Using default MKL library path. Double check the variable MKL_LIBRARY_DIR")
|
||||
set(MKL_LIBRARY_DIR "lib/intel64")
|
||||
endif()
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(MKL_PARDISO MKL_PARDISO
|
||||
MKL_PARDISO_DIR "include" mkl_pardiso.h ${MKL_LIBRARY_DIR} mkl_core
|
||||
"Paths to headers required by MKL Pardiso." "Libraries required by MKL PARDISO."
|
||||
ADD_COMPONENT MKL_LP64 "include" "" ${MKL_LIBRARY_DIR} mkl_intel_lp64
|
||||
ADD_COMPONENT MKL_SEQUENTIAL "include" "" ${MKL_LIBRARY_DIR} mkl_sequential)
|
||||
@@ -11,9 +11,8 @@
|
||||
|
||||
# Sets the following variables:
|
||||
# - MUMPS_FOUND
|
||||
# - MUMPS_LIBRARIES
|
||||
# - MUMPS_INCLUDE_DIRS
|
||||
# - MUMPS_VERSION
|
||||
# - MUMPS_LIBRARIES
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(MUMPS MUMPS MUMPS_DIR
|
||||
@@ -22,18 +21,3 @@ mfem_find_package(MUMPS MUMPS MUMPS_DIR
|
||||
"Libraries required by MUMPS."
|
||||
ADD_COMPONENT mumps_common "include" dmumps_c.h "lib" mumps_common
|
||||
ADD_COMPONENT pord "include" dmumps_c.h "lib" pord)
|
||||
|
||||
if (MUMPS_FOUND AND (NOT MUMPS_VERSION))
|
||||
try_run(MUMPS_VERSION_RUN_RESULT MUMPS_VERSION_COMPILE_RESULT
|
||||
${CMAKE_CURRENT_BINARY_DIR}/config
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/config/get_mumps_version.cpp
|
||||
CMAKE_FLAGS -DINCLUDE_DIRECTORIES:STRING=${MUMPS_INCLUDE_DIRS}
|
||||
RUN_OUTPUT_VARIABLE MUMPS_VERSION_OUTPUT)
|
||||
if ((MUMPS_VERSION_RUN_RESULT EQUAL 0) AND MUMPS_VERSION_OUTPUT)
|
||||
string(STRIP "${MUMPS_VERSION_OUTPUT}" MUMPS_VERSION)
|
||||
set(MUMPS_VERSION ${MUMPS_VERSION} CACHE STRING "MUMPS version." FORCE)
|
||||
message(STATUS "Found MUMPS version ${MUMPS_VERSION}")
|
||||
else()
|
||||
message(FATAL_ERROR "Unable to determine MUMPS version.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -22,8 +22,8 @@ mfem_find_package(SUNDIALS SUNDIALS SUNDIALS_DIR
|
||||
"include" nvector/nvector_serial.h "lib" sundials_nvecserial
|
||||
ADD_COMPONENT NVector_Cuda
|
||||
"include" nvector/nvector_cuda.h "lib" sundials_nveccuda
|
||||
ADD_COMPONENT NVector_Hip
|
||||
"include" nvector/nvector_hip.h "lib" sundials_nvechip
|
||||
ADD_COMPONENT NVector_ParHyp
|
||||
"include" nvector/nvector_parhyp.h "lib" sundials_nvecparhyp
|
||||
ADD_COMPONENT NVector_Parallel
|
||||
"include" nvector/nvector_parallel.h "lib" sundials_nvecparallel
|
||||
ADD_COMPONENT NVector_MPIPlusX
|
||||
|
||||
+16
-19
@@ -30,10 +30,10 @@
|
||||
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
|
||||
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
|
||||
|
||||
// The absolute path of the MFEM source prefix.
|
||||
// The absolute path of the MFEM source prefix
|
||||
// #define MFEM_SOURCE_DIR "@MFEM_SOURCE_DIR@"
|
||||
|
||||
// The absolute path of the MFEM installation prefix.
|
||||
// The absolute path of the MFEM installation prefix
|
||||
// #define MFEM_INSTALL_DIR "@MFEM_INSTALL_DIR@"
|
||||
|
||||
// Description of the git commit used to build MFEM.
|
||||
@@ -91,7 +91,7 @@
|
||||
// Enable MFEM functionality based on the SuiteSparse library.
|
||||
// #define MFEM_USE_SUITESPARSE
|
||||
|
||||
// Enable MFEM functionality based on the SuperLU_DIST library.
|
||||
// Enable MFEM functionality based on the SuperLU library.
|
||||
// #define MFEM_USE_SUPERLU
|
||||
// #define MFEM_USE_SUPERLU5
|
||||
|
||||
@@ -102,40 +102,40 @@
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
// #define MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable MFEM features based on the Ginkgo library.
|
||||
// Enable MFEM features based on the Ginkgo library
|
||||
// #define MFEM_USE_GINKGO
|
||||
|
||||
// Enable MFEM functionality based on the AmgX library.
|
||||
// #define MFEM_USE_AMGX
|
||||
|
||||
// Enable secure socket streams based on the GNUTLS library.
|
||||
// Enable secure socket streams based on the GNUTLS library
|
||||
// #define MFEM_USE_GNUTLS
|
||||
|
||||
// Enable Sidre support.
|
||||
// Enable Sidre support
|
||||
// #define MFEM_USE_SIDRE
|
||||
|
||||
// Enable the use of SIMD in the high performance templated classes.
|
||||
// Enable the use of SIMD in the high performance templated classes
|
||||
// #define MFEM_USE_SIMD
|
||||
|
||||
// Enable FMS support.
|
||||
// Enable FMS support
|
||||
// #define MFEM_USE_FMS
|
||||
|
||||
// Enable Conduit support.
|
||||
// Enable Conduit support
|
||||
// #define MFEM_USE_CONDUIT
|
||||
|
||||
// Enable functionality based on the NetCDF library (reading CUBIT files).
|
||||
// Enable functionality based on the NetCDF library (reading CUBIT files)
|
||||
// #define MFEM_USE_NETCDF
|
||||
|
||||
// Enable functionality based on the PETSc library.
|
||||
// Enable functionality based on the PETSc library
|
||||
// #define MFEM_USE_PETSC
|
||||
|
||||
// Enable functionality based on the SLEPc library.
|
||||
// Enable functionality based on the SLEPc library
|
||||
// #define MFEM_USE_SLEPC
|
||||
|
||||
// Enable functionality based on the MPFR library.
|
||||
// #define MFEM_USE_MPFR
|
||||
|
||||
// Enable MFEM functionality based on the PUMI library.
|
||||
// Enable MFEM functionality based on the PUMI library
|
||||
// #define MFEM_USE_PUMI
|
||||
|
||||
// Enable Moonolith-based general interpolation between finite element spaces.
|
||||
@@ -144,7 +144,7 @@
|
||||
// Enable MFEM functionality based on the HIOP library.
|
||||
// #define MFEM_USE_HIOP
|
||||
|
||||
// Enable MFEM functionality based on the GSLIB library.
|
||||
// Enable MFEM functionality based on the GSLIB library
|
||||
// #define MFEM_USE_GSLIB
|
||||
|
||||
// Build the NVIDIA GPU/CUDA-enabled version of the MFEM library.
|
||||
@@ -186,13 +186,10 @@
|
||||
// Enable interface to the MKL CPardiso library.
|
||||
// #define MFEM_USE_MKL_CPARDISO
|
||||
|
||||
// Enable interface to the MKL Pardiso library.
|
||||
// #define MFEM_USE_MKL_PARDISO
|
||||
|
||||
// Use forward mode for automatic differentiation.
|
||||
// Use forward mode for automatic differentiation
|
||||
// #define MFEM_USE_ADFORWARD
|
||||
|
||||
// Enable the use of the CoDiPack library for AD.
|
||||
// Enable the use of the CoDiPack library for AD
|
||||
// #define MFEM_USE_CODIPACK
|
||||
|
||||
// Enable functionality based on the Google Benchmark library.
|
||||
|
||||
@@ -57,7 +57,6 @@ MFEM_USE_UMPIRE = @MFEM_USE_UMPIRE@
|
||||
MFEM_USE_SIMD = @MFEM_USE_SIMD@
|
||||
MFEM_USE_ADIOS2 = @MFEM_USE_ADIOS2@
|
||||
MFEM_USE_MKL_CPARDISO = @MFEM_USE_MKL_CPARDISO@
|
||||
MFEM_USE_MKL_PARDISO = @MFEM_USE_MKL_PARDISO@
|
||||
MFEM_USE_MOONOLITH = @MFEM_USE_MOONOLITH@
|
||||
MFEM_USE_ADFORWARD = @MFEM_USE_ADFORWARD@
|
||||
MFEM_USE_CODIPACK = @MFEM_USE_CODIPACK@
|
||||
|
||||
+5
-10
@@ -60,7 +60,6 @@ option(MFEM_USE_ADIOS2 "Enable ADIOS2" OFF)
|
||||
option(MFEM_USE_CALIPER "Enable Caliper support" OFF)
|
||||
option(MFEM_USE_ALGOIM "Enable Algoim support" OFF)
|
||||
option(MFEM_USE_MKL_CPARDISO "Enable MKL CPardiso" OFF)
|
||||
option(MFEM_USE_MKL_PARDISO "Enable MKL Pardiso" OFF)
|
||||
option(MFEM_USE_ADFORWARD "Enable forward mode for AD" OFF)
|
||||
option(MFEM_USE_CODIPACK "Enable automatic differentiation (AD) using CoDiPack" OFF)
|
||||
option(MFEM_USE_BENCHMARK "Enable Google Benchmark" OFF)
|
||||
@@ -135,18 +134,16 @@ set(ParMETIS_DIR "${MFEM_DIR}/../parmetis-4.0.3" CACHE PATH
|
||||
set(ParMETIS_REQUIRED_PACKAGES "METIS" CACHE STRING
|
||||
"Additional packages required by ParMETIS.")
|
||||
|
||||
set(SuperLUDist_DIR "${MFEM_DIR}/../SuperLU_DIST_8.1.2" CACHE PATH
|
||||
set(SuperLUDist_DIR "${MFEM_DIR}/../SuperLU_DIST_6.3.1" CACHE PATH
|
||||
"Path to the SuperLU_DIST library.")
|
||||
# SuperLU_DIST may also depend on "OpenMP", depending on how it was compiled.
|
||||
set(SuperLUDist_REQUIRED_PACKAGES "MPI" "ParMETIS" "METIS"
|
||||
"LAPACK" "BLAS" CACHE STRING
|
||||
set(SuperLUDist_REQUIRED_PACKAGES "MPI" "BLAS" "ParMETIS" CACHE STRING
|
||||
"Additional packages required by SuperLU_DIST.")
|
||||
|
||||
set(MUMPS_DIR "${MFEM_DIR}/../MUMPS_5.5.0" CACHE PATH
|
||||
set(MUMPS_DIR "${MFEM_DIR}/../MUMPS_5.2.0" CACHE PATH
|
||||
"Path to the MUMPS library.")
|
||||
# MUMPS may also depend on "OpenMP", depending on how it was compiled.
|
||||
set(MUMPS_REQUIRED_PACKAGES "MPI" "MPI_Fortran" "ParMETIS" "METIS"
|
||||
"ScaLAPACK" "LAPACK" "BLAS" CACHE STRING
|
||||
# Packages required by MUMPS, depending on how it was compiled.
|
||||
set(MUMPS_REQUIRED_PACKAGES "MPI" "BLAS" "METIS" "ScaLAPACK" CACHE STRING
|
||||
"Additional packages required by MUMPS.")
|
||||
# If the MPI package does not find all required Fortran libraries:
|
||||
# set(MUMPS_REQUIRED_LIBRARIES "gfortran" "mpi_mpifh" CACHE STRING
|
||||
@@ -229,8 +226,6 @@ set(MKL_CPARDISO_DIR "" CACHE STRING "MKL installation path.")
|
||||
set(MKL_MPI_WRAPPER_LIB "mkl_blacs_mpich_lp64" CACHE STRING "MKL MPI wrapper library")
|
||||
set(MKL_LIBRARY_DIR "" CACHE STRING "Custom library subdirectory")
|
||||
|
||||
set(MKL_PARDISO_DIR "" CACHE STRING "MKL installation path.")
|
||||
|
||||
set(OCCA_DIR "${MFEM_DIR}/../occa" CACHE PATH "Path to OCCA")
|
||||
set(RAJA_DIR "${MFEM_DIR}/../raja" CACHE PATH "Path to RAJA")
|
||||
set(CEED_DIR "${MFEM_DIR}/../libCEED" CACHE PATH "Path to libCEED")
|
||||
|
||||
+4
-15
@@ -160,7 +160,6 @@ MFEM_USE_UMPIRE = NO
|
||||
MFEM_USE_SIMD = NO
|
||||
MFEM_USE_ADIOS2 = NO
|
||||
MFEM_USE_MKL_CPARDISO = NO
|
||||
MFEM_USE_MKL_PARDISO = NO
|
||||
MFEM_USE_MOONOLITH = NO
|
||||
MFEM_USE_ADFORWARD = NO
|
||||
MFEM_USE_CODIPACK = NO
|
||||
@@ -267,9 +266,6 @@ endif
|
||||
ifeq ($(MFEM_USE_CUDA),YES)
|
||||
SUNDIALS_LIB += -lsundials_nveccuda
|
||||
endif
|
||||
ifeq ($(MFEM_USE_HIP),YES)
|
||||
SUNDIALS_LIB += -lsundials_nvechip
|
||||
endif
|
||||
# If SUNDIALS was built with KLU:
|
||||
# MFEM_USE_SUITESPARSE = YES
|
||||
|
||||
@@ -288,10 +284,10 @@ ifeq ($(MFEM_USE_SUPERLU5),YES)
|
||||
SUPERLU_LIB = $(XLINKER)-rpath,$(SUPERLU_DIR)/lib -L$(SUPERLU_DIR)/lib\
|
||||
-lsuperlu_dist_5.1.0
|
||||
else
|
||||
SUPERLU_DIR = @MFEM_DIR@/../SuperLU_DIST_8.1.2
|
||||
SUPERLU_DIR = @MFEM_DIR@/../SuperLU_DIST_6.3.1
|
||||
SUPERLU_OPT = -I$(SUPERLU_DIR)/include
|
||||
SUPERLU_LIB = $(XLINKER)-rpath,$(SUPERLU_DIR)/lib64 -L$(SUPERLU_DIR)/lib64\
|
||||
-lsuperlu_dist $(LAPACK_LIB)
|
||||
-lsuperlu_dist -lblas
|
||||
endif
|
||||
|
||||
# SCOTCH library configuration (required by STRUMPACK <= v2.1.0, optional in
|
||||
@@ -315,7 +311,7 @@ MPI_FORTRAN_LIB = -lmpifort
|
||||
# MPI_FORTRAN_LIB += -lgfortran
|
||||
|
||||
# MUMPS library configuration
|
||||
MUMPS_DIR = @MFEM_DIR@/../MUMPS_5.5.0
|
||||
MUMPS_DIR = @MFEM_DIR@/../MUMPS_5.2.0
|
||||
MUMPS_OPT = -I$(MUMPS_DIR)/include
|
||||
MUMPS_LIB = $(XLINKER)-rpath,$(MUMPS_DIR)/lib -L$(MUMPS_DIR)/lib -ldmumps\
|
||||
-lmumps_common -lpord $(SCALAPACK_LIB) $(LAPACK_LIB) $(MPI_FORTRAN_LIB)
|
||||
@@ -488,6 +484,7 @@ ifdef GOTCHA_DIR
|
||||
CALIPER_LIB += $(XLINKER)-rpath,$(GOTCHA_DIR)/lib64 $(XLINKER)-rpath,$(GOTCHA_DIR)/lib -L$(GOTCHA_DIR)/lib64 -L$(GOTCHA_DIR)/lib -lgotcha
|
||||
endif
|
||||
|
||||
|
||||
# BLITZ library configuration
|
||||
BLITZ_DIR = @MFEM_DIR@/../blitz
|
||||
BLITZ_OPT = -I$(BLITZ_DIR)/include
|
||||
@@ -542,14 +539,6 @@ MKL_CPARDISO_LIB = $(XLINKER)-rpath,$(MKL_CPARDISO_DIR)/$(MKL_LIBRARY_SUBDIR)\
|
||||
-L$(MKL_CPARDISO_DIR)/$(MKL_LIBRARY_SUBDIR) -l$(MKL_MPI_WRAPPER)\
|
||||
-lmkl_intel_lp64 -lmkl_sequential -lmkl_core
|
||||
|
||||
# MKL Pardiso library configuration
|
||||
MKL_PARDISO_DIR ?=
|
||||
MKL_LIBRARY_SUBDIR ?= lib
|
||||
MKL_PARDISO_OPT = -I$(MKL_PARDISO_DIR)/include
|
||||
MKL_PARDISO_LIB = $(XLINKER)-rpath,$(MKL_PARDISO_DIR)/$(MKL_LIBRARY_SUBDIR)\
|
||||
-L$(MKL_PARDISO_DIR)/$(MKL_LIBRARY_SUBDIR)\
|
||||
-lmkl_intel_lp64 -lmkl_sequential -lmkl_core
|
||||
|
||||
# PARELAG library configuration
|
||||
PARELAG_DIR = @MFEM_DIR@/../parelag
|
||||
PARELAG_OPT = -I$(PARELAG_DIR)/src -I$(PARELAG_DIR)/build/src
|
||||
|
||||
@@ -138,7 +138,7 @@ groups_parallel=(
|
||||
'"meshing"
|
||||
"Meshing miniapps:"
|
||||
"miniapps/meshing"
|
||||
"pmesh-optimizer.cpp pmesh-fitting.cpp pminimal-surface.cpp"'
|
||||
"pmesh-optimizer.cpp pminimal-surface.cpp"'
|
||||
'"electromagnetics"
|
||||
"Electromagnetics miniapps:"
|
||||
"miniapps/electromagnetics"
|
||||
@@ -227,7 +227,7 @@ groups_all=(
|
||||
"Meshing miniapps:"
|
||||
"miniapps/meshing"
|
||||
"mobius-strip.cpp klein-bottle.cpp extruder.cpp toroid.cpp
|
||||
{,p}mesh-optimizer.cpp pmesh-fitting.cpp {,p}minimal-surface.cpp"'
|
||||
{,p}mesh-optimizer.cpp {,p}minimal-surface.cpp"'
|
||||
'"electromagnetics"
|
||||
"Electromagnetics miniapps:"
|
||||
"miniapps/electromagnetics"
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
MFEM mesh v1.0
|
||||
#
|
||||
# MFEM Geometry Types (see mesh/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
#
|
||||
|
||||
dimension
|
||||
2
|
||||
|
||||
elements
|
||||
6
|
||||
1 3 0 1 4 3
|
||||
1 3 2 3 6 5
|
||||
1 2 3 4 8
|
||||
1 2 4 7 8
|
||||
1 2 7 6 8
|
||||
1 2 6 3 8
|
||||
|
||||
boundary
|
||||
8
|
||||
1 1 0 1
|
||||
2 1 1 4
|
||||
3 1 4 7
|
||||
4 1 7 6
|
||||
5 1 6 5
|
||||
6 1 5 2
|
||||
7 1 2 3
|
||||
8 1 3 0
|
||||
|
||||
vertices
|
||||
9
|
||||
2
|
||||
0.5 0
|
||||
1 0
|
||||
0 0.5
|
||||
0.5 0.5
|
||||
1 0.5
|
||||
0 1
|
||||
0.5 1
|
||||
1 1
|
||||
0.75 0.75
|
||||
@@ -1,44 +0,0 @@
|
||||
MFEM mesh v1.0
|
||||
#
|
||||
# MFEM Geometry Types (see mesh/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
#
|
||||
|
||||
dimension
|
||||
2
|
||||
|
||||
elements
|
||||
3
|
||||
1 3 0 1 4 3
|
||||
1 3 2 3 6 5
|
||||
1 3 3 4 7 6
|
||||
|
||||
boundary
|
||||
8
|
||||
1 1 0 1
|
||||
2 1 1 4
|
||||
3 1 4 7
|
||||
4 1 7 6
|
||||
5 1 6 5
|
||||
6 1 5 2
|
||||
7 1 2 3
|
||||
8 1 3 0
|
||||
|
||||
vertices
|
||||
8
|
||||
2
|
||||
0.5 0
|
||||
1 0
|
||||
0 0.5
|
||||
0.5 0.5
|
||||
1 0.5
|
||||
0 1
|
||||
0.5 1
|
||||
1 1
|
||||
@@ -1,322 +0,0 @@
|
||||
MFEM mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see mesh/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
# PYRAMID = 7
|
||||
#
|
||||
|
||||
dimension
|
||||
2
|
||||
|
||||
elements
|
||||
26
|
||||
1 2 1 18 0
|
||||
1 3 1 3 19 18
|
||||
2 3 3 6 20 19
|
||||
1 3 6 9 21 20
|
||||
2 3 9 12 22 21
|
||||
1 3 12 15 23 22
|
||||
2 2 23 15 24
|
||||
1 2 1 4 3
|
||||
2 3 4 7 6 3
|
||||
1 3 7 10 9 6
|
||||
2 3 10 13 12 9
|
||||
1 3 13 16 15 12
|
||||
2 3 16 25 24 15
|
||||
1 3 2 5 4 1
|
||||
1 3 5 8 7 4
|
||||
1 3 8 11 10 7
|
||||
1 3 11 14 13 10
|
||||
1 3 14 17 16 13
|
||||
1 2 25 16 17
|
||||
1 3 18 19 27 26
|
||||
2 3 19 20 28 27
|
||||
1 3 20 21 29 28
|
||||
2 3 21 22 30 29
|
||||
1 3 22 23 31 30
|
||||
2 3 23 24 32 31
|
||||
1 3 24 25 33 32
|
||||
|
||||
boundary
|
||||
18
|
||||
1 1 28 27
|
||||
2 1 30 29
|
||||
3 1 32 31
|
||||
4 1 0 1
|
||||
4 1 1 2
|
||||
4 1 2 5
|
||||
4 1 5 8
|
||||
4 1 8 11
|
||||
4 1 11 14
|
||||
4 1 14 17
|
||||
4 1 17 25
|
||||
4 1 25 33
|
||||
4 1 33 32
|
||||
4 1 31 30
|
||||
4 1 29 28
|
||||
4 1 27 26
|
||||
4 1 26 18
|
||||
4 1 18 0
|
||||
|
||||
vertices
|
||||
34
|
||||
|
||||
nodes
|
||||
FiniteElementSpace
|
||||
FiniteElementCollection: H1_2D_P3
|
||||
VDim: 2
|
||||
Ordering: 1
|
||||
|
||||
0 0
|
||||
0.53125 0
|
||||
1 0
|
||||
0.53125 0.09375
|
||||
0.5625 0.09375
|
||||
1 0.09375
|
||||
0.53125 0.21875
|
||||
0.6875 0.21875
|
||||
1 0.1875
|
||||
0.53125 0.25
|
||||
0.71875 0.25
|
||||
1 0.25
|
||||
0.53125 0.375
|
||||
0.84375 0.375
|
||||
1 0.34375
|
||||
0.53125 0.40625
|
||||
0.875 0.40625
|
||||
1 0.40625
|
||||
0 0.53125
|
||||
0.09375 0.53125
|
||||
0.21875 0.53125
|
||||
0.25 0.53125
|
||||
0.375 0.53125
|
||||
0.40625 0.53125
|
||||
0.53125 0.53125
|
||||
1 0.53125
|
||||
0 1
|
||||
0.09375 1
|
||||
0.21875 1
|
||||
0.25 1
|
||||
0.375 1
|
||||
0.40625 1
|
||||
0.53125 1
|
||||
1 1
|
||||
|
||||
0.33175106835972 0.094168845750364
|
||||
0.094168845750364 0.33175106835972
|
||||
-5.1759634627347e-17 0.14683388869532
|
||||
6.5255471622478e-17 0.38441611130468
|
||||
0.14683388869532 6.0713766400335e-17
|
||||
0.38441611130468 8.2458945395444e-19
|
||||
0.53125 0.025911862710939
|
||||
0.53125 0.067838137289061
|
||||
0.34721731046049 0.13433915461926
|
||||
0.13433915461926 0.34721731046049
|
||||
0.025911862710939 0.53125
|
||||
0.067838137289061 0.53125
|
||||
0.53125 0.12829915028125
|
||||
0.53125 0.18420084971875
|
||||
0.39979807890035 0.24774225329947
|
||||
0.24774225329947 0.39979807890035
|
||||
0.12829915028125 0.53125
|
||||
0.18420084971875 0.53125
|
||||
0.53125 0.22738728757031
|
||||
0.53125 0.24136271242969
|
||||
0.41294327101031 0.27609302796952
|
||||
0.27609302796952 0.41294327101031
|
||||
0.22738728757031 0.53125
|
||||
0.24136271242969 0.53125
|
||||
0.53125 0.28454915028125
|
||||
0.53125 0.34045084971875
|
||||
0.46552403945017 0.38949612664974
|
||||
0.38949612664974 0.46552403945017
|
||||
0.28454915028125 0.53125
|
||||
0.34045084971875 0.53125
|
||||
0.53125 0.38363728757031
|
||||
0.53125 0.39761271242969
|
||||
0.47866923156014 0.41784690131979
|
||||
0.41784690131979 0.47866923156014
|
||||
0.38363728757031 0.53125
|
||||
0.39761271242969 0.53125
|
||||
0.53125 0.44079915028125
|
||||
0.53125 0.49670084971875
|
||||
0.44079915028125 0.53125
|
||||
0.49670084971875 0.53125
|
||||
0.53988728757031 0.025911862710939
|
||||
0.55386271242969 0.067838137289061
|
||||
0.53988728757031 0.09375
|
||||
0.55386271242969 0.09375
|
||||
0.59704915028125 0.12829915028125
|
||||
0.65295084971875 0.18420084971875
|
||||
0.57443643785157 0.21875
|
||||
0.64431356214843 0.21875
|
||||
0.69613728757031 0.22738728757031
|
||||
0.71011271242969 0.24136271242969
|
||||
0.58307372542188 0.25
|
||||
0.66692627457812 0.25
|
||||
0.75329915028125 0.28454915028125
|
||||
0.80920084971875 0.34045084971875
|
||||
0.61762287570313 0.375
|
||||
0.75737712429687 0.375
|
||||
0.85238728757031 0.38363728757031
|
||||
0.86636271242969 0.39761271242969
|
||||
0.62626016327344 0.40625
|
||||
0.77998983672656 0.40625
|
||||
0.90954915028125 0.44079915028125
|
||||
0.96545084971875 0.49670084971875
|
||||
0.6608093135547 0.53125
|
||||
0.8704406864453 0.53125
|
||||
1 0.025911862710939
|
||||
1 0.067838137289061
|
||||
0.68342202598438 0.09375
|
||||
0.87907797401562 0.09375
|
||||
0.6608093135547 0
|
||||
0.8704406864453 0
|
||||
1 0.11966186271094
|
||||
1 0.16158813728906
|
||||
0.77387287570313 0.21011271242969
|
||||
0.91362712429687 0.19613728757031
|
||||
1 0.20477457514063
|
||||
1 0.23272542485937
|
||||
0.79648558813282 0.25
|
||||
0.92226441186718 0.25
|
||||
1 0.27591186271094
|
||||
1 0.31783813728906
|
||||
0.88693643785157 0.36636271242969
|
||||
0.95681356214843 0.35238728757031
|
||||
1 0.36102457514063
|
||||
1 0.38897542485937
|
||||
0.90954915028125 0.40625
|
||||
0.96545084971875 0.40625
|
||||
1 0.44079915028125
|
||||
1 0.49670084971875
|
||||
0.09375 0.6608093135547
|
||||
0.09375 0.8704406864453
|
||||
0.025911862710939 1
|
||||
0.067838137289061 1
|
||||
0 0.6608093135547
|
||||
0 0.8704406864453
|
||||
0.21875 0.6608093135547
|
||||
0.21875 0.8704406864453
|
||||
0.12829915028125 1
|
||||
0.18420084971875 1
|
||||
0.25 0.6608093135547
|
||||
0.25 0.8704406864453
|
||||
0.22738728757031 1
|
||||
0.24136271242969 1
|
||||
0.375 0.6608093135547
|
||||
0.375 0.8704406864453
|
||||
0.28454915028125 1
|
||||
0.34045084971875 1
|
||||
0.40625 0.6608093135547
|
||||
0.40625 0.8704406864453
|
||||
0.38363728757031 1
|
||||
0.39761271242969 1
|
||||
0.53125 0.6608093135547
|
||||
0.53125 0.8704406864453
|
||||
0.44079915028125 1
|
||||
0.49670084971875 1
|
||||
1 0.6608093135547
|
||||
1 0.8704406864453
|
||||
0.6608093135547 1
|
||||
0.8704406864453 1
|
||||
0.14782497614169 0.14782497614169
|
||||
0.3364183509774 0.10629113008478
|
||||
0.3439701728879 0.12590539815918
|
||||
0.10629113008478 0.3364183509774
|
||||
0.12590539815918 0.3439701728879
|
||||
0.36175027742635 0.16568300020856
|
||||
0.38526511193449 0.21639840771017
|
||||
0.16568300020856 0.36175027742635
|
||||
0.21639840771017 0.38526511193449
|
||||
0.40343132064181 0.2555782146968
|
||||
0.40931002926885 0.2682570665722
|
||||
0.2555782146968 0.40343132064181
|
||||
0.2682570665722 0.40931002926885
|
||||
0.42747623797617 0.30743687355882
|
||||
0.45099107248431 0.35815228106044
|
||||
0.30743687355882 0.42747623797617
|
||||
0.35815228106044 0.45099107248431
|
||||
0.46915728119164 0.39733208804706
|
||||
0.47503598981867 0.41001093992246
|
||||
0.39733208804706 0.46915728119164
|
||||
0.41001093992246 0.47503598981867
|
||||
0.47232443490112 0.47232443490112
|
||||
0.54166666666667 0.0625
|
||||
0.57886271242969 0.12829915028125
|
||||
0.61931356214843 0.18420084971875
|
||||
0.54943643785157 0.12829915028125
|
||||
0.56488728757031 0.18420084971875
|
||||
0.65056356214843 0.22738728757031
|
||||
0.66067627457812 0.24136271242969
|
||||
0.57682372542188 0.22738728757031
|
||||
0.58068643785157 0.24136271242969
|
||||
0.69192627457812 0.28454915028125
|
||||
0.73237712429687 0.34045084971875
|
||||
0.59262287570313 0.28454915028125
|
||||
0.60807372542188 0.34045084971875
|
||||
0.76362712429687 0.38363728757031
|
||||
0.77373983672656 0.39761271242969
|
||||
0.62001016327344 0.38363728757031
|
||||
0.62387287570313 0.39761271242969
|
||||
0.80498983672656 0.44079915028125
|
||||
0.8454406864453 0.49670084971875
|
||||
0.6358093135547 0.44079915028125
|
||||
0.65126016327344 0.49670084971875
|
||||
0.87282797401562 0.025911862710939
|
||||
0.8766906864453 0.067838137289061
|
||||
0.6670593135547 0.025911862710939
|
||||
0.67717202598438 0.067838137289061
|
||||
0.88862712429687 0.12204915028125
|
||||
0.90407797401562 0.16783813728906
|
||||
0.70842202598438 0.12591186271094
|
||||
0.74887287570313 0.17795084971875
|
||||
0.91601441186718 0.21102457514063
|
||||
0.91987712429687 0.23511271242969
|
||||
0.78012287570313 0.22113728757031
|
||||
0.79023558813282 0.23897542485937
|
||||
0.93181356214843 0.27829915028125
|
||||
0.94726441186718 0.32408813728906
|
||||
0.82148558813282 0.28216186271094
|
||||
0.86193643785157 0.33420084971875
|
||||
0.95920084971875 0.36727457514063
|
||||
0.96306356214843 0.39136271242969
|
||||
0.89318643785157 0.37738728757031
|
||||
0.90329915028125 0.39522542485937
|
||||
0.95833333333333 0.44791666666667
|
||||
0.025911862710939 0.6608093135547
|
||||
0.067838137289061 0.6608093135547
|
||||
0.025911862710939 0.8704406864453
|
||||
0.067838137289061 0.8704406864453
|
||||
0.12829915028125 0.6608093135547
|
||||
0.18420084971875 0.6608093135547
|
||||
0.12829915028125 0.8704406864453
|
||||
0.18420084971875 0.8704406864453
|
||||
0.22738728757031 0.6608093135547
|
||||
0.24136271242969 0.6608093135547
|
||||
0.22738728757031 0.8704406864453
|
||||
0.24136271242969 0.8704406864453
|
||||
0.28454915028125 0.6608093135547
|
||||
0.34045084971875 0.6608093135547
|
||||
0.28454915028125 0.8704406864453
|
||||
0.34045084971875 0.8704406864453
|
||||
0.38363728757031 0.6608093135547
|
||||
0.39761271242969 0.6608093135547
|
||||
0.38363728757031 0.8704406864453
|
||||
0.39761271242969 0.8704406864453
|
||||
0.44079915028125 0.6608093135547
|
||||
0.49670084971875 0.6608093135547
|
||||
0.44079915028125 0.8704406864453
|
||||
0.49670084971875 0.8704406864453
|
||||
0.6608093135547 0.6608093135547
|
||||
0.8704406864453 0.6608093135547
|
||||
0.6608093135547 0.8704406864453
|
||||
0.8704406864453 0.8704406864453
|
||||
@@ -1,46 +0,0 @@
|
||||
MFEM NURBS mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see mesh/geom.hpp):
|
||||
#
|
||||
# SEGMENT = 1
|
||||
# SQUARE = 3
|
||||
# CUBE = 5
|
||||
#
|
||||
|
||||
dimension
|
||||
1
|
||||
|
||||
elements
|
||||
1
|
||||
1 1 0 1
|
||||
|
||||
boundary
|
||||
2
|
||||
1 0 0
|
||||
2 0 1
|
||||
|
||||
edges
|
||||
1
|
||||
0 0 1
|
||||
|
||||
vertices
|
||||
2
|
||||
|
||||
knotvectors
|
||||
1
|
||||
1 2 0 0 1 1
|
||||
|
||||
weights
|
||||
1
|
||||
1
|
||||
|
||||
FiniteElementSpace
|
||||
FiniteElementCollection: NURBS1
|
||||
VDim: 1
|
||||
Ordering: 1
|
||||
|
||||
0
|
||||
1
|
||||
|
||||
|
||||
@@ -795,7 +795,6 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/miniapps/common \
|
||||
@MFEM_SOURCE_DIR@/miniapps/electromagnetics \
|
||||
@MFEM_SOURCE_DIR@/miniapps/gslib \
|
||||
@MFEM_SOURCE_DIR@/miniapps/hdiv-linear-solver \
|
||||
@MFEM_SOURCE_DIR@/miniapps/hooke \
|
||||
@MFEM_SOURCE_DIR@/miniapps/hooke/kernels \
|
||||
@MFEM_SOURCE_DIR@/miniapps/hooke/materials \
|
||||
@@ -811,10 +810,7 @@ INPUT = @MFEM_SOURCE_DIR@/doc/CodeDocumentation.dox \
|
||||
@MFEM_SOURCE_DIR@/miniapps/shifted \
|
||||
@MFEM_SOURCE_DIR@/miniapps/solvers \
|
||||
@MFEM_SOURCE_DIR@/miniapps/tools \
|
||||
@MFEM_SOURCE_DIR@/miniapps/toys \
|
||||
@MFEM_SOURCE_DIR@/miniapps/spde \
|
||||
@MFEM_SOURCE_DIR@/miniapps/dpg \
|
||||
@MFEM_SOURCE_DIR@/miniapps/dpg/util
|
||||
@MFEM_SOURCE_DIR@/miniapps/toys
|
||||
|
||||
# This tag can be used to specify the character encoding of the source files
|
||||
# that doxygen parses. Internally doxygen uses the UTF-8 encoding. Doxygen uses
|
||||
|
||||
@@ -39,7 +39,7 @@ namespace mfem {
|
||||
* - Device
|
||||
* - Memory
|
||||
* - MemoryManager
|
||||
* - mfem::forall functions in forall.hpp
|
||||
* - MFEM_FORALL macro in forall.hpp
|
||||
*
|
||||
* <H3>Example codes</H3>
|
||||
* - <a class="el" href="ex0_8cpp_source.html">Example 0</a>: simplest example, nodal H1 FEM for the Laplace problem
|
||||
@@ -105,8 +105,6 @@ namespace mfem {
|
||||
* - <a class="el" href="ex32p_8cpp_source.html">Example 32p</a>: parallel anisotropic Maxwell eigensolver
|
||||
* - <a class="el" href="ex33_8cpp_source.html">Example 33</a>: nodal H1 FEM for the fractional Laplacian problem
|
||||
* - <a class="el" href="ex33p_8cpp_source.html">Example 33p</a>: parallel nodal H1 FEM for the fractional Laplacian problem
|
||||
* - <a class="el" href="ex36_8cpp_source.html">Example 36</a>: Proximal Galerkin FEM for the obstacle problem
|
||||
* - <a class="el" href="ex36p_8cpp_source.html">Example 36p</a>: parallel Proximal Galerkin FEM for the obstacle problem
|
||||
*
|
||||
* <H4>AmgX Examples</H4>
|
||||
* - Variants of Examples
|
||||
@@ -188,7 +186,6 @@ namespace mfem {
|
||||
* - <a class="el" href="extruder_8cpp_source.html">Extruder</a>: extrude a low-dimensional mesh into a higher dimension
|
||||
* - <a class="el" href="mesh-explorer_8cpp_source.html">Mesh Explorer</a>: visualize and manipulate meshes
|
||||
* - <a class="el" href="mesh-optimizer_8cpp_source.html">Mesh Optimizer</a>: optimize high-order meshes, <a class="el" href="mesh-optimizer_8cpp_source.html">serial</a> and <a class="el" href="pmesh-optimizer_8cpp_source.html">parallel</a> versions
|
||||
* - <a class="el" href="mesh-quality_8cpp_source.html">Mesh Quality</a>: visualize and check mesh quality
|
||||
* - <a class="el" href="trimmer_8cpp_source.html">Trimmer</a>: trim elements from existing meshes
|
||||
* - <a class="el" href="display-basis_8cpp_source.html">Display Basis</a>: visualize finite element basis functions
|
||||
* - <a class="el" href="get-values_8cpp_source.html">Get Values</a>: extract field values via DataCollection classes
|
||||
@@ -201,15 +198,12 @@ namespace mfem {
|
||||
* - <a class="el" href="distance_8cpp_source.html">Distance</a>: finite element distance function solver
|
||||
* - <a class="el" href="diffusion_8cpp_source.html">Shifted Diffusion</a>: shifted boundary diffusion solver
|
||||
* - <a class="el" href="extrapolate_8cpp_source.html">Extrapolation</a>: PDE-based extrapolation of finite element functions
|
||||
* - <a class="el" href="block-solvers_8cpp_source.html">Block Solvers</a>: comparison of saddle point system solvers
|
||||
* - <a class="el" href="distance_8cpp_source.html">Block Solvers</a>: comparison of saddle point system solvers
|
||||
* - <a class="el" href="parheat_8cpp_source.html">Optimization gradients</a>: Gradients of PDE-constrained function
|
||||
* - <a class="el" href="par__example_8cpp_source.html">Parallel AD</a>: Parallel p-Laplacian example
|
||||
* - <a class="el" href="seq__example_8cpp_source.html">Serial AD</a>: Serial p-Laplacian example
|
||||
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
|
||||
* - <a class="el" href="generate__random__field_8cpp_source.html">SPDE Solvers</a>: SPDE solver random field generation
|
||||
* - <a class="el" href="pdiffusion_8cpp_source.html">DPG Diffusion example</a>: DPG formulation for the diffusion problem
|
||||
* - <a class="el" href="pmaxwell_8cpp_source.html">DPG Maxwell example</a>: DPG formulation for the indefinite Maxwell problem
|
||||
*
|
||||
* See also the <a class="el" href="https://mfem.org/examples/">examples documentation</a> online.
|
||||
*/
|
||||
|
||||
+2
-17
@@ -40,8 +40,6 @@ list(APPEND ALL_EXE_SRCS
|
||||
ex30.cpp
|
||||
ex31.cpp
|
||||
ex33.cpp
|
||||
ex34.cpp
|
||||
ex36.cpp
|
||||
)
|
||||
|
||||
if (MFEM_USE_MPI)
|
||||
@@ -79,9 +77,6 @@ if (MFEM_USE_MPI)
|
||||
ex31p.cpp
|
||||
ex32p.cpp
|
||||
ex33p.cpp
|
||||
ex34p.cpp
|
||||
ex35p.cpp
|
||||
ex36p.cpp
|
||||
)
|
||||
endif()
|
||||
|
||||
@@ -124,10 +119,9 @@ if (MFEM_ENABLE_TESTING)
|
||||
# Add CUDA/HIP tests.
|
||||
set(DEVICE_EXAMPLES
|
||||
# serial examples with device support:
|
||||
ex1 ex3 ex4 ex5 ex6 ex9 ex22 ex24 ex25 ex26 ex34
|
||||
ex1 ex3 ex4 ex5 ex6 ex9 ex22 ex24 ex25 ex26
|
||||
# parallel examples with device support:
|
||||
ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex9p ex13p ex22p ex24p ex25p ex26p
|
||||
ex34p ex35p)
|
||||
ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex9p ex13p ex22p ex24p ex25p ex26p)
|
||||
set(MFEM_TEST_DEVICE)
|
||||
if (MFEM_USE_CUDA)
|
||||
set(MFEM_TEST_DEVICE "cuda")
|
||||
@@ -167,15 +161,6 @@ if (MFEM_ENABLE_TESTING)
|
||||
$<TARGET_FILE:ex11p> "-no-vis" "--superlu"
|
||||
${MPIEXEC_POSTFLAGS})
|
||||
endif()
|
||||
|
||||
# If MUMPS is enabled, add a test run that uses it.
|
||||
if (MFEM_USE_MUMPS)
|
||||
add_test(NAME ex25p_mumps_np=${MFEM_MPI_NP}
|
||||
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
|
||||
${MPIEXEC_PREFLAGS}
|
||||
$<TARGET_FILE:ex25p> "-no-vis" "--mumps-solver"
|
||||
${MPIEXEC_POSTFLAGS})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Include the examples/amgx directory if AmgX is enabled
|
||||
|
||||
@@ -1,907 +0,0 @@
|
||||
#include "mfem.hpp"
|
||||
#include "IPsolver.hpp"
|
||||
#include "problems.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <cstdlib>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
|
||||
|
||||
InteriorPointSolver::InteriorPointSolver(OptProblem * Problem, ParFiniteElementSpace *Vhin)
|
||||
: problem(Problem), block_offsetsumlz(5), block_offsetsuml(4), block_offsetsx(3),
|
||||
saveLogBarrierIterates(false), Vh(Vhin)
|
||||
{
|
||||
tol = 1.e-2;
|
||||
max_iter = 20;
|
||||
mu_k = 1.0;
|
||||
|
||||
sMax = 1.e2;
|
||||
kSig = 1.e10; // control deviation from primal Hessian
|
||||
tauMin = 0.8; // control rate at which iterates can approach the boundary
|
||||
eta = 1.e-4; // backtracking constant
|
||||
thetaMin = 1.e-4; // allowed violation of the equality constraints
|
||||
|
||||
// constants in line-step A-5.4
|
||||
delta = 1.0;
|
||||
sTheta = 1.1;
|
||||
sPhi = 2.3;
|
||||
|
||||
// control the rate at which the penalty parameter is decreased
|
||||
kMu = 0.2;
|
||||
thetaMu = 1.5;
|
||||
|
||||
|
||||
thetaMax = 1.e6; // maximum constraint violation
|
||||
// data for the second order correction
|
||||
kSoc = 0.99;
|
||||
|
||||
// equation (18)
|
||||
gTheta = 1.e-5;
|
||||
gPhi = 1.e-5;
|
||||
|
||||
kEps = 1.e1;
|
||||
|
||||
dimU = problem->GetDimU();
|
||||
dimM = problem->GetDimM();
|
||||
dimC = problem->GetDimC();
|
||||
ckSoc.SetSize(dimC);
|
||||
|
||||
block_offsetsumlz[0] = 0;
|
||||
block_offsetsumlz[1] = dimU; // u
|
||||
block_offsetsumlz[2] = dimM; // m
|
||||
block_offsetsumlz[3] = dimC; // lambda
|
||||
block_offsetsumlz[4] = dimM; // zl
|
||||
block_offsetsumlz.PartialSum();
|
||||
|
||||
for(int i = 0; i < block_offsetsuml.Size(); i++) { block_offsetsuml[i] = block_offsetsumlz[i]; }
|
||||
for(int i = 0; i < block_offsetsx.Size(); i++) { block_offsetsx[i] = block_offsetsuml[i] ; }
|
||||
|
||||
// lower-bound for the inequality constraint m >= ml
|
||||
ml = problem->Getml();
|
||||
|
||||
lk.SetSize(dimC); lk = 0.0;
|
||||
zlk.SetSize(dimM); zlk = 0.0;
|
||||
mf.SetSize(dimM); mf = 0.0;
|
||||
|
||||
linSolver = 0;
|
||||
MyRank = 0;
|
||||
iAmRoot = MyRank == 0 ? true : false;
|
||||
}
|
||||
|
||||
double InteriorPointSolver::MaxStepSize(Vector &x, Vector &xl, Vector &xhat, double tau)
|
||||
{
|
||||
double alphaMaxloc = 1.0;
|
||||
double alphaTmp;
|
||||
for(int i = 0; i < x.Size(); i++)
|
||||
{
|
||||
if( xhat(i) < 0. )
|
||||
{
|
||||
alphaTmp = -1. * tau * (x(i) - xl(i)) / xhat(i);
|
||||
alphaMaxloc = min(alphaMaxloc, alphaTmp);
|
||||
}
|
||||
}
|
||||
|
||||
// alphaMaxloc is the local maximum step size which is
|
||||
// distinct on each MPI process. Need to compute
|
||||
// the global maximum step size
|
||||
double alphaMaxglb;
|
||||
alphaMaxglb = alphaMaxloc;
|
||||
return alphaMaxglb;
|
||||
}
|
||||
|
||||
double InteriorPointSolver::MaxStepSize(Vector &x, Vector &xhat, double tau)
|
||||
{
|
||||
Vector zero(x.Size()); zero = 0.0;
|
||||
return MaxStepSize(x, zero, xhat, tau);
|
||||
}
|
||||
|
||||
|
||||
void InteriorPointSolver::Mult(const Vector &x0, Vector &xf)
|
||||
{
|
||||
BlockVector x0block(block_offsetsx); x0block = 0.0;
|
||||
x0block.GetBlock(0).Set(1.0, x0);
|
||||
// hard coded initialization :(
|
||||
x0block.GetBlock(1) = 1.0;
|
||||
x0block.GetBlock(1).Add(1.0, ml);
|
||||
BlockVector xfblock(block_offsetsx); xfblock = 0.0;
|
||||
Mult(x0block, xfblock);
|
||||
xf.Set(1.0, xfblock.GetBlock(0));
|
||||
mf.Set(1.0, xfblock.GetBlock(1));
|
||||
}
|
||||
|
||||
void InteriorPointSolver::Mult(const BlockVector &x0, BlockVector &xf)
|
||||
{
|
||||
converged = false;
|
||||
IPNewtonKrylovIters.open("IPNewtonKrylovIters.dat", ios::out | ios::trunc);
|
||||
BlockVector xk(block_offsetsx), xhat(block_offsetsx); xk = 0; xhat = 0.0;
|
||||
BlockVector Xk(block_offsetsumlz), Xhat(block_offsetsumlz); Xk = 0.0; Xhat = 0.0;
|
||||
BlockVector Xhatuml(block_offsetsuml); Xhatuml = 0.0;
|
||||
Vector zlhat(dimM); zlhat = 0.0;
|
||||
|
||||
xk.GetBlock(0).Set(1.0, x0.GetBlock(0));
|
||||
xk.GetBlock(1).Set(1.0, x0.GetBlock(1));
|
||||
// running estimate of the final values of the Lagrange multipliers
|
||||
lk = 0.0;
|
||||
zlk = 0.0;
|
||||
|
||||
for(int i = 0; i < dimM; i++)
|
||||
{
|
||||
zlk(i) = 1.e1 * mu_k / (xk(i+dimU) - ml(i));
|
||||
}
|
||||
|
||||
Xk.GetBlock(0).Set(1.0, xk.GetBlock(0));
|
||||
Xk.GetBlock(1).Set(1.0, xk.GetBlock(1));
|
||||
Xk.GetBlock(2).Set(1.0, lk);
|
||||
Xk.GetBlock(3).Set(1.0, zlk);
|
||||
|
||||
/* set theta0 = theta(x0)
|
||||
* thetaMin
|
||||
* thetaMax
|
||||
* when theta(xk) < thetaMin and the switching condition holds
|
||||
* then we ask for the Armijo sufficient decrease of the barrier
|
||||
* objective to be satisfied, in order to accept the trial step length alphakl
|
||||
*
|
||||
* thetaMax controls how the filter is initialized for each log-barrier subproblem
|
||||
* F0 = {(th, phi) s.t. th > thetaMax}
|
||||
* that is the filter does not allow for iterates where the constraint violation
|
||||
* is larger than that of thetaMax
|
||||
*/
|
||||
double theta0 = theta(xk);
|
||||
thetaMin = 1.e-4 * max(1.0, theta0);
|
||||
thetaMax = 1.e8 * thetaMin;
|
||||
|
||||
double Eeval, maxBarrierSolves, Eevalmu0;
|
||||
bool printOptimalityError; // control optimality error print to console for log-barrier subproblems
|
||||
|
||||
maxBarrierSolves = 10;
|
||||
|
||||
for(jOpt = 0; jOpt < max_iter; jOpt++)
|
||||
{
|
||||
if(iAmRoot)
|
||||
{
|
||||
cout << "interior-point solve step " << jOpt << endl;
|
||||
}
|
||||
// A-2. Check convergence of overall optimization problem
|
||||
printOptimalityError = false;
|
||||
Eevalmu0 = E(xk, lk, zlk, printOptimalityError);
|
||||
if(Eevalmu0 < tol)
|
||||
{
|
||||
converged = true;
|
||||
if(iAmRoot)
|
||||
{
|
||||
IPNewtonKrylovIters.close();
|
||||
cout << "solved optimization problem :)\n";
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
if(jOpt > 0) { maxBarrierSolves = 1; }
|
||||
|
||||
for(int i = 0; i < maxBarrierSolves; i++)
|
||||
{
|
||||
// A-3. Check convergence of the barrier subproblem
|
||||
printOptimalityError = true;
|
||||
Eeval = E(xk, lk, zlk, mu_k, printOptimalityError);
|
||||
if(Eeval < kEps * mu_k)
|
||||
{
|
||||
if(iAmRoot)
|
||||
{
|
||||
cout << "solved barrier subproblem :), for mu = " << mu_k << endl;
|
||||
}
|
||||
// A-3.1. Recompute the barrier parameter
|
||||
mu_k = max(tol / 10., min(kMu * mu_k, pow(mu_k, thetaMu)));
|
||||
// A-3.2. Re-initialize the filter
|
||||
F1.DeleteAll();
|
||||
F2.DeleteAll();
|
||||
}
|
||||
else
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// A-4. Compute the search direction
|
||||
// solve for (uhat, mhat, lhat)
|
||||
if(iAmRoot)
|
||||
{
|
||||
cout << "\n** A-4. IP-Newton solve **\n";
|
||||
}
|
||||
zlhat = 0.0; Xhatuml = 0.0;
|
||||
// why do we have Xhatuml ....???
|
||||
// TO DO: remove Xhatuml in favor of passing Xhat
|
||||
IPNewtonSolve(xk, lk, zlk, zlhat, Xhatuml, mu_k, false);
|
||||
|
||||
|
||||
// assign data stack, X = (u, m, l, zl)
|
||||
Xk = 0.0;
|
||||
Xk.GetBlock(0).Set(1.0, xk.GetBlock(0));
|
||||
Xk.GetBlock(1).Set(1.0, xk.GetBlock(1));
|
||||
Xk.GetBlock(2).Set(1.0, lk);
|
||||
Xk.GetBlock(3).Set(1.0, zlk);
|
||||
|
||||
// assign data stack, Xhat = (uhat, mhat, lhat, zlhat)
|
||||
Xhat = 0.0;
|
||||
for(int i = 0; i < 3; i++)
|
||||
{
|
||||
Xhat.GetBlock(i).Set(1.0, Xhatuml.GetBlock(i));
|
||||
}
|
||||
Xhat.GetBlock(3).Set(1.0, zlhat);
|
||||
|
||||
|
||||
// A-5. Backtracking line search.
|
||||
if(iAmRoot)
|
||||
{
|
||||
cout << "\n** A-5. Linesearch **\n";
|
||||
cout << "mu = " << mu_k << endl;
|
||||
}
|
||||
lineSearch(Xk, Xhat, mu_k);
|
||||
|
||||
if(lineSearchSuccess)
|
||||
{
|
||||
if(iAmRoot)
|
||||
{
|
||||
cout << "lineSearch successful :)\n";
|
||||
}
|
||||
if(!switchCondition || !sufficientDecrease)
|
||||
{
|
||||
F1.Append( (1. - gTheta) * thx0);
|
||||
F2.Append( phx0 - gPhi * thx0);
|
||||
}
|
||||
// ----- A-6: Accept the trial point
|
||||
// print info regarding zl...
|
||||
xk.GetBlock(0).Add(alpha, Xhat.GetBlock(0));
|
||||
xk.GetBlock(1).Add(alpha, Xhat.GetBlock(1));
|
||||
lk.Add(alpha, Xhat.GetBlock(2));
|
||||
zlk.Add(alphaz, Xhat.GetBlock(3));
|
||||
projectZ(xk, zlk, mu_k);
|
||||
}
|
||||
else
|
||||
{
|
||||
if(iAmRoot)
|
||||
{
|
||||
cout << "lineSearch not successful :(\n";
|
||||
cout << "attempting feasibility restoration with theta = " << thx0 << endl;
|
||||
cout << "no feasibility restoration implemented, exiting now \n";
|
||||
}
|
||||
break;
|
||||
//cout << "feasibility restoration!!! :( :( :(\n";
|
||||
//problem->feasibilityRestoration(x, 1.e-12);
|
||||
// break;
|
||||
}
|
||||
//
|
||||
if(jOpt + 1 == max_iter && iAmRoot)
|
||||
{
|
||||
cout << "maximum optimization iterations :(\n";
|
||||
IPNewtonKrylovIters.close();
|
||||
}
|
||||
}
|
||||
// done with optimization routine, just reassign data to xf reference so
|
||||
// that the application code has access to the optimal point
|
||||
xf = 0.0;
|
||||
xf.GetBlock(0).Set(1.0, xk.GetBlock(0));
|
||||
xf.GetBlock(1).Set(1.0, xk.GetBlock(1));
|
||||
}
|
||||
|
||||
void InteriorPointSolver::FormIPNewtonMat(BlockVector & x, Vector & l, Vector &zl, BlockOperator &Ak)
|
||||
{
|
||||
// WARNING: Huu, Hum, Hmu, Hmm should all be Hessian terms of the Lagrangian, currently we
|
||||
// them by Hessian terms of the objective function and neglect the Hessian of l^T c
|
||||
|
||||
Huu = problem->Duuf(x); Hum = problem->Dumf(x);
|
||||
Hmu = problem->Dmuf(x); Hmm = problem->Dmmf(x);
|
||||
|
||||
Vector DiagLogBar(dimM); DiagLogBar = 0.0;
|
||||
for(int ii = 0; ii < dimM; ii++)
|
||||
{
|
||||
DiagLogBar(ii) = zl(ii) / (x(ii+dimU) - ml(ii));
|
||||
}
|
||||
if(saveLogBarrierIterates)
|
||||
{
|
||||
std::ofstream diagStream;
|
||||
char diagString[100];
|
||||
snprintf(diagString, 100, "logBarrierHessiandata/D%d.dat", jOpt);
|
||||
diagStream.open(diagString, ios::out | ios::trunc);
|
||||
for(int ii = 0; ii < dimM; ii++)
|
||||
{
|
||||
diagStream << setprecision(30) << DiagLogBar(ii) << endl;
|
||||
}
|
||||
diagStream.close();
|
||||
}
|
||||
|
||||
delete Wmm;
|
||||
if(Hmm != nullptr)
|
||||
{
|
||||
SparseMatrix * D = new SparseMatrix(DiagLogBar);
|
||||
Wmm = Add(*Hmm, *D);
|
||||
delete D;
|
||||
}
|
||||
else
|
||||
{
|
||||
Wmm = new SparseMatrix(DiagLogBar);
|
||||
}
|
||||
|
||||
delete JuT;
|
||||
delete JmT;
|
||||
Ju = problem->Duc(x); JuT = Transpose(*Ju);
|
||||
Jm = problem->Dmc(x); JmT = Transpose(*Jm);
|
||||
|
||||
// IP-Newton system matrix
|
||||
// Ak = [[H_(u,u) H_(u,m) J_u^T]
|
||||
// [H_(m,u) W_(m,m) J_m^T]
|
||||
// [ J_u J_m 0 ]]
|
||||
|
||||
Ak.SetBlock(0, 0, Huu); Ak.SetBlock(0, 2, JuT);
|
||||
Ak.SetBlock(1, 1, Wmm); Ak.SetBlock(1, 2, JmT);
|
||||
Ak.SetBlock(2, 0, Ju); Ak.SetBlock(2, 1, Jm);
|
||||
|
||||
if(Hum != nullptr) { Ak.SetBlock(0, 1, Hum); Ak.SetBlock(1, 0, Hmu); }
|
||||
}
|
||||
|
||||
|
||||
// perturbed KKT system solve
|
||||
// determine the search direction
|
||||
void InteriorPointSolver::IPNewtonSolve(BlockVector &x, Vector &l, Vector &zl, Vector &zlhat, BlockVector &Xhat, double mu, bool socSolve)
|
||||
{
|
||||
// solve A x = b, where A is the IP-Newton matrix
|
||||
BlockOperator A(block_offsetsuml, block_offsetsuml); BlockVector b(block_offsetsuml); b = 0.0;
|
||||
FormIPNewtonMat(x, l, zl, A);
|
||||
|
||||
// [grad_u phi + Ju^T l]
|
||||
// b = - [grad_m phi + Jm^T l]
|
||||
// [ c ]
|
||||
BlockVector gradphi(block_offsetsx); gradphi = 0.0;
|
||||
BlockVector JTl(block_offsetsx); JTl = 0.0;
|
||||
Dxphi(x, mu, gradphi);
|
||||
|
||||
(A.GetBlock(0,2)).Mult(l, JTl.GetBlock(0));
|
||||
(A.GetBlock(1,2)).Mult(l, JTl.GetBlock(1));
|
||||
|
||||
for(int ii = 0; ii < 2; ii++)
|
||||
{
|
||||
b.GetBlock(ii).Set(1.0, gradphi.GetBlock(ii));
|
||||
b.GetBlock(ii).Add(1.0, JTl.GetBlock(ii));
|
||||
}
|
||||
if(!socSolve)
|
||||
{
|
||||
problem->c(x, b.GetBlock(2));
|
||||
}
|
||||
else
|
||||
{
|
||||
b.GetBlock(2).Set(1.0, ckSoc);
|
||||
}
|
||||
b *= -1.0;
|
||||
Xhat = 0.0;
|
||||
|
||||
#ifdef MFEM_USE_SUITESPARSE
|
||||
// Direct solve for IP-Newton saddle-point system
|
||||
// A = [ [ Huu 0 Ju^T]
|
||||
// [ 0 D -I ]
|
||||
// [ Ju -I 0 ]]
|
||||
if(linSolver == 0)
|
||||
{
|
||||
BlockMatrix ABlockMatrix(block_offsetsuml, block_offsetsuml);
|
||||
for(int ii = 0; ii < 3; ii++)
|
||||
{
|
||||
for(int jj = 0; jj < 3; jj++)
|
||||
{
|
||||
if(!A.IsZeroBlock(ii, jj))
|
||||
{
|
||||
ABlockMatrix.SetBlock(ii, jj, dynamic_cast<SparseMatrix *>(&(A.GetBlock(ii, jj))));
|
||||
}
|
||||
}
|
||||
}
|
||||
/* direct solve of the 3x3 IP-Newton linear system */
|
||||
UMFPackSolver ASolver;
|
||||
SparseMatrix *ASparse = ABlockMatrix.CreateMonolithic();
|
||||
ASolver.SetOperator(*ASparse);
|
||||
ASolver.Mult(b, Xhat);
|
||||
|
||||
Vector residual(Xhat.Size());
|
||||
ASparse->Mult(Xhat, residual);
|
||||
residual.Add(-1.0, b);
|
||||
delete ASparse;
|
||||
}
|
||||
else if(linSolver == 1)
|
||||
{
|
||||
// Direct solve for 0,0 Schur complement of IP-Newton system, Huu + Ju^T Wmm Ju,
|
||||
// where Wmm = D for contact problems
|
||||
SparseMatrix * Huuloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(0, 0))));
|
||||
SparseMatrix * Wmmloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(1, 1))));
|
||||
SparseMatrix * Juloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(2, 0))));
|
||||
SparseMatrix * JuTloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(0, 2))));
|
||||
Vector Dvec(dimM); Dvec = 0.0;
|
||||
Vector one(dimM); one = 1.0;
|
||||
Wmmloc->Mult(one, Dvec);
|
||||
SparseMatrix *JuTDJu = Mult_AtDA(*Juloc, Dvec); // Ju^T D Ju
|
||||
SparseMatrix *Areduced = Add(*Huuloc, *JuTDJu); // Huu + Ju^T D Ju
|
||||
|
||||
|
||||
/* prepare the reduced rhs */
|
||||
// breduced = bu + Ju^T (bm + Wmm bl)
|
||||
Vector breduced(dimU); breduced = 0.0;
|
||||
Vector tempVec(dimM); tempVec = 0.0;
|
||||
Wmmloc->Mult(b.GetBlock(2), tempVec);
|
||||
tempVec.Add(1.0, b.GetBlock(1));
|
||||
JuTloc->Mult(tempVec, breduced);
|
||||
breduced.Add(1.0, b.GetBlock(0));
|
||||
|
||||
// solve the reduced linear system
|
||||
UMFPackSolver AreducedSolver;
|
||||
AreducedSolver.SetOperator(*Areduced);
|
||||
AreducedSolver.Mult(breduced, Xhat.GetBlock(0));
|
||||
|
||||
// now propagate solved uhat to obtain mhat and lhat
|
||||
// xm = Ju xu - bl
|
||||
Juloc->Mult(Xhat.GetBlock(0), Xhat.GetBlock(1));
|
||||
Xhat.GetBlock(1).Add(-1.0, b.GetBlock(2));
|
||||
|
||||
// xl = Wmm xm - bm
|
||||
Wmmloc->Mult(Xhat.GetBlock(1), Xhat.GetBlock(2));
|
||||
Xhat.GetBlock(2).Add(-1.0, b.GetBlock(1));
|
||||
|
||||
delete Wmmloc;
|
||||
delete Huuloc;
|
||||
delete JuTDJu;
|
||||
delete Juloc;
|
||||
delete Areduced;
|
||||
}
|
||||
#else
|
||||
MFEM_VERIFY(linSolver > 1, "linSolver = 0, 1 require MFEM_USE_SUITESPARSE=YES");
|
||||
#endif
|
||||
if (linSolver == 2 || linSolver == 3)
|
||||
{
|
||||
// Iterative solve for 0,0 Schur complement of IP-Newton system, Huu + Ju^T Wmm Ju,
|
||||
// where Wmm = D for contact problems
|
||||
// here the iterative solver is a Jacobi-preconditioned CG-solve
|
||||
SparseMatrix * Huuloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(0, 0))));
|
||||
SparseMatrix * Wmmloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(1, 1))));
|
||||
SparseMatrix * Juloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(2, 0))));
|
||||
SparseMatrix * JuTloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(0, 2))));
|
||||
// Vector Dvec(dimM); Dvec = 0.0;
|
||||
// Vector one(dimM); one = 1.0;
|
||||
// Wmmloc->Mult(one, Dvec);
|
||||
|
||||
// SparseMatrix *JuTDJu = Mult_AtDA(*Juloc, Dvec); // Ju^T D Ju
|
||||
SparseMatrix *JuTDJu = RAP(*Juloc,*Wmmloc,*Juloc); // Ju^T D Ju
|
||||
SparseMatrix *Areduced = Add(*Huuloc, *JuTDJu); // Huu + Ju^T D Ju
|
||||
/* prepare the reduced rhs */
|
||||
// breduced = bu + Ju^T (bm + Wmm bl)
|
||||
Vector breduced(dimU); breduced = 0.0;
|
||||
Vector tempVec(dimM); tempVec = 0.0;
|
||||
Wmmloc->Mult(b.GetBlock(2), tempVec);
|
||||
tempVec.Add(1.0, b.GetBlock(1));
|
||||
JuTloc->Mult(tempVec, breduced);
|
||||
breduced.Add(1.0, b.GetBlock(0));
|
||||
|
||||
/* set up an iterative solver */
|
||||
int globalNumRows = dimU;
|
||||
HYPRE_BigInt rowStarts[2];
|
||||
rowStarts[0] = 0;
|
||||
rowStarts[1] = dimU;
|
||||
HypreParMatrix * Ahypre = new HypreParMatrix(MPI_COMM_WORLD, globalNumRows, rowStarts, Areduced);
|
||||
// CGSolver Asolver(MPI_COMM_WORLD);
|
||||
HyprePCG Asolver(MPI_COMM_WORLD);
|
||||
HypreBoomerAMG * Aprec = new HypreBoomerAMG(*Ahypre);
|
||||
Aprec->SetPrintLevel(0);
|
||||
if(linSolver == 3)
|
||||
{
|
||||
Aprec->SetElasticityOptions(Vh);
|
||||
}
|
||||
Aprec->SetSystemsOptions(3,false);
|
||||
|
||||
Asolver.SetOperator(*Ahypre);
|
||||
Asolver.SetPrintLevel(2);
|
||||
Asolver.SetMaxIter(1000);
|
||||
// Asolver.SetResidualConvergenceOptions();
|
||||
Asolver.SetTol(1.e-6);
|
||||
Asolver.SetPreconditioner(*Aprec);
|
||||
// Asolver.SetResidualConvergenceOptions();
|
||||
|
||||
Asolver.Mult(breduced, Xhat.GetBlock(0));
|
||||
int num_iterations;
|
||||
Asolver.GetNumIterations(num_iterations);
|
||||
cgnum_iterations.Append(num_iterations);
|
||||
// int numNewtonKrylovIters = -1;
|
||||
// numNewtonKrylovIters = Asolver.GetNumIterations();
|
||||
// IPNewtonKrylovIters << numNewtonKrylovIters << endl;
|
||||
|
||||
delete Aprec;
|
||||
delete Ahypre;
|
||||
|
||||
// now propagate solved uhat to obtain mhat and lhat
|
||||
// xm = Ju xu - bl
|
||||
Juloc->Mult(Xhat.GetBlock(0), Xhat.GetBlock(1));
|
||||
Xhat.GetBlock(1).Add(-1.0, b.GetBlock(2));
|
||||
|
||||
// // xl = Wmm xm - bm
|
||||
Wmmloc->Mult(Xhat.GetBlock(1), Xhat.GetBlock(2));
|
||||
Xhat.GetBlock(2).Add(-1.0, b.GetBlock(1));
|
||||
|
||||
|
||||
delete Wmmloc;
|
||||
delete Huuloc;
|
||||
delete JuTDJu;
|
||||
delete Juloc;
|
||||
delete Areduced;
|
||||
}
|
||||
else if(linSolver > 2)
|
||||
{
|
||||
// Iterative solve for 0,0 Schur complement of IP-Newton system, Huu + Ju^T Wmm Ju,
|
||||
// where Wmm = D for contact problems
|
||||
// here the iterative solver is a Jacobi-preconditioned CG-solve
|
||||
SparseMatrix * Huuloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(0, 0))));
|
||||
SparseMatrix * Wmmloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(1, 1))));
|
||||
SparseMatrix * Juloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(2, 0))));
|
||||
SparseMatrix * JuTloc = new SparseMatrix(*dynamic_cast<SparseMatrix *>(&(A.GetBlock(0, 2))));
|
||||
Vector Dvec(dimM); Dvec = 0.0;
|
||||
Vector one(dimM); one = 1.0;
|
||||
Wmmloc->Mult(one, Dvec);
|
||||
SparseMatrix *JuTDJu = Mult_AtDA(*Juloc, Dvec); // Ju^T D Ju
|
||||
SparseMatrix *Areduced = Add(*Huuloc, *JuTDJu); // Huu + Ju^T D Ju
|
||||
|
||||
/* prepare the reduced rhs */
|
||||
// breduced = bu + Ju^T (bm + Wmm bl)
|
||||
Vector breduced(dimU); breduced = 0.0;
|
||||
Vector tempVec(dimM); tempVec = 0.0;
|
||||
Wmmloc->Mult(b.GetBlock(2), tempVec);
|
||||
tempVec.Add(1.0, b.GetBlock(1));
|
||||
JuTloc->Mult(tempVec, breduced);
|
||||
breduced.Add(1.0, b.GetBlock(0));
|
||||
|
||||
/* set up an iterative solver */
|
||||
GSSmoother AreducedPrec((SparseMatrix &)(*Areduced));
|
||||
GMRESSolver AreducedSolver;
|
||||
AreducedSolver.SetOperator(*Areduced);
|
||||
AreducedSolver.SetAbsTol(1.e-12);
|
||||
AreducedSolver.SetRelTol(1.e-8);
|
||||
AreducedSolver.SetMaxIter(500);
|
||||
AreducedSolver.SetPreconditioner(AreducedPrec);
|
||||
AreducedSolver.SetPrintLevel(1);
|
||||
AreducedSolver.Mult(breduced, Xhat.GetBlock(0));
|
||||
|
||||
// now propagate solved uhat to obtain mhat and lhat
|
||||
// xm = Ju xu - bl
|
||||
Juloc->Mult(Xhat.GetBlock(0), Xhat.GetBlock(1));
|
||||
Xhat.GetBlock(1).Add(-1.0, b.GetBlock(2));
|
||||
|
||||
// xl = Wmm xm - bm
|
||||
Wmmloc->Mult(Xhat.GetBlock(1), Xhat.GetBlock(2));
|
||||
Xhat.GetBlock(2).Add(-1.0, b.GetBlock(1));
|
||||
|
||||
delete Wmmloc;
|
||||
delete Huuloc;
|
||||
delete JuTDJu;
|
||||
delete Juloc;
|
||||
delete Areduced;
|
||||
}
|
||||
|
||||
|
||||
/* backsolve to determine zlhat */
|
||||
for(int ii = 0; ii < dimM; ii++)
|
||||
{
|
||||
zlhat(ii) = -1.*(zl(ii) + (zl(ii) * Xhat(ii + dimU) - mu) / (x(ii + dimU) - ml(ii)) );
|
||||
}
|
||||
}
|
||||
|
||||
// here Xhat, X will be BlockVectors w.r.t. the 4 partitioning X = (u, m, l, zl)
|
||||
|
||||
void InteriorPointSolver::lineSearch(BlockVector& X0, BlockVector& Xhat, double mu)
|
||||
{
|
||||
double tau = max(tauMin, 1.0 - mu);
|
||||
Vector u0 = X0.GetBlock(0);
|
||||
Vector m0 = X0.GetBlock(1);
|
||||
Vector l0 = X0.GetBlock(2);
|
||||
Vector z0 = X0.GetBlock(3);
|
||||
Vector uhat = Xhat.GetBlock(0);
|
||||
Vector mhat = Xhat.GetBlock(1);
|
||||
Vector lhat = Xhat.GetBlock(2);
|
||||
Vector zhat = Xhat.GetBlock(3);
|
||||
double alphaMax = MaxStepSize(m0, ml, mhat, tau);
|
||||
double alphaMaxz = MaxStepSize(z0, zhat, tau);
|
||||
alphaz = alphaMaxz;
|
||||
|
||||
|
||||
BlockVector x0(block_offsetsx); x0 = 0.0;
|
||||
x0.GetBlock(0).Set(1.0, u0);
|
||||
x0.GetBlock(1).Set(1.0, m0);
|
||||
|
||||
BlockVector xhat(block_offsetsx); xhat = 0.0;
|
||||
xhat.GetBlock(0).Set(1.0, uhat);
|
||||
xhat.GetBlock(1).Set(1.0, mhat);
|
||||
|
||||
BlockVector xtrial(block_offsetsx); xtrial = 0.0;
|
||||
BlockVector Dxphi0(block_offsetsx); Dxphi0 = 0.0;
|
||||
int maxBacktrack = 20;
|
||||
alpha = alphaMax;
|
||||
|
||||
|
||||
Vector ck0(dimC); ck0 = 0.0;
|
||||
Vector zhatsoc(dimM); zhatsoc = 0.0;
|
||||
BlockVector Xhatumlsoc(block_offsetsuml); Xhatumlsoc = 0.0;
|
||||
BlockVector xhatsoc(block_offsetsx); xhatsoc = 0.0;
|
||||
Vector uhatsoc(dimU); uhatsoc = 0.0;
|
||||
Vector mhatsoc(dimM); mhatsoc = 0.0;
|
||||
|
||||
Dxphi(x0, mu, Dxphi0);
|
||||
Dxphi0_xhat = InnerProduct(Dxphi0, xhat);
|
||||
double xhat_L2norm = sqrt(InnerProduct(xhat, xhat));
|
||||
double Dxphi_L2norm = sqrt(InnerProduct(Dxphi0, Dxphi0));
|
||||
descentDirection = Dxphi0_xhat < 0. ? true : false;
|
||||
if(descentDirection)
|
||||
{
|
||||
cout << "is a descent direction for the log-barrier objective\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
cout << "is not a descent direction for the log-barrier objective\n";
|
||||
}
|
||||
cout << "Dxphi^T xhat / (|| Dxphi ||_2 * || xhat ||_2) = " << Dxphi0_xhat / (xhat_L2norm * Dxphi_L2norm) << endl;
|
||||
thx0 = theta(x0);
|
||||
phx0 = phi(x0, mu);
|
||||
|
||||
lineSearchSuccess = false;
|
||||
for(int i = 0; i < maxBacktrack; i++)
|
||||
{
|
||||
cout << "\n--------- alpha = " << alpha << " ---------\n";
|
||||
|
||||
// ----- A-5.2. Compute trial point: xtrial = x0 + alpha_i xhat
|
||||
xtrial.Set(1.0, x0);
|
||||
xtrial.Add(alpha, xhat);
|
||||
|
||||
// ------ A-5.3. if not in filter region go to A.5.4 otherwise go to A-5.5.
|
||||
thxtrial = theta(xtrial);
|
||||
phxtrial = phi(xtrial, mu);
|
||||
filterCheck(thxtrial, phxtrial);
|
||||
if(!inFilterRegion)
|
||||
{
|
||||
cout << "not in filter region :)\n";
|
||||
// ------ A.5.4: Check sufficient decrease
|
||||
if(!descentDirection)
|
||||
{
|
||||
switchCondition = false;
|
||||
}
|
||||
else
|
||||
{
|
||||
switchCondition = (alpha * pow(abs(Dxphi0_xhat), sPhi) > delta * pow(thx0, sTheta)) ? true : false;
|
||||
}
|
||||
cout << "alpha |Dxphi(x0)^T xhat|^sPhi = " << alpha * pow(abs(Dxphi0_xhat), sPhi) << endl;
|
||||
cout << "delta * theta(x0)^sTheta = " << delta * pow(thx0, sTheta) << endl;
|
||||
cout << "theta(x0) = " << thx0 << ", thetaMin = " << thetaMin << endl;
|
||||
cout << "theta(xtrial) = " << thxtrial << ", (1-gTheta) *theta(x0) = " << (1. - gTheta) * thx0 << endl;
|
||||
cout << "phi(xtrial) = " << phxtrial << ", phi(x0) - gPhi *theta(x0) = " << phx0 - gPhi * thx0 << endl;
|
||||
|
||||
// Case I
|
||||
if(thx0 <= thetaMin && switchCondition)
|
||||
{
|
||||
sufficientDecrease = phxtrial <= phx0 + eta * alpha * Dxphi0_xhat ? true : false;
|
||||
if(sufficientDecrease)
|
||||
{
|
||||
if(iAmRoot) { cout << "A-5.4. Case I -- accepted step length.\n"; }
|
||||
// accept the trial step
|
||||
lineSearchSuccess = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if(thxtrial <= (1. - gTheta) * thx0 || phxtrial <= phx0 - gPhi * thx0)
|
||||
{
|
||||
if(iAmRoot) { cout << "A-5.4. Case II -- accepted step length.\n"; }
|
||||
// accept the trial step
|
||||
lineSearchSuccess = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
// A-5.5: Initialize the second-order correction
|
||||
if((!(thx0 < thxtrial)) && i == 0)
|
||||
{
|
||||
cout << "second order correction\n";
|
||||
problem->c(xtrial, ckSoc);
|
||||
problem->c(x0, ck0);
|
||||
ckSoc.Add(alphaMax, ck0);
|
||||
// A-5.6 Compute the second-order correction.
|
||||
IPNewtonSolve(x0, l0, z0, zhatsoc, Xhatumlsoc, mu, true);
|
||||
mhatsoc.Set(1.0, Xhatumlsoc.GetBlock(1));
|
||||
// alphasoc = MaxStepSize(m0, ml, mhatsoc, tau);
|
||||
//WARNING: not complete but currently solver isn't entering this region
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
cout << "in filter region :(\n";
|
||||
}
|
||||
|
||||
// include more if needed
|
||||
alpha *= 0.5;
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
void InteriorPointSolver::projectZ(const Vector &x, Vector &z, double mu)
|
||||
{
|
||||
double zi;
|
||||
double mudivmml;
|
||||
for(int i = 0; i < dimM; i++)
|
||||
{
|
||||
zi = z(i);
|
||||
mudivmml = mu / (x(i + dimU) - ml(i));
|
||||
z(i) = max(min(zi, kSig * mudivmml), mudivmml / kSig);
|
||||
}
|
||||
}
|
||||
|
||||
void InteriorPointSolver::filterCheck(double th, double ph)
|
||||
{
|
||||
inFilterRegion = false;
|
||||
if(th > thetaMax)
|
||||
{
|
||||
inFilterRegion = true;
|
||||
}
|
||||
else
|
||||
{
|
||||
for(int i = 0; i < F1.Size(); i++)
|
||||
{
|
||||
if(th >= F1[i] && ph >= F2[i])
|
||||
{
|
||||
inFilterRegion = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
double InteriorPointSolver::E(const BlockVector &x, const Vector &l, const Vector &zl, double mu, bool print)
|
||||
{
|
||||
double E1, E2, E3;
|
||||
double sc, sd;
|
||||
BlockVector gradL(block_offsetsx); gradL = 0.0; // stationarity grad L = grad f + J^T l - z
|
||||
Vector cx(dimC); cx = 0.0; // feasibility c = c(x)
|
||||
Vector comp(dimM); comp = 0.0; // complementarity M Z - mu 1
|
||||
|
||||
DxL(x, l, zl, gradL);
|
||||
E1 = gradL.Normlinf();
|
||||
|
||||
problem->c(x, cx);
|
||||
E2 = cx.Normlinf();
|
||||
|
||||
for(int ii = 0; ii < dimM; ii++)
|
||||
{
|
||||
comp(ii) = x(dimU + ii) * zl(ii) - mu;
|
||||
}
|
||||
E3 = comp.Normlinf();
|
||||
|
||||
double ll1, zl1;
|
||||
zl1 = zl.Norml1() / double(dimC + dimM);
|
||||
ll1 = l.Norml1();
|
||||
sc = max(sMax, zl1 / (double(dimM)) ) / sMax;
|
||||
sd = max(sMax, (ll1 + zl1) / (double(dimC + dimM))) / sMax;
|
||||
if(iAmRoot && print)
|
||||
{
|
||||
cout << "evaluating optimality error for mu = " << mu << endl;
|
||||
cout << "stationarity measure = " << E1 / sd << endl;
|
||||
cout << "feasibility measure = " << E2 << endl;
|
||||
cout << "complimentarity measure = " << E3 / sc << endl;
|
||||
}
|
||||
return max(max(E1 / sd, E2), E3 / sc);
|
||||
}
|
||||
|
||||
double InteriorPointSolver::E(const BlockVector &x, const Vector &l, const Vector &zl, bool print)
|
||||
{
|
||||
return E(x, l, zl, 0.0, print);
|
||||
}
|
||||
|
||||
double InteriorPointSolver::theta(const BlockVector &x)
|
||||
{
|
||||
Vector cx(dimC); cx = 0.0;
|
||||
problem->c(x, cx);
|
||||
return sqrt(InnerProduct(cx, cx));
|
||||
}
|
||||
|
||||
// log-barrier objective
|
||||
double InteriorPointSolver::phi(const BlockVector &x, double mu)
|
||||
{
|
||||
double fx = problem->CalcObjective(x);
|
||||
double logBarrierLoc = 0.0;
|
||||
for(int i = 0; i < dimM; i++)
|
||||
{
|
||||
logBarrierLoc += log(x(dimU+i)-ml(i));
|
||||
}
|
||||
double logBarrierGlb = 0.0;
|
||||
logBarrierGlb = logBarrierLoc;
|
||||
return fx - mu * logBarrierGlb;
|
||||
}
|
||||
|
||||
|
||||
// gradient of log-barrier objective with respect to x = (u, m)
|
||||
void InteriorPointSolver::Dxphi(const BlockVector &x, double mu, BlockVector &y)
|
||||
{
|
||||
problem->CalcObjectiveGrad(x, y);
|
||||
for(int i = 0; i < dimM; i++)
|
||||
{
|
||||
y(dimU + i) -= mu / (x(dimU + i) - ml(i));
|
||||
}
|
||||
}
|
||||
|
||||
// Lagrangian function evaluation
|
||||
// L(x, l, zl) = f(x) + l^T c(x) - zl^T m
|
||||
double InteriorPointSolver::L(const BlockVector &x, const Vector &l, const Vector &zl)
|
||||
{
|
||||
double fx = problem->CalcObjective(x);
|
||||
Vector cx(dimC); problem->c(x, cx);
|
||||
return (fx + InnerProduct(cx, l) - InnerProduct(x.GetBlock(1), zl));
|
||||
}
|
||||
|
||||
void InteriorPointSolver::DxL(const BlockVector &x, const Vector &l, const Vector &zl, BlockVector &y)
|
||||
{
|
||||
// evaluate the gradient of the objective with respect to the primal variables x = (u, m)
|
||||
BlockVector gradxf(block_offsetsx); gradxf = 0.0;
|
||||
problem->CalcObjectiveGrad(x, gradxf);
|
||||
|
||||
SparseMatrix *Jacu, *Jacm, *JacuT, *JacmT;
|
||||
Jacu = problem->Duc(x); Jacm = problem->Dmc(x);
|
||||
JacuT = Transpose(*Jacu);
|
||||
JacmT = Transpose(*Jacm);
|
||||
JacuT->Mult(l, y.GetBlock(0));
|
||||
JacmT->Mult(l, y.GetBlock(1));
|
||||
delete Jacu; delete JacuT;
|
||||
delete Jacm; delete JacmT;
|
||||
y.Add(1.0, gradxf);
|
||||
(y.GetBlock(1)).Add(-1.0, zl);
|
||||
}
|
||||
|
||||
|
||||
bool InteriorPointSolver::GetConverged() const
|
||||
{
|
||||
return converged;
|
||||
}
|
||||
|
||||
void InteriorPointSolver::SetTol(double Tol)
|
||||
{
|
||||
tol = Tol;
|
||||
}
|
||||
|
||||
void InteriorPointSolver::SetMaxIter(int max_it)
|
||||
{
|
||||
max_iter = max_it;
|
||||
}
|
||||
|
||||
void InteriorPointSolver::SetBarrierParameter(double mu_0)
|
||||
{
|
||||
mu_k = mu_0;
|
||||
}
|
||||
|
||||
void InteriorPointSolver::SaveLogBarrierHessianIterates(bool save)
|
||||
{
|
||||
MFEM_ASSERT(MyRank == 0 || save == false, "currently can only save logbarrier hessian in serial codes");
|
||||
saveLogBarrierIterates = save;
|
||||
}
|
||||
|
||||
void InteriorPointSolver::SetLinearSolver(int LinSolver)
|
||||
{
|
||||
linSolver = LinSolver;
|
||||
}
|
||||
|
||||
|
||||
|
||||
InteriorPointSolver::~InteriorPointSolver()
|
||||
{
|
||||
delete Wmm;
|
||||
delete Huu;
|
||||
delete Hum;
|
||||
delete Hmu;
|
||||
delete Hmm;
|
||||
delete Hum;
|
||||
delete Ju;
|
||||
delete Jm;
|
||||
delete JuT;
|
||||
delete JmT;
|
||||
|
||||
F1.DeleteAll();
|
||||
F2.DeleteAll();
|
||||
block_offsetsx.DeleteAll();
|
||||
block_offsetsumlz.DeleteAll();
|
||||
block_offsetsuml.DeleteAll();
|
||||
ml.SetSize(0);
|
||||
}
|
||||
@@ -1,103 +0,0 @@
|
||||
#include "mfem.hpp"
|
||||
#include "problems.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
|
||||
#ifndef IPSOLVER
|
||||
#define IPSOLVER
|
||||
|
||||
class InteriorPointSolver
|
||||
{
|
||||
protected:
|
||||
OptProblem* problem;
|
||||
double tol;
|
||||
int max_iter;
|
||||
double mu_k; // \mu_k
|
||||
Vector lk, zlk, mf;
|
||||
|
||||
double sMax, kSig, tauMin, eta, thetaMin, delta, sTheta, sPhi, kMu, thetaMu;
|
||||
double thetaMax, kSoc, gTheta, gPhi, kEps;
|
||||
|
||||
// filter
|
||||
Array<double> F1, F2;
|
||||
|
||||
// quantities computed in lineSearch
|
||||
double alpha, alphaz;
|
||||
double thx0, thxtrial;
|
||||
double phx0, phxtrial;
|
||||
bool descentDirection, switchCondition, sufficientDecrease, lineSearchSuccess, inFilterRegion;
|
||||
double Dxphi0_xhat;
|
||||
|
||||
int dimU, dimM, dimC;
|
||||
Array<int> block_offsetsumlz, block_offsetsuml, block_offsetsx;
|
||||
Vector ml;
|
||||
|
||||
Vector ckSoc;
|
||||
SparseMatrix * Huu = nullptr;
|
||||
SparseMatrix * Hum = nullptr;
|
||||
SparseMatrix * Hmu = nullptr;
|
||||
SparseMatrix * Hmm = nullptr;
|
||||
SparseMatrix * Wmm = nullptr;
|
||||
SparseMatrix * Ju = nullptr;
|
||||
SparseMatrix * Jm = nullptr;
|
||||
SparseMatrix * JuT = nullptr;
|
||||
SparseMatrix * JmT = nullptr;;
|
||||
|
||||
int jOpt;
|
||||
bool converged;
|
||||
|
||||
int MyRank;
|
||||
bool iAmRoot;
|
||||
|
||||
bool saveLogBarrierIterates;
|
||||
|
||||
int linSolver;
|
||||
std::ofstream IPNewtonKrylovIters;
|
||||
|
||||
ParFiniteElementSpace *Vh;
|
||||
Array<int> cgnum_iterations;
|
||||
|
||||
|
||||
// not sure if this data is needed or if it can
|
||||
// all be accounted for in the problem class
|
||||
// which variables have equality constraints
|
||||
//Array<int> eqConstrainedVariables;
|
||||
//Array<double> eqConstrainedValues;
|
||||
|
||||
|
||||
|
||||
public:
|
||||
InteriorPointSolver(OptProblem*, ParFiniteElementSpace *);
|
||||
void Mult(const BlockVector& , BlockVector&); // used when the user wants to be aware of bound-constrained variable m >= ml
|
||||
void Mult(const Vector&, Vector &); // useful when the user doesn't need to know about bound-constrained variable m >= ml
|
||||
double MaxStepSize(Vector& , Vector& , Vector& , double);
|
||||
double MaxStepSize(Vector& , Vector& , double);
|
||||
void FormIPNewtonMat(BlockVector& , Vector& , Vector& , BlockOperator &);
|
||||
void IPNewtonSolve(BlockVector& , Vector& , Vector& , Vector&, BlockVector& , double, bool);
|
||||
void lineSearch(BlockVector& , BlockVector& , double);
|
||||
void projectZ(const Vector & , Vector &, double);
|
||||
void filterCheck(double, double);
|
||||
double E(const BlockVector &, const Vector &, const Vector &, double, bool);
|
||||
double E(const BlockVector &, const Vector &, const Vector &, bool);
|
||||
bool GetConverged() const;
|
||||
// TO DO: include Hessian of Lagrangian
|
||||
double theta(const BlockVector &);
|
||||
double phi(const BlockVector &, double);
|
||||
void Dxphi(const BlockVector &, double, BlockVector &);
|
||||
double L(const BlockVector &, const Vector &, const Vector &);
|
||||
void DxL(const BlockVector &, const Vector &, const Vector &, BlockVector &);
|
||||
void SetTol(double);
|
||||
void SetMaxIter(int);
|
||||
void SetBarrierParameter(double);
|
||||
void SaveLogBarrierHessianIterates(bool);
|
||||
void SetLinearSolver(int);
|
||||
Vector GetBoundConstrainedVariable() {return mf;}
|
||||
Array<int> & GetCGIterNumbers() {return cgnum_iterations;}
|
||||
virtual ~InteriorPointSolver();
|
||||
};
|
||||
|
||||
#endif
|
||||
@@ -1,17 +0,0 @@
|
||||
# OneProcessAMGContact
|
||||
|
||||
|
||||
|
||||
Be sure to edit the makefile so that it points to a parallel MFEM build
|
||||
|
||||
specifically the MFEM_BUILD_DIR
|
||||
|
||||
|
||||
after building exQPContactBlockTL one can
|
||||
|
||||
1. run the bash script scalingJobArray.bat via `source scalingJobArray.bat' which will populate the CG iterations required to solve
|
||||
various linear systems into the data/ subdirectory
|
||||
2. run the python script data/process.py in order to put the scaling information into the single files algorithmicScaling_Elasticity.dat and algorithmicScaling_noElasticity.dat
|
||||
in order to see the number of average AMG-CG iterations per optimization solve.
|
||||
|
||||
|
||||
@@ -1,274 +0,0 @@
|
||||
// Contact example
|
||||
//
|
||||
// Compile with: make contact
|
||||
//
|
||||
// Sample runs: ./contact -m1 block1.mesh -m2 block2.mesh -at "5 6 7 8"
|
||||
// Sample runs: ./contact -m1 block1_d.mesh -m2 block2_d.mesh -at "5 6 7 8"
|
||||
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <array>
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include "problems.hpp"
|
||||
#include "IPsolver.hpp"
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
Mpi::Init(argc, argv);
|
||||
Hypre::Init();
|
||||
int linSolver = 2;
|
||||
int maxIPMiters = 30;
|
||||
bool iAmRoot = true;
|
||||
int ref_levels = 0;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&linSolver, "-linSolver", "--linearSolver", \
|
||||
"IP-Newton linear system solution strategy.");
|
||||
args.AddOption(&maxIPMiters, "-IPMiters", "--IPMiters",\
|
||||
"Maximum number of IPM iterations");
|
||||
args.AddOption(&ref_levels, "-r", "--mesh_refinement", \
|
||||
"Mesh Refinement");
|
||||
|
||||
|
||||
args.Parse();
|
||||
if(!args.Good())
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
return 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
if( iAmRoot )
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
}
|
||||
|
||||
// Create an instance of the nlp
|
||||
ExContactBlockTL * contact = new ExContactBlockTL(ref_levels);
|
||||
int ndofs = contact->GetDimD();
|
||||
int nconstraints = contact->GetDimS();
|
||||
std::ofstream problemDimStream;
|
||||
problemDimStream.open("problemDim.dat", ios::out | ios::trunc);
|
||||
problemDimStream << ndofs << endl;
|
||||
problemDimStream.close();
|
||||
std::ofstream problemDimConstraintsStream;
|
||||
problemDimConstraintsStream.open("problemDimConstraints.dat", ios::out | ios::trunc);
|
||||
problemDimConstraintsStream << nconstraints << endl;
|
||||
problemDimConstraintsStream.close();
|
||||
|
||||
// set up a QP-problem
|
||||
// E(d) = 1 / 2 d^T K d + f^T d
|
||||
// g(d) = J d + g0
|
||||
// where K, J, f and g0 are evaluated at d0 (a valid configuration)
|
||||
|
||||
// to do: seems more appropriate to evaluate at a valid configuration...
|
||||
// that is one where the Dirichlet conditions hold... need to pull
|
||||
// this data from contactBlockTL...
|
||||
Vector d0(ndofs); d0 = 0.0;
|
||||
Array<int> DirichletDofs = contact->GetDirichletDofs();
|
||||
Array<double> DirichletVals = contact->GetDirichletVals();
|
||||
SparseMatrix *K;
|
||||
Vector f(ndofs); f = 0.0;
|
||||
contact->DdE(d0, f); K = contact->DddE(d0);
|
||||
for(int i = 0; i < DirichletDofs.Size(); i++)
|
||||
{
|
||||
d0(DirichletDofs[i]) = DirichletVals[i];
|
||||
}
|
||||
SparseMatrix *J;
|
||||
Vector g0(nconstraints); g0 = 0.0;
|
||||
J = contact->Ddg(d0); contact->g(d0, g0);
|
||||
Vector temp(nconstraints);
|
||||
J->Mult(d0, temp);
|
||||
g0.Add(-1.0, temp);
|
||||
|
||||
// check which rows of the Jacobian are zero!
|
||||
Vector ei(nconstraints); ei = 0.0;
|
||||
Vector JTei(ndofs); JTei = 0.0;
|
||||
|
||||
double normJTei;
|
||||
|
||||
int reduced_nconstraints = 0; // find actual number of constraints
|
||||
|
||||
|
||||
Array<int> nonZeroRows;
|
||||
for(int i = 0; i < nconstraints; i++)
|
||||
{
|
||||
ei(i) = 1.0;
|
||||
J->MultTranspose(ei, JTei);
|
||||
// nullify contributions from Dirichlet constrined dofs
|
||||
for(int j = 0; j < DirichletDofs.Size(); j++)
|
||||
{
|
||||
JTei(DirichletDofs[j]) = 0.0;
|
||||
}
|
||||
normJTei = sqrt(InnerProduct(JTei, JTei));
|
||||
if (normJTei > 1.e-12)
|
||||
{
|
||||
reduced_nconstraints += 1;
|
||||
nonZeroRows.Append(i);
|
||||
}
|
||||
ei(i) = 0.0;
|
||||
}
|
||||
cout << "number of linearized constraints = " << reduced_nconstraints << endl; // 9 constraints
|
||||
|
||||
// remove zero rows of the gap function Jacobian and corresponding gap function entries
|
||||
SparseMatrix * Jreduced = new SparseMatrix(reduced_nconstraints, ndofs);
|
||||
Vector g0reduced(reduced_nconstraints); g0reduced = 0.0;
|
||||
|
||||
|
||||
for(int i = 0; i < reduced_nconstraints; i++)
|
||||
{
|
||||
Array<int> col_tmp;
|
||||
Vector v_tmp; v_tmp = 0.0;
|
||||
J->GetRow(nonZeroRows[i], col_tmp, v_tmp);
|
||||
|
||||
/* obtain subset of columns of the given nonZero Jacobian row that are not Dirichlet constrained */
|
||||
bool freeDof;
|
||||
Array<int> loc_indicies;
|
||||
for(int j = 0; j < col_tmp.Size(); j++)
|
||||
{
|
||||
freeDof = true;
|
||||
for(int k = 0; k < DirichletDofs.Size(); k++)
|
||||
{
|
||||
if(col_tmp[j] == DirichletDofs[k])
|
||||
{
|
||||
freeDof = false;
|
||||
}
|
||||
}
|
||||
if(freeDof)
|
||||
{
|
||||
loc_indicies.Append(j);
|
||||
}
|
||||
}
|
||||
|
||||
Array<int> col_tmp_reduced(loc_indicies.Size());
|
||||
Vector v_tmp_reduced(loc_indicies.Size());
|
||||
for(int j = 0; j < loc_indicies.Size(); j++)
|
||||
{
|
||||
col_tmp_reduced[j] = col_tmp[loc_indicies[j]];
|
||||
v_tmp_reduced(j) = v_tmp(loc_indicies[j]);
|
||||
}
|
||||
|
||||
Jreduced->SetRow(i, col_tmp_reduced, v_tmp_reduced);
|
||||
g0reduced(i) = g0(nonZeroRows[i]);
|
||||
}
|
||||
|
||||
|
||||
QPContactProblem *QPContact = new QPContactProblem(*K, *Jreduced, f, g0reduced);
|
||||
|
||||
Mesh * mesh1 = new Mesh("meshes/block1.mesh", 1, 1);
|
||||
Mesh * mesh2 = new Mesh("meshes/rotatedblock2.mesh", 1, 1);
|
||||
for(int i = 0; i < ref_levels; i++)
|
||||
{
|
||||
mesh1->UniformRefinement();
|
||||
mesh2->UniformRefinement();
|
||||
}
|
||||
|
||||
int numMeshes = 2;
|
||||
Mesh *meshArray[numMeshes];
|
||||
meshArray[0] = mesh1;
|
||||
meshArray[1] = mesh2;
|
||||
Mesh mesh(meshArray, numMeshes);
|
||||
|
||||
ParMesh pmesh(MPI_COMM_WORLD, mesh);
|
||||
H1_FECollection fec(1, mesh.Dimension());
|
||||
ParFiniteElementSpace fespace(&pmesh, &fec, mesh.Dimension(), Ordering::byVDIM);
|
||||
|
||||
InteriorPointSolver * QPContactOptimizer = new InteriorPointSolver(QPContact, &fespace);
|
||||
QPContactOptimizer->SetTol(1.e-6);
|
||||
QPContactOptimizer->SetLinearSolver(linSolver);
|
||||
QPContactOptimizer->SetMaxIter(50);
|
||||
Vector x0(ndofs); x0 = 0.0;
|
||||
for(int i = 0; i < DirichletDofs.Size(); i++)
|
||||
{
|
||||
x0(DirichletDofs[i]) = DirichletVals[i];
|
||||
}
|
||||
Vector xf(ndofs); xf = 0.0;
|
||||
QPContactOptimizer->Mult(x0, xf);
|
||||
|
||||
double Einitial = QPContact->E(x0);
|
||||
double Efinal = QPContact->E(xf);
|
||||
cout << "Energy objective at initial point = " << Einitial << endl;
|
||||
cout << "Energy objective at QP optimizer = " << Efinal << endl;
|
||||
QPContactOptimizer->GetCGIterNumbers().Print(mfem::out, 20);
|
||||
MFEM_VERIFY(QPContactOptimizer->GetConverged(), "Interior point solver did not converge.");
|
||||
|
||||
|
||||
//Mesh * mesh1 = new Mesh("meshes/block1.mesh", 1, 1);
|
||||
//Mesh * mesh2 = new Mesh("meshes/rotatedblock2.mesh", 1, 1);
|
||||
//for(int i = 0; i < ref_levels; i++)
|
||||
//{
|
||||
// mesh1->UniformRefinement();
|
||||
// mesh2->UniformRefinement();
|
||||
//}
|
||||
//int gdim = mesh1->Dimension();
|
||||
//FiniteElementCollection * fec = new H1_FECollection(1, gdim);
|
||||
//FiniteElementSpace * fespace1 = new FiniteElementSpace(mesh1, fec, gdim, Ordering::byVDIM);
|
||||
//FiniteElementSpace * fespace2 = new FiniteElementSpace(mesh2, fec, gdim, Ordering::byVDIM);
|
||||
//
|
||||
//GridFunction x1_gf(fespace1);
|
||||
//GridFunction x2_gf(fespace2);
|
||||
|
||||
//int ndof1 = fespace1->GetTrueVSize();
|
||||
//int ndof2 = fespace2->GetTrueVSize();
|
||||
//int ndof = ndof1 + ndof2;
|
||||
//for(int i = 0; i < ndof1; i++)
|
||||
//{
|
||||
// x1_gf(i) = xf(i);
|
||||
//}
|
||||
//for(int i = ndof1; i < ndof; i++)
|
||||
//{
|
||||
// x2_gf(i - ndof1) = xf(i);
|
||||
//}
|
||||
|
||||
//mesh1->SetNodalFESpace(fespace1);
|
||||
//mesh2->SetNodalFESpace(fespace2);
|
||||
//GridFunction *nodes1 = mesh1->GetNodes();
|
||||
//GridFunction *nodes2 = mesh2->GetNodes();
|
||||
|
||||
//{
|
||||
// *nodes1 += x1_gf;
|
||||
// *nodes2 += x2_gf;
|
||||
//}
|
||||
//
|
||||
|
||||
//ParaViewDataCollection paraview_dc1("QPContactBody1", mesh1);
|
||||
//paraview_dc1.SetPrefixPath("ParaView");
|
||||
//paraview_dc1.SetLevelsOfDetail(1);
|
||||
//paraview_dc1.SetDataFormat(VTKFormat::BINARY);
|
||||
//paraview_dc1.SetHighOrderOutput(true);
|
||||
//paraview_dc1.SetCycle(0);
|
||||
//paraview_dc1.SetTime(0.0);
|
||||
//paraview_dc1.RegisterField("Body1", &x1_gf);
|
||||
//paraview_dc1.Save();
|
||||
//
|
||||
//ParaViewDataCollection paraview_dc2("QPContactBody2", mesh2);
|
||||
//paraview_dc2.SetPrefixPath("ParaView");
|
||||
//paraview_dc2.SetLevelsOfDetail(1);
|
||||
//paraview_dc2.SetDataFormat(VTKFormat::BINARY);
|
||||
//paraview_dc2.SetHighOrderOutput(true);
|
||||
//paraview_dc2.SetCycle(0);
|
||||
//paraview_dc2.SetTime(0.0);
|
||||
//paraview_dc2.RegisterField("Body2", &x2_gf);
|
||||
//paraview_dc2.Save();
|
||||
|
||||
//delete fespace1;
|
||||
//delete fespace2;
|
||||
//delete fec;
|
||||
//delete mesh1;
|
||||
//delete mesh2;
|
||||
|
||||
delete QPContact;
|
||||
delete QPContactOptimizer;
|
||||
|
||||
delete K;
|
||||
delete J;
|
||||
delete Jreduced;
|
||||
delete contact;
|
||||
return 0;
|
||||
}
|
||||
@@ -1,36 +0,0 @@
|
||||
# Use the MFEM build directory
|
||||
MFEM_DIR ?= ../..
|
||||
MFEM_BUILD_DIR ?= ../..
|
||||
SRC = ./
|
||||
|
||||
CONFIG_MK = $(MFEM_BUILD_DIR)/config/config.mk
|
||||
|
||||
MFEM_LIB_FILE = mfem_is_not_built
|
||||
-include $(CONFIG_MK)
|
||||
|
||||
# Remove built-in rule
|
||||
#%: %.cpp
|
||||
|
||||
exQPContactBlockTL: exQPContactBlockTL.o problems.o IPsolver.o $(MFEM_LIB_FILE)
|
||||
$(MFEM_CXX) $(MFEM_FLAGS) exQPContactBlockTL.o problems.o IPsolver.o -o $@ $(MFEM_LIBS)
|
||||
|
||||
|
||||
|
||||
exQPContactBlockTL.o: exQPContactBlockTL.cpp $(CONFIG_MK)
|
||||
$(MFEM_CXX) $(MFEM_FLAGS) -c $<
|
||||
|
||||
problems.o: problems.cpp $(CONFIG_MK)
|
||||
$(MFEM_CXX) $(MFEM_FLAGS) -c $<
|
||||
|
||||
IPsolver.o: IPsolver.cpp $(CONFIG_MK)
|
||||
$(MFEM_CXX) $(MFEM_FLAGS) -c $<
|
||||
|
||||
# Generate an error message if the MFEM library is not built and exit
|
||||
$(MFEM_LIB_FILE):
|
||||
$(error The MFEM library is not built)
|
||||
|
||||
.PHONY: clean
|
||||
clean:
|
||||
rm -f *.o exQPContactBlockTL
|
||||
|
||||
|
||||
@@ -1,103 +0,0 @@
|
||||
MFEM mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see mesh/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
#
|
||||
|
||||
dimension
|
||||
3
|
||||
|
||||
elements
|
||||
9
|
||||
1 5 0 1 3 2 8 9 11 10
|
||||
1 5 2 3 5 4 10 11 13 12
|
||||
1 5 4 5 7 6 12 13 15 14
|
||||
1 5 8 9 11 10 16 17 19 18
|
||||
1 5 10 11 13 12 18 19 21 20
|
||||
1 5 12 13 15 14 20 21 23 22
|
||||
1 5 16 17 19 18 24 25 27 26
|
||||
1 5 18 19 21 20 26 27 29 28
|
||||
1 5 20 21 23 22 28 29 31 30
|
||||
|
||||
|
||||
|
||||
# 0 nothing
|
||||
# 1 dirichlet bc
|
||||
# 2 contact
|
||||
boundary
|
||||
30
|
||||
1 3 1 0 2 3
|
||||
1 3 3 2 4 5
|
||||
1 3 5 4 6 7
|
||||
1 3 24 25 27 26
|
||||
1 3 26 27 29 28
|
||||
1 3 28 29 31 30
|
||||
2 3 2 0 8 10
|
||||
2 3 4 2 10 12
|
||||
2 3 6 4 12 14
|
||||
2 3 10 8 16 18
|
||||
2 3 12 10 18 20
|
||||
2 3 14 12 20 22
|
||||
2 3 18 16 24 26
|
||||
2 3 20 18 26 28
|
||||
2 3 22 20 28 30
|
||||
3 3 1 3 11 9
|
||||
3 3 3 5 13 11
|
||||
3 3 5 7 15 13
|
||||
3 3 9 11 19 17
|
||||
3 3 11 13 21 19
|
||||
3 3 13 15 23 21
|
||||
3 3 17 19 27 25
|
||||
3 3 19 21 29 27
|
||||
3 3 21 23 31 29
|
||||
1 3 8 0 1 9
|
||||
1 3 16 8 9 17
|
||||
1 3 24 16 17 25
|
||||
1 3 6 14 15 7
|
||||
1 3 14 22 23 15
|
||||
1 3 22 30 31 23
|
||||
|
||||
|
||||
vertices
|
||||
32
|
||||
3
|
||||
-1.0000 0 0
|
||||
0 0 0
|
||||
-1.0000 0.3000 0
|
||||
0 0.3000 0
|
||||
-1.0000 0.6500 0
|
||||
0 0.6500 0
|
||||
-1.0000 1.0000 0
|
||||
0 1.0000 0
|
||||
-1.0000 0 0.3000
|
||||
0 0 0.3000
|
||||
-1.0000 0.3000 0.3500
|
||||
0 0.3000 0.3500
|
||||
-1.0000 0.6500 0.3000
|
||||
0 0.6500 0.3000
|
||||
-1.0000 1.0000 0.3000
|
||||
0 1.0000 0.3000
|
||||
-1.0000 0 0.6500
|
||||
0 0 0.6500
|
||||
-1.0000 0.3000 0.6500
|
||||
0 0.3000 0.6500
|
||||
-1.0000 0.6500 0.6500
|
||||
0 0.6500 0.6500
|
||||
-1.0000 1.0000 0.6500
|
||||
0 1.0000 0.6500
|
||||
-1.0000 0 1.0000
|
||||
0 0 1.0000
|
||||
-1.0000 0.3000 1.0000
|
||||
0 0.3000 1.0000
|
||||
-1.0000 0.6500 1.0000
|
||||
0 0.6500 1.0000
|
||||
-1.0000 1.0000 1.0000
|
||||
0 1.0000 1.0000
|
||||
@@ -1,70 +0,0 @@
|
||||
|
||||
MFEM mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see mesh/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
#
|
||||
|
||||
dimension
|
||||
3
|
||||
|
||||
# 1 nothing
|
||||
elements
|
||||
4
|
||||
1 5 0 1 3 2 6 7 9 8
|
||||
1 5 2 3 5 4 8 9 11 10
|
||||
1 5 6 7 9 8 12 13 15 14
|
||||
1 5 8 9 11 10 14 15 17 16
|
||||
|
||||
# 0 nothing
|
||||
# 1 dirichlet bc
|
||||
# 2 contact
|
||||
boundary
|
||||
16
|
||||
1 3 1 0 2 3
|
||||
1 3 3 2 4 5
|
||||
1 3 12 13 15 14
|
||||
1 3 14 15 17 16
|
||||
3 3 2 0 6 8
|
||||
3 3 4 2 8 10
|
||||
3 3 8 6 12 14
|
||||
3 3 10 8 14 16
|
||||
2 3 1 3 9 7
|
||||
2 3 3 5 11 9
|
||||
2 3 7 9 15 13
|
||||
2 3 9 11 17 15
|
||||
1 3 6 0 1 7
|
||||
1 3 12 6 7 13
|
||||
1 3 4 10 11 5
|
||||
1 3 10 16 17 11
|
||||
|
||||
vertices
|
||||
18
|
||||
3
|
||||
|
||||
0.000000000000 0.145770950245 0.443895630208
|
||||
0.507100000000 0.145770950245 0.443895630208
|
||||
0.000000000000 0.350937660019 0.294833290227
|
||||
0.507100000000 0.350937660019 0.294833290227
|
||||
0.000000000000 0.556104369792 0.145770950245
|
||||
0.507100000000 0.556104369792 0.145770950245
|
||||
0.000000000000 0.294833290227 0.649062339981
|
||||
0.507100000000 0.294833290227 0.649062339981
|
||||
0.000000000000 0.500000000000 0.500000000000
|
||||
0.507100000000 0.500000000000 0.500000000000
|
||||
0.000000000000 0.705166709773 0.350937660019
|
||||
0.507100000000 0.705166709773 0.350937660019
|
||||
0.000000000000 0.443895630208 0.854229049755
|
||||
0.507100000000 0.443895630208 0.854229049755
|
||||
0.000000000000 0.649062339981 0.705166709773
|
||||
0.507100000000 0.649062339981 0.705166709773
|
||||
0.000000000000 0.854229049755 0.556104369792
|
||||
0.507100000000 0.854229049755 0.556104369792
|
||||
@@ -1,897 +0,0 @@
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
void BasisEval(const Vector xi, Vector &N, DenseMatrix &dNdxi) // dNdxi is 2*4
|
||||
{
|
||||
N[0] = 0.25*(1-xi[0])*(1-xi[1]);
|
||||
N[1] = 0.25*(1+xi[0])*(1-xi[1]);
|
||||
N[2] = 0.25*(1+xi[0])*(1+xi[1]);
|
||||
N[3] = 0.25*(1-xi[0])*(1+xi[1]);
|
||||
|
||||
dNdxi(0,0) = 0.25*(-1+xi[1]);
|
||||
dNdxi(0,1) = 0.25*(1-xi[1]);
|
||||
dNdxi(0,2) = 0.25*(1+xi[1]);
|
||||
dNdxi(0,3) = 0.25*(-1-xi[1]);
|
||||
dNdxi(1,0) = 0.25*(-1+xi[0]);
|
||||
dNdxi(1,1) = 0.25*(-1-xi[0]);
|
||||
dNdxi(1,2) = 0.25*(1+xi[0]);
|
||||
dNdxi(1,3) = 0.25*(1-xi[0]);
|
||||
}
|
||||
|
||||
|
||||
void BasisEvalDerivs(const Vector xi, Vector& N, DenseMatrix& dNdxi,
|
||||
DenseMatrix& dN2dxi)
|
||||
{
|
||||
N[0] = 0.25*(1-xi[0])*(1-xi[1]);
|
||||
N[1] = 0.25*(1+xi[0])*(1-xi[1]);
|
||||
N[2] = 0.25*(1+xi[0])*(1+xi[1]);
|
||||
N[3] = 0.25*(1-xi[0])*(1+xi[1]);
|
||||
|
||||
dNdxi.SetSize(2,4); dNdxi = 0.0;
|
||||
dN2dxi.SetSize(3,4);
|
||||
dN2dxi = 0.0; // first row dxi2, second detadxi, third deta2
|
||||
|
||||
dNdxi(0,0) = 0.25*(-1+xi[1]); dNdxi(0,1) = 0.25*(1-xi[1]);
|
||||
dNdxi(0,2) = 0.25*(1+xi[1]); dNdxi(0,3) = 0.25*(-1-xi[1]);
|
||||
dNdxi(1,0) = 0.25*(-1+xi[0]); dNdxi(1,1) = 0.25*(-1-xi[0]);
|
||||
dNdxi(1,2) = 0.25*(1+xi[0]); dNdxi(1,3) = 0.25*(1-xi[0]);
|
||||
|
||||
dN2dxi(1,0) = 0.25; dN2dxi(1,1) = -0.25; dN2dxi(1,2) = 0.25;
|
||||
dN2dxi(1,3) = -0.25;
|
||||
}
|
||||
|
||||
// returns the vector and matrix form of the shape functions and its derivative
|
||||
void BasisVectorDerivs(const Vector xi, DenseMatrix& N, DenseMatrix& dNdxi,
|
||||
DenseMatrix& ddNdxi)
|
||||
{
|
||||
N.SetSize(3,12); N = 0.0;
|
||||
N(0,0) = 0.25*(1-xi[0])*(1-xi[1]); N(0,3) = 0.25*(1+xi[0])*(1-xi[1]);
|
||||
N(0,6) = 0.25*(1+xi[0])*(1+xi[1]); N(0,9) = 0.25*(1-xi[0])*(1+xi[1]);
|
||||
|
||||
N(1,1) = 0.25*(1-xi[0])*(1-xi[1]); N(1,4) = 0.25*(1+xi[0])*(1-xi[1]);
|
||||
N(1,7) = 0.25*(1+xi[0])*(1+xi[1]); N(1,10) = 0.25*(1-xi[0])*(1+xi[1]);
|
||||
|
||||
N(2,2) = 0.25*(1-xi[0])*(1-xi[1]); N(2,5) = 0.25*(1+xi[0])*(1-xi[1]);
|
||||
N(2,8) = 0.25*(1+xi[0])*(1+xi[1]); N(2,11) = 0.25*(1-xi[0])*(1+xi[1]);
|
||||
|
||||
dNdxi.SetSize(3*2, 3*4); dNdxi = 0.0;
|
||||
dNdxi(0,0) = 0.25*(-1+xi[1]); dNdxi(0,3) = 0.25*(1-xi[1]);
|
||||
dNdxi(0,6) = 0.25*(1+xi[1]); dNdxi(0,9) = 0.25*(-1-xi[1]);
|
||||
dNdxi(1,1) = 0.25*(-1+xi[1]); dNdxi(1,4) = 0.25*(1-xi[1]);
|
||||
dNdxi(1,7) = 0.25*(1+xi[1]); dNdxi(1,10) = 0.25*(-1-xi[1]);
|
||||
dNdxi(2,2) = 0.25*(-1+xi[1]); dNdxi(2,5) = 0.25*(1-xi[1]);
|
||||
dNdxi(2,8) = 0.25*(1+xi[1]); dNdxi(2,11) = 0.25*(-1-xi[1]);
|
||||
|
||||
dNdxi(3,0) = 0.25*(-1+xi[0]); dNdxi(3,3) = 0.25*(-1-xi[0]);
|
||||
dNdxi(3,6) = 0.25*(1+xi[0]); dNdxi(3,9) = 0.25*(1-xi[0]);
|
||||
dNdxi(4,1) = 0.25*(-1+xi[0]); dNdxi(4,4) = 0.25*(-1-xi[0]);
|
||||
dNdxi(4,7) = 0.25*(1+xi[0]); dNdxi(4,10) = 0.25*(1-xi[0]);
|
||||
dNdxi(5,2) = 0.25*(-1+xi[0]); dNdxi(5,5) = 0.25*(-1-xi[0]);
|
||||
dNdxi(5,8) = 0.25*(1+xi[0]); dNdxi(5,11) = 0.25*(1-xi[0]);
|
||||
|
||||
ddNdxi.SetSize(3*4, 3*4); ddNdxi = 0.0;
|
||||
ddNdxi(3,0) = 0.25; ddNdxi(3,3) = -0.25;
|
||||
ddNdxi(3,6) = 0.25; ddNdxi(3,9) = -0.25;
|
||||
ddNdxi(4,1) = 0.25; ddNdxi(4,4) = -0.25;
|
||||
ddNdxi(4,7) = 0.25; ddNdxi(4,10) = -0.25;
|
||||
ddNdxi(5,2) = 0.25; ddNdxi(5,5) = -0.25;
|
||||
ddNdxi(5,8) = 0.25; ddNdxi(5,11) = -0.25;
|
||||
|
||||
ddNdxi(6,0) = 0.25; ddNdxi(6,3) = -0.25;
|
||||
ddNdxi(6,6) = 0.25; ddNdxi(6,9) = -0.25;
|
||||
ddNdxi(7,1) = 0.25; ddNdxi(7,4) = -0.25;
|
||||
ddNdxi(7,7) = 0.25; ddNdxi(7,10) = -0.25;
|
||||
ddNdxi(8,2) = 0.25; ddNdxi(8,5) = -0.25;
|
||||
ddNdxi(8,8) = 0.25; ddNdxi(8,11) = -0.25;
|
||||
}
|
||||
|
||||
|
||||
void cross(const Vector a, const Vector b, Vector& c)
|
||||
{
|
||||
assert(a.Size()==3);
|
||||
c.SetSize(3);
|
||||
c[0] = a[1]*b[2] - a[2]*b[1];
|
||||
c[1] = -a[0]*b[2] + b[0]*a[2];
|
||||
c[2] = a[0]*b[1] - a[1]*b[0];
|
||||
|
||||
}
|
||||
// a outer b
|
||||
void outer(const Vector a, const Vector b, DenseMatrix& c)
|
||||
{
|
||||
int m = a.Size();
|
||||
int n = b.Size();
|
||||
assert(c.Height()==m);
|
||||
assert(c.Width() ==n);
|
||||
for (int i=0; i<m; i++)
|
||||
{
|
||||
for (int j=0; j<n; j++)
|
||||
{
|
||||
c(i,j) = a[i]*b[j];
|
||||
}
|
||||
}
|
||||
}
|
||||
// dphidxi 2*4
|
||||
// coords 4*3
|
||||
void ComputeNormal(const DenseMatrix& dphidxi, const DenseMatrix& coords,
|
||||
Vector& normal, double& nnorm)
|
||||
{
|
||||
|
||||
DenseMatrix dxdxi(2,3);
|
||||
Mult(dphidxi, coords, dxdxi);
|
||||
Vector dxdxi1(3);
|
||||
Vector dxdxi2(3);
|
||||
|
||||
dxdxi.GetRow(0,dxdxi1);
|
||||
dxdxi.GetRow(1,dxdxi2);
|
||||
|
||||
cross(dxdxi1, dxdxi2, normal); // is there a cross product? no
|
||||
// VectorCrossProductCoefficient::Eval has hard-coded cross product
|
||||
nnorm = normal.Norml2( );
|
||||
normal /= nnorm;
|
||||
}
|
||||
|
||||
void SlaveToMaster(const DenseMatrix& m_coords, const Vector& s_x, Vector& xi)
|
||||
{
|
||||
bool converged = false;
|
||||
bool pt_on_elem = false;
|
||||
int dim = 3;
|
||||
xi.SetSize(dim-1);
|
||||
xi = 0.0;
|
||||
int max_iter = 15;
|
||||
double off_el_xi = 1e-2;
|
||||
double proj_newton_tol = 1e-13;
|
||||
double proj_max_gap = 0.5;
|
||||
Vector gap_v(dim);
|
||||
// warm start from linear solution
|
||||
|
||||
for (int it=0; it<max_iter; it++)
|
||||
{
|
||||
//cout<<it<<endl;
|
||||
Vector m_N(4);
|
||||
m_N = 0.;
|
||||
DenseMatrix m_dN(2,4);
|
||||
m_dN = 0.;
|
||||
DenseMatrix m_dN2(3,4);
|
||||
m_dN2 = 0.;
|
||||
BasisEvalDerivs(xi, m_N, m_dN, m_dN2);
|
||||
|
||||
Vector x_c(dim);
|
||||
m_coords.MultTranspose(m_N, x_c);
|
||||
|
||||
gap_v = s_x;
|
||||
gap_v -= x_c;
|
||||
|
||||
DenseMatrix m_dx(2,3);
|
||||
m_dx = 0.;
|
||||
Mult(m_dN, m_coords, m_dx);
|
||||
|
||||
Vector r(dim-1);
|
||||
r = 0.0;
|
||||
m_dx.Mult(gap_v, r);
|
||||
|
||||
if (r.Normlinf() < proj_newton_tol)
|
||||
{
|
||||
converged = true;
|
||||
break;
|
||||
}
|
||||
|
||||
DenseMatrix drdxi(dim-1,dim-1);
|
||||
drdxi = 0.;
|
||||
MultABt(m_dx, m_dx, drdxi); // m_dx * m_dx.T
|
||||
drdxi *= -1.0;
|
||||
|
||||
DenseMatrix m_dx2(3,3); m_dx2 = 0.0;
|
||||
Mult(m_dN2,m_coords, m_dx2);
|
||||
|
||||
//m_d2x = m_dN(:,:,2) * m_elem_coords(1:4,:); //m_dN(:,:,2) is 3*4
|
||||
for (int d=0; d<3; d++)
|
||||
{
|
||||
DenseMatrix Mtemp(2,2); Mtemp = 0.0;
|
||||
Mtemp(0,0) = m_dx2(0,d); Mtemp(0,1) = m_dx2(1,d);
|
||||
Mtemp(1,0) = m_dx2(1,d); Mtemp(1,1) = m_dx2(2,d);
|
||||
|
||||
drdxi.Add(gap_v[d], Mtemp);
|
||||
}
|
||||
|
||||
//cond_num = rcond(drdxi); condition number?
|
||||
//drdxi.TestInversion();
|
||||
DenseMatrixInverse drdxi_inv(drdxi);
|
||||
Vector xi_tmp(dim-1);
|
||||
|
||||
drdxi_inv.Mult(r,xi_tmp);
|
||||
xi -= xi_tmp;
|
||||
}
|
||||
if (!converged)
|
||||
{
|
||||
xi = 0.0;
|
||||
}
|
||||
off_el_xi += 1 ; // tolerance of offset of xi outside [-1,1]
|
||||
|
||||
//cout<<gap_v.Norml2()<<" " <<xi.Normlinf()<<endl;
|
||||
//
|
||||
// Discuss with Frank... what is happening here
|
||||
if (gap_v.Norml2() < proj_max_gap && xi.Normlinf() <= off_el_xi)
|
||||
{
|
||||
pt_on_elem = true;
|
||||
}
|
||||
|
||||
if (pt_on_elem)
|
||||
{
|
||||
//cout << "convergence of node to segment projection? " << converged << endl;
|
||||
//for(int i = 0; i < 2; i++)
|
||||
//{
|
||||
// cout << "xi_" << i << " = " << xi(i) << endl;
|
||||
//}
|
||||
}
|
||||
MFEM_VERIFY(pt_on_elem == true, "xi went out of bounds");
|
||||
MFEM_VERIFY(converged == true, "projection didn't converge");
|
||||
}
|
||||
|
||||
|
||||
|
||||
// m_coords is expected to be 4 * 3
|
||||
void ComputeGapJacobian(const Vector x_s, const Vector xi,
|
||||
const DenseMatrix m_coords,
|
||||
double& gap, Vector& normal, Vector& dgdxm, Vector& dgdxs)
|
||||
{
|
||||
Vector m_N(4);
|
||||
DenseMatrix m_dN(2,4);
|
||||
DenseMatrix m_dN2(3,4);
|
||||
BasisEvalDerivs(xi, m_N, m_dN, m_dN2);
|
||||
|
||||
Vector x_c(3);
|
||||
m_coords.MultTranspose(m_N, x_c);
|
||||
|
||||
Vector gap_v(3); gap_v = 0.0;
|
||||
gap_v = x_s;
|
||||
gap_v -= x_c;
|
||||
|
||||
DenseMatrix m_dx(2,3);
|
||||
Mult(m_dN, m_coords, m_dx);
|
||||
|
||||
double nnorm = 0;
|
||||
ComputeNormal(m_dN, m_coords, normal, nnorm);
|
||||
|
||||
gap = gap_v * normal; // gap function value, dot product between vectors
|
||||
|
||||
//dr_dx = zeros(2,4,3); % nsegment, nodes in quad, ndim
|
||||
|
||||
DenseMatrix dr_dx_res1(4,3); dr_dx_res1 = 0.;
|
||||
DenseMatrix dr_dx_res2(4,3); dr_dx_res2 = 0.;
|
||||
|
||||
Vector m_dxrow1(3);
|
||||
m_dx.GetRow(0, m_dxrow1);
|
||||
outer(m_N, m_dxrow1, dr_dx_res1);// 4*1 times 1*3
|
||||
dr_dx_res1 *= -1.0;
|
||||
|
||||
Vector m_dxrow2(3);
|
||||
m_dx.GetRow(1, m_dxrow2);
|
||||
outer(m_N, m_dxrow2, dr_dx_res2);// 4*1 times 1*3
|
||||
dr_dx_res2 *= -1.0;
|
||||
|
||||
Vector m_dNrow1(4); m_dN.GetRow(0, m_dNrow1);
|
||||
Vector m_dNrow2(4); m_dN.GetRow(1, m_dNrow2);
|
||||
|
||||
DenseMatrix dr_dx_res1_tmp(4,3); dr_dx_res1_tmp = 0.;
|
||||
DenseMatrix dr_dx_res2_tmp(4,3); dr_dx_res2_tmp = 0.;
|
||||
outer(m_dNrow1, gap_v, dr_dx_res1_tmp);// 4*1 times 1*3
|
||||
outer(m_dNrow2, gap_v, dr_dx_res2_tmp);// 4*1 times 1*3
|
||||
|
||||
dr_dx_res1 += dr_dx_res1_tmp; // outer product in vector?
|
||||
dr_dx_res2 += dr_dx_res2_tmp;
|
||||
|
||||
|
||||
DenseMatrix K_dxidx1(2,2); // 2*2
|
||||
K_dxidx1 = 0.;
|
||||
MultABt(m_dx, m_dx, K_dxidx1); // m_dx * m_dx.T
|
||||
|
||||
Vector v_dxidx2(4);
|
||||
m_coords.Mult(gap_v, v_dxidx2); // m_coords * gap_v; // 4*3 * 3 = 4
|
||||
|
||||
DenseMatrix K_dxidx2(2,2); K_dxidx2 = 0.0;
|
||||
|
||||
Vector m_dN2row1(4); m_dN2.GetRow(0, m_dN2row1);
|
||||
Vector m_dN2row2(4); m_dN2.GetRow(1, m_dN2row2);
|
||||
Vector m_dN2row3(4); m_dN2.GetRow(2, m_dN2row3);
|
||||
// how to get 2nd order? multidimensional matrix?
|
||||
K_dxidx2(0,0) = m_dN2row1 * v_dxidx2; // how would 4*1 * 1*4 be computed?
|
||||
K_dxidx2(0,1) = m_dN2row2 * v_dxidx2;
|
||||
K_dxidx2(1,0) = m_dN2row2 * v_dxidx2;
|
||||
K_dxidx2(1,1) = m_dN2row3 * v_dxidx2;
|
||||
|
||||
DenseMatrix K_dxidx(2,2);
|
||||
K_dxidx -= K_dxidx1;
|
||||
K_dxidx += K_dxidx2;
|
||||
|
||||
// resize the vectors and matrices
|
||||
Vector dxidx(24); dxidx = 0.0;
|
||||
Vector drdx_r(24); drdx_r = 0.0;
|
||||
|
||||
for (int i=0; i<4; i++)
|
||||
{
|
||||
for (int j=0; j<3; j++)
|
||||
{
|
||||
drdx_r[4*j+i] = dr_dx_res1(i,j);
|
||||
drdx_r[4*j+i+12] = dr_dx_res2(i,j);
|
||||
|
||||
}
|
||||
}
|
||||
//drdx_r(1:4*3,1) = reshape(dr_dx_res(:,:,1),4*3,1);
|
||||
//drdx_r(4*3+1:2*4*3,1) = reshape(dr_dx_res(:,:,2),4*3,1);
|
||||
DenseMatrix drdx_K(24,24); drdx_K = 0.;
|
||||
for (int i =0; i<12; i++)
|
||||
{
|
||||
drdx_K(i,i) = K_dxidx(0,0);
|
||||
drdx_K(i,12+i) = K_dxidx(0,1);
|
||||
drdx_K(12+i,i) = K_dxidx(1,0);
|
||||
drdx_K(12+i,12+i) = K_dxidx(1,1);
|
||||
}
|
||||
|
||||
DenseMatrixInverse drdxK_inv(drdx_K);
|
||||
drdxK_inv.Mult(drdx_r,dxidx);
|
||||
// LinearSolve (drdx_K,drdx_r, dxidx) ; //???
|
||||
dxidx *= -1.0;
|
||||
|
||||
|
||||
|
||||
Vector drdxs_r(6);
|
||||
drdxs_r[0] = m_dx(0,0); drdxs_r[1] = m_dx(0,1); drdxs_r[2] = m_dx(0,2);
|
||||
drdxs_r[3] = m_dx(1,0); drdxs_r[4] = m_dx(1,1); drdxs_r[5] = m_dx(1,2);
|
||||
|
||||
DenseMatrix drdxs_K(6,6); drdxs_K = 0.;
|
||||
for (int i=0; i<3; i++)
|
||||
{
|
||||
drdxs_K(i,i) = K_dxidx(0,0);
|
||||
drdxs_K(i,3+i) = K_dxidx(0,1);
|
||||
drdxs_K(i+3,i) = K_dxidx(1,0);
|
||||
drdxs_K(i+3,i+3) = K_dxidx(1,1);
|
||||
}
|
||||
|
||||
Vector dxidxs(6); dxidxs = 0.0;
|
||||
DenseMatrixInverse drdxsK_inv(drdxs_K);
|
||||
drdxsK_inv.Mult(drdxs_r,dxidxs);
|
||||
dxidxs *= -1.0;
|
||||
//dxidxs = -drdxs_K\drdxs_r;
|
||||
|
||||
//dxidx = reshape(dxidx, 4,3,2); dxidxs = reshape(dxidxs, 1,3,2);
|
||||
|
||||
dgdxm.SetSize(12); dgdxm = 0.;
|
||||
DenseMatrix dgdxm_tmp(4,3);
|
||||
outer(m_N, normal,dgdxm_tmp);
|
||||
for (int i=0; i<4; i++)
|
||||
{
|
||||
for (int j=0; j<3; j++)
|
||||
{
|
||||
dgdxm[3*i+j] = -dgdxm_tmp(i,j);
|
||||
}
|
||||
}
|
||||
//dxidx_M = -m_dN(1:2,:,1) * (m_coords(1:4,:)*normal'); % this turns out to be 0
|
||||
|
||||
dgdxs.SetSize(3);
|
||||
dgdxs += normal;
|
||||
//dgdxs = dgdxs + dxidx_M(1) * dxidxs(:,:,1) + dxidx_M(2) * dxidxs(:,:,2);
|
||||
};
|
||||
|
||||
void ComputeGapHessian(const Vector x_s, const Vector xi,
|
||||
const DenseMatrix m_coords,
|
||||
DenseMatrix& dg2dx)
|
||||
{
|
||||
Vector m_N(4);
|
||||
DenseMatrix m_dN(2,4);
|
||||
DenseMatrix m_dN2(3,4);
|
||||
BasisEvalDerivs(xi, m_N, m_dN, m_dN2);
|
||||
|
||||
int dim = 3;
|
||||
int num_dofs1 = dim;
|
||||
int num_dofs2 = 4*dim;
|
||||
int num_dofs = num_dofs1 + num_dofs2;
|
||||
dg2dx.SetSize(num_dofs,num_dofs); dg2dx = 0.0;
|
||||
|
||||
Vector x_c(3);
|
||||
m_coords.MultTranspose(m_N,x_c);
|
||||
|
||||
Vector gap_v(3); gap_v = 0.0;
|
||||
gap_v = x_s;
|
||||
gap_v -= x_c;
|
||||
|
||||
DenseMatrix m_dx(2,3);
|
||||
Mult(m_dN, m_coords, m_dx);
|
||||
|
||||
DenseMatrix m_dx2(3,3); m_dx2 = 0.0;
|
||||
Mult(m_dN2,m_coords, m_dx2);
|
||||
double nnorm = 0.0;
|
||||
Vector normal(3); normal = 0.0;
|
||||
ComputeNormal(m_dN, m_coords, normal, nnorm);
|
||||
|
||||
double gap = gap_v * normal; // gap function value, dot product between vectors
|
||||
|
||||
DenseMatrix M(2,2); M = 0.0;
|
||||
MultABt(m_dx, m_dx, M);
|
||||
|
||||
DenseMatrix f(2, num_dofs2); f = 0.0;
|
||||
|
||||
for (int d=0; d<3; d++)
|
||||
{
|
||||
DenseMatrix Mtemp(2,2); Mtemp = 0.0;
|
||||
Mtemp(0,0) = m_dx2(0,d); Mtemp(0,1) = m_dx2(1,d);
|
||||
Mtemp(1,0) = m_dx2(1,d); Mtemp(1,1) = m_dx2(2,d);
|
||||
|
||||
M.Add(-gap_v[d], Mtemp);
|
||||
|
||||
Vector m_dxcol(2); m_dx.GetColumn(d, m_dxcol);
|
||||
DenseMatrix ftmp(2,4);
|
||||
outer(m_dxcol, m_N, ftmp);
|
||||
ftmp *= -1;
|
||||
ftmp.Add( gap_v[d], m_dN); // 2*4
|
||||
|
||||
for (int j=0; j<4; j++)
|
||||
{
|
||||
assert(d+3*j<num_dofs2);
|
||||
f(0,d+j*3) = ftmp(0,j);
|
||||
f(1,d+j*3) = ftmp(1,j);
|
||||
}
|
||||
}
|
||||
//fprintf('hess dxidxm\n');
|
||||
DenseMatrixInverse Minv(M);
|
||||
DenseMatrix dxidxm(2,num_dofs2); dxidxm = 0.0;
|
||||
Minv.Mult(f, dxidxm);
|
||||
//LinearSolve??
|
||||
//dxidxm = M\f;
|
||||
|
||||
DenseMatrix nde2(2,2); nde2 = 0.0;
|
||||
DenseMatrix Nndx2(2,num_dofs2); Nndx2 = 0.0;
|
||||
|
||||
for (int d=0; d<3; d++)
|
||||
{
|
||||
DenseMatrix ndetmp(2,2); ndetmp = 0.0;
|
||||
ndetmp(0,0) = normal(d)*m_dx2(0,d); ndetmp(0,1) = normal(d)*m_dx2(1,d);
|
||||
ndetmp(1,0) = normal(d)*m_dx2(1,d); ndetmp(1,1) = normal(d)*m_dx2(2,d);
|
||||
|
||||
nde2 += ndetmp;
|
||||
|
||||
for (int j=0; j<4; j++)
|
||||
{
|
||||
assert(d+3*j<num_dofs2);
|
||||
Nndx2(0,d+j*3) = normal[d]*m_dN(0,j);
|
||||
Nndx2(1,d+j*3) = normal[d]*m_dN(1,j);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
DenseMatrix Ndn(2,num_dofs2); Ndn = 0.0;
|
||||
Ndn += Nndx2;
|
||||
AddMult(nde2, dxidxm, Ndn);
|
||||
|
||||
|
||||
DenseMatrix M2(2,2); M2 = 0.0;
|
||||
MultABt(m_dx, m_dx, M2);
|
||||
DenseMatrixInverse M2inv(M2);
|
||||
DenseMatrix diag2(2,2); diag2(0,0) = 1.0; diag2(1,1) = 1.0;
|
||||
DenseMatrix m_con(2,2); m_con = 0.0;
|
||||
|
||||
M2inv.Mult(diag2, m_con);
|
||||
|
||||
DenseMatrix dg2dxm(num_dofs2, num_dofs2); dg2dxm = 0.0;
|
||||
|
||||
DenseMatrix dg2dxm_tmp(num_dofs2,2); dg2dxm_tmp = 0.0;
|
||||
MultAtB(Ndn, m_con, dg2dxm_tmp);
|
||||
Mult(dg2dxm_tmp, Ndn, dg2dxm);
|
||||
dg2dxm *= gap;
|
||||
|
||||
DenseMatrix dg2dxm_tmp2(num_dofs2,num_dofs2); dg2dxm_tmp2 = 0.0;
|
||||
MultAtB(Nndx2, dxidxm, dg2dxm_tmp2);
|
||||
dg2dxm.Add(-1.0, dg2dxm_tmp2);
|
||||
|
||||
dg2dxm_tmp = 0.0;
|
||||
MultAtB(dxidxm, nde2, dg2dxm_tmp);
|
||||
|
||||
AddMult_a(-1.0, dg2dxm_tmp, dxidxm, dg2dxm);
|
||||
|
||||
dg2dxm_tmp2 = 0.0;
|
||||
MultAtB(dxidxm, Nndx2, dg2dxm_tmp2);
|
||||
dg2dxm.Add(-1.0, dg2dxm_tmp2);
|
||||
|
||||
Vector v_dxidx2(4);
|
||||
m_coords.Mult(gap_v, v_dxidx2); // m_coords * gap_v; // 4*3 * 3 = 4
|
||||
|
||||
DenseMatrix K_dxidx2(2,2); K_dxidx2 = 0.0;
|
||||
|
||||
Vector m_dN2row1(4); m_dN2.GetRow(0, m_dN2row1);
|
||||
Vector m_dN2row2(4); m_dN2.GetRow(1, m_dN2row2);
|
||||
Vector m_dN2row3(4); m_dN2.GetRow(2, m_dN2row3);
|
||||
K_dxidx2(0,0) = m_dN2row1 * v_dxidx2; // how would 4*1 * 1*4 be computed?
|
||||
K_dxidx2(0,1) = m_dN2row2 * v_dxidx2;
|
||||
K_dxidx2(1,0) = m_dN2row2 * v_dxidx2;
|
||||
K_dxidx2(1,1) = m_dN2row3 * v_dxidx2;
|
||||
|
||||
DenseMatrix K_dxidx(2,2);
|
||||
K_dxidx -= M2;
|
||||
K_dxidx += K_dxidx2;
|
||||
|
||||
Vector drdxs_r(6);
|
||||
drdxs_r[0] = m_dx(0,0); drdxs_r[1] = m_dx(0,1); drdxs_r[2] = m_dx(0,2);
|
||||
drdxs_r[3] = m_dx(1,0); drdxs_r[4] = m_dx(1,1); drdxs_r[5] = m_dx(1,2);
|
||||
|
||||
DenseMatrix drdxs_K(6,6); drdxs_K = 0.;
|
||||
for (int i=0; i<3; i++)
|
||||
{
|
||||
drdxs_K(i,i) = K_dxidx(0,0);
|
||||
drdxs_K(i,3+i) = K_dxidx(0,1);
|
||||
drdxs_K(i+3,i) = K_dxidx(1,0);
|
||||
drdxs_K(i+3,i+3) = K_dxidx(1,1);
|
||||
}
|
||||
Vector dxidxs(6);
|
||||
|
||||
DenseMatrixInverse drdxsK_inv(drdxs_K);
|
||||
drdxsK_inv.Mult(drdxs_r,dxidxs);
|
||||
dxidxs *= -1.0;
|
||||
//dxidxs = -drdxs_K\drdxs_r;
|
||||
|
||||
DenseMatrix dxidxs_m(2,3); dxidxs_m = 0.0;
|
||||
dxidxs_m(0,0) = dxidxs[0]; dxidxs_m(0,1) = dxidxs[1]; dxidxs_m(0,2) = dxidxs[2];
|
||||
dxidxs_m(1,0) = dxidxs[3]; dxidxs_m(1,1) = dxidxs[4]; dxidxs_m(1,2) = dxidxs[5];
|
||||
|
||||
DenseMatrix dtao1dxs(3,3); dtao1dxs = 0.0;
|
||||
DenseMatrix dtao2dxs(3,3); dtao2dxs = 0.0;
|
||||
|
||||
Vector dxidxs_row1(3); dxidxs_row1 = 0.0; Vector dxidxs_row2(3);
|
||||
dxidxs_row2 = 0.0;
|
||||
Vector mdx2_row1(3); mdx2_row1 = 0.0; Vector mdx2_row2(3); mdx2_row2 = 0.0;
|
||||
Vector mdx2_row3(3); mdx2_row3 = 0.0;
|
||||
dxidxs_m.GetRow(0,dxidxs_row1);
|
||||
dxidxs_m.GetRow(1,dxidxs_row2);
|
||||
m_dx2.GetRow(0,mdx2_row1);
|
||||
m_dx2.GetRow(1,mdx2_row2);
|
||||
m_dx2.GetRow(2,mdx2_row3);
|
||||
|
||||
DenseMatrix dtaotmp(3,3); dtaotmp = 0.0;
|
||||
outer(mdx2_row1, dxidxs_row1,dtaotmp);
|
||||
dtao1dxs += dtaotmp; dtaotmp = 0.0;
|
||||
outer(mdx2_row2, dxidxs_row1,dtaotmp);
|
||||
dtao1dxs += dtaotmp; dtaotmp = 0.0;
|
||||
|
||||
outer(mdx2_row2, dxidxs_row2, dtaotmp);
|
||||
dtao2dxs += dtaotmp; dtaotmp = 0.0;
|
||||
outer(mdx2_row3, dxidxs_row2, dtaotmp);
|
||||
dtao2dxs += dtaotmp; dtaotmp = 0.0;
|
||||
|
||||
DenseMatrix dtaodxs(3,3); dtaodxs = 0.0; //tao = tao1 cross tao2
|
||||
|
||||
for (int d=0; d<3; d++)
|
||||
{
|
||||
Vector dtao1dxs_tmp(3); dtao1dxs_tmp = 0.0;
|
||||
dtao1dxs.GetColumn(d,dtao1dxs_tmp);
|
||||
Vector m_dxrow(3); m_dx.GetRow(1, m_dxrow);
|
||||
|
||||
Vector dtaodxs_tmp(3); dtaodxs_tmp = 0.0;
|
||||
cross(dtao1dxs_tmp, m_dxrow, dtaodxs_tmp);
|
||||
|
||||
Vector dtaodxs_tmp2(3); dtaodxs_tmp2 = 0.0;
|
||||
m_dx.GetRow(0, m_dxrow);
|
||||
dtao1dxs_tmp = 0.0; // reuse the same vector for dtao2
|
||||
dtao2dxs.GetColumn(d,dtao1dxs_tmp);
|
||||
cross(m_dxrow, dtao1dxs_tmp, dtaodxs_tmp2);
|
||||
|
||||
dtaodxs_tmp2 += dtaodxs_tmp;
|
||||
dtaodxs.SetCol(d, dtaodxs_tmp2);
|
||||
}
|
||||
|
||||
DenseMatrix dndxs(3,3); dndxs = 0.0; dndxs += dtaodxs; dndxs *= 1.0/nnorm;
|
||||
DenseMatrix dndxs_tmp(3,3); dndxs_tmp = 0.0;
|
||||
outer(normal, normal, dndxs_tmp);
|
||||
AddMult_a(-1/nnorm, dndxs_tmp, dtaodxs, dndxs);
|
||||
|
||||
DenseMatrix dgvdxs(3,3); dgvdxs = 0.0;
|
||||
MultAtB(m_dx, dxidxs_m, dgvdxs);
|
||||
dgvdxs *= -1;
|
||||
for (int d=0; d<3; d++)
|
||||
{
|
||||
dgvdxs(d,d) += 1.0;
|
||||
}
|
||||
//dxidxs: 2*3
|
||||
|
||||
DenseMatrix dg2dxs(3,3); dg2dxs = 0.0;
|
||||
DenseMatrix dg2dxs_tmp(3,2); dg2dxs_tmp = 0.0;
|
||||
MultAtB(dxidxs_m, nde2, dg2dxs_tmp);
|
||||
AddMult_a(-1.0, dg2dxs_tmp, dxidxs_m, dg2dxs);
|
||||
DenseMatrix dg2dxs_tmp2(3,3); dg2dxs_tmp2 = 0.0;
|
||||
MultAtB(dgvdxs, dndxs, dg2dxs_tmp2);
|
||||
dg2dxs += dg2dxs_tmp2;
|
||||
dg2dxs_tmp2 = 0.0;
|
||||
MultAtB(dndxs, dndxs_tmp, dg2dxs_tmp2);
|
||||
AddMult(dg2dxs_tmp2, dgvdxs, dg2dxs);
|
||||
|
||||
DenseMatrix Ne(3,12), Be(6,12), dBe(12,12);
|
||||
BasisVectorDerivs(xi, Ne, Be, dBe);
|
||||
|
||||
DenseMatrix dtao1dxm(3,12); dtao1dxm.CopyRows(Be, 0, 2);
|
||||
DenseMatrix dtao2dxm(3,12); dtao2dxm.CopyRows(Be, 3, 5);
|
||||
|
||||
Vector m_coords_v(12);
|
||||
for (int i=0; i<4; i++)
|
||||
{
|
||||
for (int j=0; j<3; j++)
|
||||
{
|
||||
m_coords_v[i*3+j] = m_coords(i,j);
|
||||
}
|
||||
}
|
||||
|
||||
for (int i=0; i<2; i++)
|
||||
{
|
||||
Vector dxidxm_tmp(num_dofs2); dxidxm_tmp = 0.0;
|
||||
dxidxm.GetRow(i,dxidxm_tmp);
|
||||
|
||||
DenseMatrix dBe_tmp(3,12);
|
||||
dBe_tmp.CopyRows(dBe,i*3,(i+1)*3-1);
|
||||
|
||||
DenseMatrix dtaodxm_tmp(12,12); dtaodxm_tmp = 0.0;
|
||||
outer(m_coords_v, dxidxm_tmp, dtaodxm_tmp);
|
||||
AddMult(dBe_tmp, dtaodxm_tmp, dtao1dxm);
|
||||
|
||||
//dtao1dxm += dBe(:,:,i)*reshape(m_coords(1:4,:)',12,1)*reshape(dxidxm(i,:),1,12); % 3*12
|
||||
dBe_tmp = 0.0;
|
||||
dBe_tmp.CopyRows(dBe,(i+2)*3,(i+3)*3-1);
|
||||
AddMult(dBe_tmp, dtaodxm_tmp, dtao2dxm);
|
||||
|
||||
}
|
||||
|
||||
DenseMatrix dtaodxm(3,12); dtaodxm = 0.0;//tao = tao1 cross tao2
|
||||
|
||||
for (int d=0; d<12; d++)
|
||||
{
|
||||
Vector dtaodxm_tmp(3); dtaodxm_tmp = 0.0;
|
||||
Vector dtaodxm_tmp2(3); dtaodxm_tmp2 = 0.0;
|
||||
Vector tmp1(3); tmp1 = 0.0; dtao1dxm.GetColumn(d,tmp1);
|
||||
Vector m_dxrow2(3); m_dx.GetRow(1, m_dxrow2);
|
||||
Vector m_dxrow1(3); m_dx.GetRow(0, m_dxrow1);
|
||||
Vector tmp2(3); tmp2 = 0.0; dtao2dxm.GetColumn(d,tmp2);
|
||||
|
||||
cross(tmp1, m_dxrow2, dtaodxm_tmp);
|
||||
cross(m_dxrow1,tmp2, dtaodxm_tmp2);
|
||||
dtaodxm_tmp += dtaodxm_tmp2;
|
||||
|
||||
dtaodxm.SetCol(d, dtaodxm_tmp);
|
||||
}
|
||||
|
||||
DenseMatrix dndxm(3,12); dndxm = 0.0;
|
||||
dndxm += dtaodxm;
|
||||
dndxm *= 1.0/nnorm;
|
||||
AddMult_a(-1/nnorm, dndxs_tmp, dtaodxm, dndxm); //dndxs_tmp = normal'*normal
|
||||
|
||||
DenseMatrix dgvdxm(3,12); dgvdxm = 0.0;
|
||||
dgvdxm -= Ne;
|
||||
|
||||
for (int i=0; i<2; i++)
|
||||
{
|
||||
Vector dxidxm_tmp(num_dofs2); dxidxm_tmp = 0.0;
|
||||
dxidxm.GetRow(i,dxidxm_tmp);
|
||||
|
||||
DenseMatrix Be_tmp(3,12);
|
||||
Be_tmp.CopyRows(Be,i*3,(i+1)*3-1);
|
||||
|
||||
DenseMatrix dgvdxm_tmp(12,12); dgvdxm_tmp = 0.0;
|
||||
outer(m_coords_v, dxidxm_tmp, dgvdxm_tmp);
|
||||
AddMult_a(-1.0, Be_tmp, dgvdxm_tmp, dgvdxm);
|
||||
|
||||
}
|
||||
|
||||
DenseMatrix dg2dxsxm(3,12); dg2dxsxm = 0.0;
|
||||
DenseMatrix dg2dxsxm_tmp(3,3); dg2dxsxm_tmp = 0.0;
|
||||
MultAtB(dgvdxs, dndxm, dg2dxsxm);
|
||||
|
||||
MultAtB(dndxs, dndxs_tmp, dg2dxsxm_tmp);
|
||||
AddMult(dg2dxsxm_tmp, dgvdxm, dg2dxsxm); // += dndxs'*normal'*normal*dgvdxm;
|
||||
|
||||
DenseMatrix dgvdxsxmn(3,12); dgvdxsxmn = 0.0;
|
||||
DenseMatrix dgvdxsxmn_tmp(3,2); dgvdxsxmn_tmp = 0.0;
|
||||
MultAtB(dxidxs_m, nde2, dgvdxsxmn_tmp); //dxidxs_m: 2*3
|
||||
|
||||
AddMult_a(-1.0, dgvdxsxmn_tmp, dxidxm, dgvdxsxmn);
|
||||
|
||||
|
||||
for (int i =0; i<2; i++)
|
||||
{
|
||||
DenseMatrix Be_tmp(3,12);
|
||||
Be_tmp.CopyRows(Be,i*3,(i+1)*3-1);
|
||||
|
||||
Vector dxidxs_row(3); dxidxs_row = 0.0; dxidxs_m.GetRow(i,dxidxs_row);
|
||||
DenseMatrix dgvdxsxmn_tmp2(3,3); dgvdxsxmn_tmp2 = 0.0;
|
||||
outer(dxidxs_row, normal, dgvdxsxmn_tmp2);
|
||||
AddMult_a(-1.0, dgvdxsxmn_tmp2, Be_tmp, dgvdxsxmn);
|
||||
}
|
||||
|
||||
dg2dxsxm += dgvdxsxmn;
|
||||
|
||||
DenseMatrix dg2dxmxs(12,3); dg2dxmxs = 0.0;
|
||||
DenseMatrix dg2dxmxs_tmp(12,3); dg2dxmxs_tmp = 0.0;
|
||||
MultAtB(dgvdxm, dndxs, dg2dxmxs);
|
||||
MultAtB(dndxm, dndxs_tmp, dg2dxmxs_tmp);
|
||||
AddMult(dg2dxmxs_tmp, dgvdxs, dg2dxmxs);
|
||||
|
||||
DenseMatrix dgvdxmxsn(12,3); dgvdxmxsn = 0.0;
|
||||
DenseMatrix dgvdxmxsn_tmp(12,2); dgvdxmxsn_tmp = 0.0;
|
||||
|
||||
MultAtB(dxidxm, nde2, dgvdxmxsn_tmp);
|
||||
dgvdxmxsn_tmp *= -1.0;
|
||||
AddMult(dgvdxmxsn_tmp, dxidxs_m, dgvdxmxsn);
|
||||
|
||||
for (int i =0; i<2; i++)
|
||||
{
|
||||
DenseMatrix Be_tmp(3,12);
|
||||
Be_tmp.CopyRows(Be,i*3,(i+1)*3-1);
|
||||
Be_tmp.Transpose(); // Be is now 12*3
|
||||
|
||||
Vector dxidxs_row(3); dxidxs_row = 0.0; dxidxs_m.GetRow(i,dxidxs_row);
|
||||
DenseMatrix dgvdxmxsn_tmp2(3,3); dgvdxmxsn_tmp2 = 0.0;
|
||||
outer(normal, dxidxs_row, dgvdxmxsn_tmp2);
|
||||
AddMult_a(-1.0, Be_tmp, dgvdxmxsn_tmp2, dgvdxmxsn);
|
||||
|
||||
}
|
||||
|
||||
dg2dxmxs += dgvdxmxsn;
|
||||
|
||||
dg2dx.CopyMN(dg2dxs, 0, 0);
|
||||
dg2dx.CopyMN(dg2dxm, 3, 3);
|
||||
dg2dx.CopyMN(dg2dxsxm, 0, 3);
|
||||
dg2dx.CopyMN(dg2dxmxs, 3, 0);
|
||||
|
||||
};
|
||||
|
||||
|
||||
|
||||
void NodeSegConPairs(const Vector x1, const Vector xi2,
|
||||
const DenseMatrix coords2,
|
||||
double& node_g, Vector& node_dg, DenseMatrix& node_dg2)
|
||||
{
|
||||
double gap = 0.0;
|
||||
Vector normal(3); normal = 0.0;
|
||||
Vector dgdxm(12); dgdxm = 0.0;
|
||||
Vector dgdxs(3); dgdxs = 0.0;
|
||||
|
||||
ComputeGapJacobian(x1, xi2, coords2, gap, normal, dgdxm, dgdxs);
|
||||
node_g = gap;
|
||||
|
||||
node_dg.SetSize(12+3);
|
||||
for (int i=0; i<3; i++) { node_dg[i] = dgdxs[i]; }
|
||||
for (int i=0; i<12; i++) { node_dg[i+3] = dgdxm[i]; }
|
||||
|
||||
DenseMatrix dg2dx(15,15); dg2dx = 0.0;
|
||||
DenseMatrix dgvdxmxsn(12,3); dgvdxmxsn = 0.0;
|
||||
ComputeGapHessian(x1, xi2, coords2, dg2dx);
|
||||
|
||||
node_dg2.SetSize(15,15);
|
||||
node_dg2 = dg2dx;
|
||||
|
||||
/*
|
||||
if(obj.space1.conns{e1}(i)==150) % for debugging purpose
|
||||
|
||||
v1 = 1:3;
|
||||
v2 = 1:12;
|
||||
%v1 = ones(1,3)
|
||||
%v2 = ones(1,12)
|
||||
v2 = reshape(v2,4,3);
|
||||
x1n1 = x1 + 0.01*v1;
|
||||
coords2n1 = coords2 + 0.001*v2;
|
||||
[xi2n1, gapv1, ~, ~] = SlaveToMaster(obj, coords2n1, x1n1);
|
||||
[gapn1, n1,dgdxmn1, dgdxsn1] = ComputeGapJacobian(obj, x1n1, xi2n1, coords2n1);
|
||||
x1n2 = x1 - 0.01*v1;
|
||||
coords2n2 = coords2 - 0.001*v2;
|
||||
[xi2n2, gapv2, ~, ~] = SlaveToMaster(obj, coords2n2, x1n2);
|
||||
[gapn2, n2,dgdxmn2, dgdxsn2] = ComputeGapJacobian(obj, x1n2, xi2n2, coords2n2);
|
||||
fprintf('fd\n');
|
||||
%gapv1-gapv2
|
||||
[dgdxsn1(:)',dgdxmn1(:)'] - [dgdxsn2(:)',dgdxmn2(:)']
|
||||
|
||||
%dgdxsn1-dgdxsn2
|
||||
fprintf('code\n');
|
||||
v2n = v2';
|
||||
%dg2dx(1:3,1:3)*0.04*ones(3,1)
|
||||
temp = zeros(12,3);
|
||||
for i = 1:4
|
||||
temp1 = dg2dx(3+(i-1)*3+1:3+i*3,1:3);
|
||||
temp((i-1)*3+1:i*3,:) = temp1';
|
||||
end
|
||||
temp2 = zeros(3,12);
|
||||
for i = 1:4
|
||||
temp3 = dg2dx(1:3,3+(i-1)*3+1:3+i*3);
|
||||
temp2(:,(i-1)*3+1:i*3) = temp3';
|
||||
end
|
||||
%dg2dx
|
||||
%dg2dx(4:end,1:3) = temp;
|
||||
%dg2dx(1:3,4:end) = temp2;
|
||||
%dgvdxm * 0.002*v2n(:)
|
||||
(dg2dx*[0.02*v1(:)',0.002*v2n(:)']')'
|
||||
%dg2dx(4:end,1:3)
|
||||
end*/
|
||||
|
||||
};
|
||||
|
||||
|
||||
// coordsm : (npoints*4, 3) use what class?
|
||||
// m_conn: (npoints*4)
|
||||
void Assemble_Contact(const int m, const int npoints, const int ndofs,
|
||||
const Vector x_s,
|
||||
const Vector xi, const DenseMatrix coordsm, const Array<int> s_conn,
|
||||
const Array<int> m_conn, Vector& g, SparseMatrix& M,
|
||||
std::vector<SparseMatrix>& dM)
|
||||
{
|
||||
int ndim = 3;
|
||||
|
||||
g.SetSize(m);
|
||||
g = 0.0;
|
||||
|
||||
//SparseMatrix M(m, n); // M needs to be the correct size
|
||||
|
||||
//dM.resize(m); // needs to clear?
|
||||
|
||||
double g_tmp = 0.;
|
||||
Vector dg(4*ndim+ndim);
|
||||
dg = 0.;
|
||||
DenseMatrix dg2(4*ndim+ndim,4*ndim+ndim);
|
||||
dg2 = 0.;
|
||||
|
||||
for (int i=0; i<npoints; i++)
|
||||
{
|
||||
Vector x1(ndim);
|
||||
x1[0] = x_s[i*ndim];
|
||||
x1[1] = x_s[i*ndim+1];
|
||||
x1[2] = x_s[i*ndim+2];
|
||||
|
||||
Vector xi2(ndim-1);
|
||||
xi2[0] = xi[i*(ndim-1)];
|
||||
xi2[1] = xi[i*(ndim-1)+1];
|
||||
|
||||
DenseMatrix coords2(4,3);
|
||||
coords2.CopyRows(coordsm, i*4,(i+1)*4-1);
|
||||
|
||||
//how to get coords2?
|
||||
dg = 0.0;
|
||||
dg2 = 0.;
|
||||
NodeSegConPairs(x1, xi2, coords2, g_tmp, dg, dg2);
|
||||
g[s_conn[i]] = g_tmp; // should be unique
|
||||
Array<int> m_conn_i(4);
|
||||
m_conn.GetSubArray(4*i, 4, m_conn_i);
|
||||
|
||||
Array<int> node_conn(5);
|
||||
node_conn[0] = s_conn[i];
|
||||
for (int j=0; j<4; j++)
|
||||
{
|
||||
node_conn[j+1] = m_conn_i[j];
|
||||
}
|
||||
|
||||
Array<int> M_i_tmp(1);
|
||||
M_i_tmp[0] = s_conn[i];
|
||||
|
||||
//j_idx = (node_conn-1)*obj.disp_field.num_components +repmat((1:obj.disp_field.num_components)', 1, length(node_conn{i}));
|
||||
Array<int> j_idx(5*ndim); j_idx = 0;
|
||||
for (int j=0; j< 5; j++)
|
||||
{
|
||||
for (int k=0; k<ndim; k++)
|
||||
{
|
||||
j_idx[j*ndim+k] = node_conn[j]*ndim+k;
|
||||
}
|
||||
}
|
||||
DenseMatrix M_v_tmp(1, ndim*(4+1)); // SetData now?
|
||||
M_v_tmp.SetRow(0, dg);
|
||||
|
||||
M.AddSubMatrix(M_i_tmp, j_idx, M_v_tmp);
|
||||
|
||||
Array<int> dM_i(ndim*(4+1));
|
||||
Array<int> dM_j(ndim*(4+1));
|
||||
|
||||
for (int j=0; j< ndim*(4+1); j++)
|
||||
{
|
||||
dM_i[j] = j_idx[j];
|
||||
dM_j[j] = j_idx[j];
|
||||
}
|
||||
dM[s_conn[i]].AddSubMatrix(dM_i,dM_j, dg2);
|
||||
dM[s_conn[i]].Finalize();
|
||||
dM[s_conn[i]].Threshold(0.0);
|
||||
dM[s_conn[i]].SortColumnIndices();
|
||||
}
|
||||
M.Finalize();
|
||||
M.Threshold(0.0);
|
||||
M.SortColumnIndices();
|
||||
};
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,396 +0,0 @@
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <set>
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
#ifndef PROBLEM_DEFS
|
||||
#define PROBLEM_DEFS
|
||||
|
||||
|
||||
|
||||
// abstract OptProblem class
|
||||
// of the form
|
||||
// min_(u,m) f(u,m) s.t. c(u,m)=0 and m>=ml
|
||||
// the primal variable (u, m) is represented as a BlockVector
|
||||
|
||||
class OptProblem
|
||||
{
|
||||
protected:
|
||||
int dimU, dimM, dimC;
|
||||
Array<int> block_offsetsx;
|
||||
Vector ml;
|
||||
public:
|
||||
OptProblem();
|
||||
virtual double CalcObjective(const BlockVector &) const = 0;
|
||||
virtual void Duf(const BlockVector &, Vector &) const = 0;
|
||||
virtual void Dmf(const BlockVector &, Vector &) const = 0;
|
||||
void CalcObjectiveGrad(const BlockVector &, BlockVector &) const;
|
||||
virtual SparseMatrix* Duuf(const BlockVector &) = 0;
|
||||
virtual SparseMatrix* Dumf(const BlockVector &) = 0;
|
||||
virtual SparseMatrix* Dmuf(const BlockVector &) = 0;
|
||||
virtual SparseMatrix* Dmmf(const BlockVector &) = 0;
|
||||
virtual void c(const BlockVector &, Vector &) const = 0;
|
||||
virtual SparseMatrix* Duc(const BlockVector &) = 0;
|
||||
virtual SparseMatrix* Dmc(const BlockVector &) = 0;
|
||||
// TO DO: include Hessian terms of constraint c
|
||||
// TO DO: include log-barrier lumped-mass and pass that
|
||||
// to the optimizer
|
||||
//virtual SparseMatrix* GetLogBarrierLumpedMass() = 0;
|
||||
int GetDimU() const { return dimU; };
|
||||
int GetDimM() const { return dimM; };
|
||||
int GetDimC() const { return dimC; };
|
||||
Vector Getml() const { return ml; };
|
||||
~OptProblem();
|
||||
};
|
||||
|
||||
|
||||
// abstract ContactProblem class
|
||||
// of the form
|
||||
// min_d e(d) s.t. g(d) >= 0
|
||||
// TO DO: add functionality for gap function Hessian apply
|
||||
class ContactProblem : public OptProblem
|
||||
{
|
||||
protected:
|
||||
int dimD;
|
||||
int dimS;
|
||||
Array<int> block_offsetsx;
|
||||
public:
|
||||
//ContactProblem(int, int); // constructor
|
||||
ContactProblem();
|
||||
void InitializeParentData(int, int);
|
||||
double CalcObjective(const BlockVector &) const; // objective e
|
||||
void Duf(const BlockVector &, Vector &) const;
|
||||
void Dmf(const BlockVector &, Vector &) const;
|
||||
SparseMatrix* Duuf(const BlockVector &);
|
||||
SparseMatrix* Dumf(const BlockVector &);
|
||||
SparseMatrix* Dmuf(const BlockVector &);
|
||||
SparseMatrix* Dmmf(const BlockVector &);
|
||||
void c(const BlockVector &, Vector &) const;
|
||||
SparseMatrix* Duc(const BlockVector &);
|
||||
SparseMatrix* Dmc(const BlockVector &);
|
||||
virtual double E(const Vector &) const = 0; // objective e(d) (energy function)
|
||||
virtual void DdE(const Vector &, Vector &) const = 0; // gradient of objective De / Dd
|
||||
virtual SparseMatrix* DddE(const Vector &) = 0; // Hessian of objective D^2 e / D d^2
|
||||
virtual void g(const Vector &, Vector &) const = 0; // inequality constraint g(d) >= 0 (gap function)
|
||||
virtual SparseMatrix* Ddg(const Vector &) = 0; // Jacobian of inequality constraint Dg / Dd
|
||||
int GetDimD() const { return dimD; };
|
||||
int GetDimS() const { return dimS; };
|
||||
virtual ~ContactProblem();
|
||||
};
|
||||
|
||||
|
||||
class ObstacleProblem : public ContactProblem
|
||||
{
|
||||
protected:
|
||||
// data to define energy objective function e(d) = 0.5 d^T K d - f^T d, g(d) = d >= 0
|
||||
// stiffness matrix used to define objective
|
||||
BilinearForm *Kform;
|
||||
LinearForm *fform;
|
||||
Array<int> empty_tdof_list; // needed for calls to FormSystemMatrix
|
||||
SparseMatrix K;
|
||||
SparseMatrix *J;
|
||||
FiniteElementSpace *Vh;
|
||||
Vector f;
|
||||
public :
|
||||
ObstacleProblem(FiniteElementSpace* , double (*fSource)(const Vector &));
|
||||
double E(const Vector &) const;
|
||||
void DdE(const Vector &, Vector &) const;
|
||||
SparseMatrix* DddE(const Vector &);
|
||||
void g(const Vector &, Vector &) const;
|
||||
SparseMatrix* Ddg(const Vector &);
|
||||
// TO DO: include lumped-mass for the log-barrier term
|
||||
//SparseMatrix* GetLogBarrierLumpedMass();
|
||||
virtual ~ObstacleProblem();
|
||||
};
|
||||
|
||||
class DirichletObstacleProblem : public ContactProblem
|
||||
{
|
||||
protected:
|
||||
// data to define energy objective function e(d) = 0.5 d^T K d - f^T d, g(d) = d + \psi >= 0
|
||||
// stiffness matrix used to define objective
|
||||
BilinearForm *Kform;
|
||||
LinearForm *fform;
|
||||
Array<int> ess_tdof_list; // needed for calls to FormSystemMatrix
|
||||
SparseMatrix *K;
|
||||
SparseMatrix *J;
|
||||
FiniteElementSpace *Vh;
|
||||
Vector f;
|
||||
Vector psi;
|
||||
Vector xDC;
|
||||
public :
|
||||
DirichletObstacleProblem(FiniteElementSpace*, Vector&, double (*fSource)(const Vector &), double (*obstacleSource)(const Vector &), Array<int> tdof_list, bool);
|
||||
double E(const Vector &) const;
|
||||
void DdE(const Vector &, Vector &) const;
|
||||
SparseMatrix* DddE(const Vector &);
|
||||
void g(const Vector &, Vector &) const;
|
||||
SparseMatrix* Ddg(const Vector &);
|
||||
virtual ~DirichletObstacleProblem();
|
||||
};
|
||||
|
||||
|
||||
// abstract out technology for removing null rows of the Jacobian from an existing contact problem
|
||||
class ReducedContactProblem : public ContactProblem
|
||||
{
|
||||
protected:
|
||||
Array<int> activeConstraints;
|
||||
Array<int> fixedDofs;
|
||||
ContactProblem * contact;
|
||||
int dimSin;
|
||||
public:
|
||||
ReducedContactProblem(ContactProblem * contact, Array<int> activeConstraints, Array<int> fixedDofs);
|
||||
double E(const Vector &) const;
|
||||
void DdE(const Vector &, Vector &) const;
|
||||
SparseMatrix* DddE(const Vector &);
|
||||
void g(const Vector &, Vector &) const;
|
||||
SparseMatrix* Ddg(const Vector &);
|
||||
virtual ~ReducedContactProblem();
|
||||
};
|
||||
|
||||
|
||||
class QPContactProblem : public ContactProblem
|
||||
{
|
||||
protected:
|
||||
SparseMatrix *K;
|
||||
SparseMatrix *J;
|
||||
Vector f;
|
||||
Vector g0;
|
||||
public:
|
||||
QPContactProblem(const SparseMatrix, const SparseMatrix, const Vector, const Vector);
|
||||
double E(const Vector &) const;
|
||||
void DdE(const Vector &, Vector &) const;
|
||||
SparseMatrix* DddE(const Vector &);
|
||||
void g(const Vector &, Vector &) const;
|
||||
SparseMatrix* Ddg(const Vector &);
|
||||
virtual ~QPContactProblem();
|
||||
};
|
||||
|
||||
|
||||
typedef int Index;
|
||||
typedef double Number;
|
||||
|
||||
class ExContactBlockTL : public ContactProblem
|
||||
{
|
||||
public:
|
||||
double E(const Vector &) const;
|
||||
void DdE(const Vector &, Vector &) const;
|
||||
SparseMatrix* DddE(const Vector &);
|
||||
void g(const Vector &, Vector &) const;
|
||||
SparseMatrix* Ddg(const Vector &);
|
||||
FiniteElementSpace GetVh1();
|
||||
FiniteElementSpace GetVh2();
|
||||
|
||||
public:
|
||||
/** default constructor */
|
||||
ExContactBlockTL(int );
|
||||
|
||||
|
||||
/** default destructor */
|
||||
virtual ~ExContactBlockTL();
|
||||
|
||||
///**@name Overloaded from TNLP */
|
||||
///** Method to return some info about the nlp */
|
||||
//virtual bool get_nlp_info(
|
||||
// Index& n,
|
||||
// Index& m,
|
||||
// Index& nnz_jac_g,
|
||||
// Index& nnz_h_lag,
|
||||
// IndexStyleEnum& index_style
|
||||
//);
|
||||
|
||||
///** Method to return the bounds for my problem */
|
||||
//virtual bool get_bounds_info(
|
||||
// Index n,
|
||||
// Number* x_l,
|
||||
// Number* x_u,
|
||||
// Index m,
|
||||
// Number* g_l,
|
||||
// Number* g_u
|
||||
//);
|
||||
|
||||
///** Method to return the starting point for the algorithm */
|
||||
//virtual bool get_starting_point(
|
||||
// Index n,
|
||||
// bool init_x,
|
||||
// Number* x,
|
||||
// bool init_z,
|
||||
// Number* z_L,
|
||||
// Number* z_U,
|
||||
// Index m,
|
||||
// bool init_lambda,
|
||||
// Number* lambda
|
||||
//);
|
||||
|
||||
/* Method to return the objective value */
|
||||
virtual bool eval_f(
|
||||
Index n,
|
||||
const Number* x,
|
||||
bool new_x,
|
||||
Number& obj_value
|
||||
) const;
|
||||
|
||||
/* Method to return the gradient of the objective */
|
||||
virtual bool eval_grad_f(
|
||||
Index n,
|
||||
const Number* x,
|
||||
bool new_x,
|
||||
Number* grad_f
|
||||
) const;
|
||||
|
||||
/* Method to return the constraint residuals */
|
||||
virtual bool eval_g(
|
||||
Index n,
|
||||
const Number* x,
|
||||
bool new_x,
|
||||
Index m,
|
||||
Number* cons
|
||||
) const;
|
||||
|
||||
/* Method to return:
|
||||
1) The structure of the Jacobian (if "values" is NULL)
|
||||
2) The values of the Jacobian (if "values" is not NULL)
|
||||
*/
|
||||
virtual bool eval_jac_g(
|
||||
Index n,
|
||||
const Number* x,
|
||||
bool new_x,
|
||||
Index m,
|
||||
Index nele_jac,
|
||||
Index* iRow,
|
||||
Index* jCol,
|
||||
Number* values
|
||||
) const;
|
||||
|
||||
/* Method to return:
|
||||
* 1) The structure of the Hessian of the Lagrangian (if "values" is NULL)
|
||||
* 2) The values of the Hessian of the Lagrangian (if "values" is not NULL)
|
||||
*/
|
||||
virtual bool eval_h(
|
||||
Index n,
|
||||
const Number* x,
|
||||
bool new_x,
|
||||
Number obj_factor,
|
||||
Index m,
|
||||
const Number* lambda,
|
||||
bool new_lambda,
|
||||
Index nele_hess,
|
||||
Index* iRow,
|
||||
Index* jCol,
|
||||
Number* values
|
||||
);
|
||||
|
||||
///** This method is called when the algorithm is complete so the TNLP can store/write the solution */
|
||||
//virtual void finalize_solution(
|
||||
// SolverReturn status,
|
||||
// Index n,
|
||||
// const Number* x,
|
||||
// const Number* z_L,
|
||||
// const Number* z_U,
|
||||
// Index m,
|
||||
// const Number* g,
|
||||
// const Number* lambda,
|
||||
// Number obj_value,
|
||||
// const IpoptData* ip_data,
|
||||
// IpoptCalculatedQuantities* ip_cq
|
||||
//);
|
||||
|
||||
private:
|
||||
void update_g() const;
|
||||
void update_jac();
|
||||
void update_hess();
|
||||
|
||||
private:
|
||||
/**@name Methods to block default compiler methods.
|
||||
*
|
||||
* The compiler automatically generates the following three methods.
|
||||
* Since the default compiler implementation is generally not what
|
||||
* you want (for all but the most simple classes), we usually
|
||||
* put the declarations of these methods in the private section
|
||||
* and never implement them. This prevents the compiler from
|
||||
* implementing an incorrect "default" behavior without us
|
||||
* knowing. (See Scott Meyers book, "Effective C++")
|
||||
*/
|
||||
ExContactBlockTL(
|
||||
const ExContactBlockTL&
|
||||
);
|
||||
|
||||
ExContactBlockTL& operator=(
|
||||
const ExContactBlockTL&
|
||||
);
|
||||
|
||||
Array<int> attr;
|
||||
Array<int> m_attr;
|
||||
Array<int> s_conn; // connectivity of the second/slave mesh
|
||||
std::string mesh_file1;
|
||||
std::string mesh_file2;
|
||||
Mesh* mesh1;
|
||||
Mesh* mesh2;
|
||||
FiniteElementCollection* fec1;
|
||||
FiniteElementCollection* fec2;
|
||||
FiniteElementSpace* fespace1;
|
||||
FiniteElementSpace* fespace2;
|
||||
Array<int> ess_tdof_list1;
|
||||
Array<int> ess_tdof_list2;
|
||||
GridFunction nodes0;
|
||||
GridFunction* nodes1;
|
||||
GridFunction* nodes2;
|
||||
mutable GridFunction* x1;
|
||||
mutable GridFunction* x2;
|
||||
LinearForm* b1;
|
||||
LinearForm* b2;
|
||||
PWConstCoefficient* lambda1_func;
|
||||
PWConstCoefficient* lambda2_func;
|
||||
PWConstCoefficient* mu1_func;
|
||||
PWConstCoefficient* mu2_func;
|
||||
BilinearForm* a1;
|
||||
BilinearForm* a2;
|
||||
|
||||
mfem::Vector lambda1;
|
||||
mfem::Vector lambda2;
|
||||
mfem::Vector mu1;
|
||||
mfem::Vector mu2;
|
||||
mutable mfem::Vector xyz;
|
||||
|
||||
std::set<int> bdryVerts2;
|
||||
|
||||
int dim;
|
||||
// degrees of freedom of both meshes
|
||||
int ndof_1;
|
||||
int ndof_2;
|
||||
int ndofs;
|
||||
// number of nodes for each mesh
|
||||
int nnd_1;
|
||||
int nnd_2;
|
||||
int nnd;
|
||||
|
||||
int npoints;
|
||||
|
||||
SparseMatrix A1;
|
||||
mfem::Vector B1, X1;
|
||||
SparseMatrix A2;
|
||||
mfem::Vector B2, X2;
|
||||
|
||||
SparseMatrix* K;
|
||||
mutable mfem::Vector gapv;
|
||||
mutable mfem::Vector m_xi;
|
||||
mutable mfem::Vector xs;
|
||||
|
||||
mutable Array<int> m_conn; // only works for linear elements that have 4 vertices!
|
||||
mutable DenseMatrix* coordsm;
|
||||
mutable SparseMatrix* M;
|
||||
|
||||
mutable std::vector<SparseMatrix>* dM;
|
||||
|
||||
Array<int> Dirichlet_dof;
|
||||
Array<double> Dirichlet_val;
|
||||
|
||||
public:
|
||||
Mesh * GetMesh1() {return mesh1;}
|
||||
Mesh * GetMesh2() {return mesh2;}
|
||||
Array<int> GetDirichletDofs() {return Dirichlet_dof;}
|
||||
Array<double> GetDirichletVals() {return Dirichlet_val;}
|
||||
|
||||
};
|
||||
|
||||
#endif
|
||||
+7
-13
@@ -246,23 +246,17 @@ int main(int argc, char *argv[])
|
||||
e_var /= (nsteps + 1);
|
||||
double e_sd = sqrt(e_var);
|
||||
|
||||
double e_loc_stats[2];
|
||||
double *e_stats = (myid == 0) ? new double[2 * num_procs] : (double*)NULL;
|
||||
|
||||
e_loc_stats[0] = e_mean;
|
||||
e_loc_stats[1] = e_sd;
|
||||
MPI_Gather(e_loc_stats, 2, MPI_DOUBLE, e_stats, 2, MPI_DOUBLE, 0, comm);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << endl << "Mean and standard deviation of the energy "
|
||||
<< "for different initial conditions" << endl;
|
||||
for (int i = 0; i < num_procs; i++)
|
||||
cout << endl << "Mean and standard deviation of the energy" << endl;
|
||||
}
|
||||
for (int i = 0; i < num_procs; i++)
|
||||
{
|
||||
if (myid == i)
|
||||
{
|
||||
cout << i << ": " << e_stats[2 * i + 0]
|
||||
<< "\t" << e_stats[2 * i + 1] << endl;
|
||||
cout << myid << ": " << e_mean << "\t" << e_sd << endl;
|
||||
}
|
||||
delete [] e_stats;
|
||||
MPI_Barrier(comm);
|
||||
}
|
||||
|
||||
// 9. Finalize the GnuPlot output
|
||||
|
||||
+36
-33
@@ -32,7 +32,6 @@
|
||||
// We recommend viewing Example 22 before viewing this example.
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <memory>
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
@@ -45,7 +44,7 @@ using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
// Class for setting up a simple Cartesian PML region
|
||||
class PML
|
||||
class CartesianPML
|
||||
{
|
||||
private:
|
||||
Mesh *mesh;
|
||||
@@ -70,7 +69,7 @@ private:
|
||||
|
||||
public:
|
||||
// Constructor
|
||||
PML(Mesh *mesh_,Array2D<double> length_);
|
||||
CartesianPML(Mesh *mesh_,Array2D<double> length_);
|
||||
|
||||
// Return Computational Domain Boundary
|
||||
Array2D<double> GetCompDomainBdr() {return comp_dom_bdr;}
|
||||
@@ -92,12 +91,12 @@ public:
|
||||
class PMLDiagMatrixCoefficient : public VectorCoefficient
|
||||
{
|
||||
private:
|
||||
PML * pml = nullptr;
|
||||
void (*Function)(const Vector &, PML *, Vector &);
|
||||
CartesianPML * pml = nullptr;
|
||||
void (*Function)(const Vector &, CartesianPML *, Vector &);
|
||||
public:
|
||||
PMLDiagMatrixCoefficient(int dim, void(*F)(const Vector &, PML *,
|
||||
PMLDiagMatrixCoefficient(int dim, void(*F)(const Vector &, CartesianPML *,
|
||||
Vector &),
|
||||
PML * pml_)
|
||||
CartesianPML * pml_)
|
||||
: VectorCoefficient(dim), pml(pml_), Function(F)
|
||||
{}
|
||||
|
||||
@@ -126,13 +125,13 @@ void source(const Vector &x, Vector & f);
|
||||
|
||||
// Functions for computing the necessary coefficients after PML stretching.
|
||||
// J is the Jacobian matrix of the stretching function
|
||||
void detJ_JT_J_inv_Re(const Vector &x, PML * pml, Vector &D);
|
||||
void detJ_JT_J_inv_Im(const Vector &x, PML * pml, Vector &D);
|
||||
void detJ_JT_J_inv_abs(const Vector &x, PML * pml, Vector &D);
|
||||
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, Vector &D);
|
||||
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, Vector &D);
|
||||
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, Vector &D);
|
||||
|
||||
void detJ_inv_JT_J_Re(const Vector &x, PML * pml, Vector &D);
|
||||
void detJ_inv_JT_J_Im(const Vector &x, PML * pml, Vector &D);
|
||||
void detJ_inv_JT_J_abs(const Vector &x, PML * pml, Vector &D);
|
||||
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, Vector &D);
|
||||
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, Vector &D);
|
||||
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, Vector &D);
|
||||
|
||||
Array2D<double> comp_domain_bdr;
|
||||
Array2D<double> domain_bdr;
|
||||
@@ -268,7 +267,7 @@ int main(int argc, char *argv[])
|
||||
length = 0.25;
|
||||
break;
|
||||
}
|
||||
PML * pml = new PML(mesh,length);
|
||||
CartesianPML * pml = new CartesianPML(mesh,length);
|
||||
comp_domain_bdr = pml->GetCompDomainBdr();
|
||||
domain_bdr = pml->GetDomainBdr();
|
||||
|
||||
@@ -468,14 +467,16 @@ int main(int argc, char *argv[])
|
||||
offsets[2] = fespace->GetTrueVSize();
|
||||
offsets.PartialSum();
|
||||
|
||||
std::unique_ptr<Operator> pc_r;
|
||||
std::unique_ptr<Operator> pc_i;
|
||||
Operator *pc_r = nullptr;
|
||||
Operator *pc_i = nullptr;
|
||||
double s = (conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0;
|
||||
if (pa)
|
||||
{
|
||||
// Jacobi Smoother
|
||||
pc_r.reset(new OperatorJacobiSmoother(prec, ess_tdof_list));
|
||||
pc_i.reset(new ScaledOperator(pc_r.get(), s));
|
||||
OperatorJacobiSmoother *d00 = new OperatorJacobiSmoother(prec, ess_tdof_list);
|
||||
ScaledOperator *d11 = new ScaledOperator(d00, s);
|
||||
pc_r = d00;
|
||||
pc_i = d11;
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -484,13 +485,15 @@ int main(int argc, char *argv[])
|
||||
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
|
||||
|
||||
// Gauss-Seidel Smoother
|
||||
pc_r.reset(new GSSmoother(*PCOpAh.As<SparseMatrix>()));
|
||||
pc_i.reset(new ScaledOperator(pc_r.get(), s));
|
||||
GSSmoother *gs00 = new GSSmoother(*PCOpAh.As<SparseMatrix>());
|
||||
ScaledOperator *gs11 = new ScaledOperator(gs00, s);
|
||||
pc_r = gs00;
|
||||
pc_i = gs11;
|
||||
}
|
||||
|
||||
BlockDiagonalPreconditioner BlockDP(offsets);
|
||||
BlockDP.SetDiagonalBlock(0, pc_r.get());
|
||||
BlockDP.SetDiagonalBlock(1, pc_i.get());
|
||||
BlockDP.SetDiagonalBlock(0, pc_r);
|
||||
BlockDP.SetDiagonalBlock(1, pc_i);
|
||||
|
||||
GMRESSolver gmres;
|
||||
gmres.SetPrintLevel(1);
|
||||
@@ -804,7 +807,7 @@ void E_bdr_data_Im(const Vector &x, Vector &E)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_JT_J_inv_Re(const Vector &x, PML * pml, Vector &D)
|
||||
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, Vector &D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det(1.0, 0.0);
|
||||
@@ -821,7 +824,7 @@ void detJ_JT_J_inv_Re(const Vector &x, PML * pml, Vector &D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_JT_J_inv_Im(const Vector &x, PML * pml, Vector &D)
|
||||
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, Vector &D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det = 1.0;
|
||||
@@ -838,7 +841,7 @@ void detJ_JT_J_inv_Im(const Vector &x, PML * pml, Vector &D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_JT_J_inv_abs(const Vector &x, PML * pml, Vector &D)
|
||||
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, Vector &D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det = 1.0;
|
||||
@@ -855,7 +858,7 @@ void detJ_JT_J_inv_abs(const Vector &x, PML * pml, Vector &D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_inv_JT_J_Re(const Vector &x, PML * pml, Vector &D)
|
||||
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, Vector &D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det(1.0, 0.0);
|
||||
@@ -880,7 +883,7 @@ void detJ_inv_JT_J_Re(const Vector &x, PML * pml, Vector &D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_inv_JT_J_Im(const Vector &x, PML * pml, Vector &D)
|
||||
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, Vector &D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det = 1.0;
|
||||
@@ -904,7 +907,7 @@ void detJ_inv_JT_J_Im(const Vector &x, PML * pml, Vector &D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_inv_JT_J_abs(const Vector &x, PML * pml, Vector &D)
|
||||
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, Vector &D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det = 1.0;
|
||||
@@ -928,14 +931,14 @@ void detJ_inv_JT_J_abs(const Vector &x, PML * pml, Vector &D)
|
||||
}
|
||||
}
|
||||
|
||||
PML::PML(Mesh *mesh_, Array2D<double> length_)
|
||||
CartesianPML::CartesianPML(Mesh *mesh_, Array2D<double> length_)
|
||||
: mesh(mesh_), length(length_)
|
||||
{
|
||||
dim = mesh->Dimension();
|
||||
SetBoundaries();
|
||||
}
|
||||
|
||||
void PML::SetBoundaries()
|
||||
void CartesianPML::SetBoundaries()
|
||||
{
|
||||
comp_dom_bdr.SetSize(dim, 2);
|
||||
dom_bdr.SetSize(dim, 2);
|
||||
@@ -950,7 +953,7 @@ void PML::SetBoundaries()
|
||||
}
|
||||
}
|
||||
|
||||
void PML::SetAttributes(Mesh *mesh_)
|
||||
void CartesianPML::SetAttributes(Mesh *mesh_)
|
||||
{
|
||||
// Initialize bdr attributes
|
||||
for (int i = 0; i < mesh_->GetNBE(); ++i)
|
||||
@@ -999,8 +1002,8 @@ void PML::SetAttributes(Mesh *mesh_)
|
||||
mesh_->SetAttributes();
|
||||
}
|
||||
|
||||
void PML::StretchFunction(const Vector &x,
|
||||
vector<complex<double>> &dxs)
|
||||
void CartesianPML::StretchFunction(const Vector &x,
|
||||
vector<complex<double>> &dxs)
|
||||
{
|
||||
complex<double> zi = complex<double>(0., 1.);
|
||||
|
||||
|
||||
+38
-34
@@ -44,7 +44,7 @@ using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
// Class for setting up a simple Cartesian PML region
|
||||
class PML
|
||||
class CartesianPML
|
||||
{
|
||||
private:
|
||||
Mesh *mesh;
|
||||
@@ -69,7 +69,7 @@ private:
|
||||
|
||||
public:
|
||||
// Constructor
|
||||
PML(Mesh *mesh_,Array2D<double> length_);
|
||||
CartesianPML(Mesh *mesh_,Array2D<double> length_);
|
||||
|
||||
// Return Computational Domain Boundary
|
||||
Array2D<double> GetCompDomainBdr() {return comp_dom_bdr;}
|
||||
@@ -91,12 +91,12 @@ public:
|
||||
class PMLDiagMatrixCoefficient : public VectorCoefficient
|
||||
{
|
||||
private:
|
||||
PML * pml = nullptr;
|
||||
void (*Function)(const Vector &, PML *, Vector &);
|
||||
CartesianPML * pml = nullptr;
|
||||
void (*Function)(const Vector &, CartesianPML *, Vector &);
|
||||
public:
|
||||
PMLDiagMatrixCoefficient(int dim, void(*F)(const Vector &, PML *,
|
||||
PMLDiagMatrixCoefficient(int dim, void(*F)(const Vector &, CartesianPML *,
|
||||
Vector &),
|
||||
PML * pml_)
|
||||
CartesianPML * pml_)
|
||||
: VectorCoefficient(dim), pml(pml_), Function(F)
|
||||
{}
|
||||
|
||||
@@ -125,13 +125,13 @@ void source(const Vector &x, Vector & f);
|
||||
|
||||
// Functions for computing the necessary coefficients after PML stretching.
|
||||
// J is the Jacobian matrix of the stretching function
|
||||
void detJ_JT_J_inv_Re(const Vector &x, PML * pml, Vector & D);
|
||||
void detJ_JT_J_inv_Im(const Vector &x, PML * pml, Vector & D);
|
||||
void detJ_JT_J_inv_abs(const Vector &x, PML * pml, Vector & D);
|
||||
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, Vector & D);
|
||||
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, Vector & D);
|
||||
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, Vector & D);
|
||||
|
||||
void detJ_inv_JT_J_Re(const Vector &x, PML * pml, Vector & D);
|
||||
void detJ_inv_JT_J_Im(const Vector &x, PML * pml, Vector & D);
|
||||
void detJ_inv_JT_J_abs(const Vector &x, PML * pml, Vector & D);
|
||||
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, Vector & D);
|
||||
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, Vector & D);
|
||||
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, Vector & D);
|
||||
|
||||
Array2D<double> comp_domain_bdr;
|
||||
Array2D<double> domain_bdr;
|
||||
@@ -295,7 +295,7 @@ int main(int argc, char *argv[])
|
||||
length = 0.25;
|
||||
break;
|
||||
}
|
||||
PML * pml = new PML(mesh,length);
|
||||
CartesianPML * pml = new CartesianPML(mesh,length);
|
||||
comp_domain_bdr = pml->GetCompDomainBdr();
|
||||
domain_bdr = pml->GetDomainBdr();
|
||||
|
||||
@@ -478,11 +478,11 @@ int main(int argc, char *argv[])
|
||||
if (!pa && mumps_solver)
|
||||
{
|
||||
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
MUMPSSolver mumps(A->GetComm());
|
||||
MUMPSSolver mumps;
|
||||
mumps.SetPrintLevel(0);
|
||||
mumps.SetMatrixSymType(MUMPSSolver::MatType::UNSYMMETRIC);
|
||||
mumps.SetOperator(*A);
|
||||
mumps.Mult(B, X);
|
||||
mumps.Mult(B,X);
|
||||
delete A;
|
||||
}
|
||||
#endif
|
||||
@@ -524,14 +524,16 @@ int main(int argc, char *argv[])
|
||||
offsets[2] = fespace->GetTrueVSize();
|
||||
offsets.PartialSum();
|
||||
|
||||
std::unique_ptr<Operator> pc_r;
|
||||
std::unique_ptr<Operator> pc_i;
|
||||
Operator *pc_r = nullptr;
|
||||
Operator *pc_i = nullptr;
|
||||
int s = (conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0;
|
||||
if (pa)
|
||||
{
|
||||
// Jacobi Smoother
|
||||
pc_r.reset(new OperatorJacobiSmoother(prec, ess_tdof_list));
|
||||
pc_i.reset(new ScaledOperator(pc_r.get(), s));
|
||||
OperatorJacobiSmoother *d00 = new OperatorJacobiSmoother(prec, ess_tdof_list);
|
||||
ScaledOperator *d11 = new ScaledOperator(d00, s);
|
||||
pc_r = d00;
|
||||
pc_i = d11;
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -539,13 +541,15 @@ int main(int argc, char *argv[])
|
||||
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
|
||||
|
||||
// Hypre AMS
|
||||
pc_r.reset(new HypreAMS(*PCOpAh.As<HypreParMatrix>(), fespace));
|
||||
pc_i.reset(new ScaledOperator(pc_r.get(), s));
|
||||
HypreAMS *ams00 = new HypreAMS(*PCOpAh.As<HypreParMatrix>(), fespace);
|
||||
ScaledOperator *ams11 = new ScaledOperator(ams00, s);
|
||||
pc_r = ams00;
|
||||
pc_i = ams11;
|
||||
}
|
||||
|
||||
BlockDiagonalPreconditioner BlockDP(offsets);
|
||||
BlockDP.SetDiagonalBlock(0, pc_r.get());
|
||||
BlockDP.SetDiagonalBlock(1, pc_i.get());
|
||||
BlockDP.SetDiagonalBlock(0, pc_r);
|
||||
BlockDP.SetDiagonalBlock(1, pc_i);
|
||||
|
||||
GMRESSolver gmres(MPI_COMM_WORLD);
|
||||
gmres.SetPrintLevel(1);
|
||||
@@ -880,7 +884,7 @@ void E_bdr_data_Im(const Vector &x, Vector &E)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_JT_J_inv_Re(const Vector &x, PML * pml, Vector & D)
|
||||
void detJ_JT_J_inv_Re(const Vector &x, CartesianPML * pml, Vector & D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det(1.0, 0.0);
|
||||
@@ -897,7 +901,7 @@ void detJ_JT_J_inv_Re(const Vector &x, PML * pml, Vector & D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_JT_J_inv_Im(const Vector &x, PML * pml, Vector & D)
|
||||
void detJ_JT_J_inv_Im(const Vector &x, CartesianPML * pml, Vector & D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det = 1.0;
|
||||
@@ -914,7 +918,7 @@ void detJ_JT_J_inv_Im(const Vector &x, PML * pml, Vector & D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_JT_J_inv_abs(const Vector &x, PML * pml, Vector & D)
|
||||
void detJ_JT_J_inv_abs(const Vector &x, CartesianPML * pml, Vector & D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det = 1.0;
|
||||
@@ -931,7 +935,7 @@ void detJ_JT_J_inv_abs(const Vector &x, PML * pml, Vector & D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_inv_JT_J_Re(const Vector &x, PML * pml, Vector & D)
|
||||
void detJ_inv_JT_J_Re(const Vector &x, CartesianPML * pml, Vector & D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det(1.0, 0.0);
|
||||
@@ -956,7 +960,7 @@ void detJ_inv_JT_J_Re(const Vector &x, PML * pml, Vector & D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_inv_JT_J_Im(const Vector &x, PML * pml, Vector & D)
|
||||
void detJ_inv_JT_J_Im(const Vector &x, CartesianPML * pml, Vector & D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det = 1.0;
|
||||
@@ -980,7 +984,7 @@ void detJ_inv_JT_J_Im(const Vector &x, PML * pml, Vector & D)
|
||||
}
|
||||
}
|
||||
|
||||
void detJ_inv_JT_J_abs(const Vector &x, PML * pml, Vector & D)
|
||||
void detJ_inv_JT_J_abs(const Vector &x, CartesianPML * pml, Vector & D)
|
||||
{
|
||||
vector<complex<double>> dxs(dim);
|
||||
complex<double> det = 1.0;
|
||||
@@ -1004,14 +1008,14 @@ void detJ_inv_JT_J_abs(const Vector &x, PML * pml, Vector & D)
|
||||
}
|
||||
}
|
||||
|
||||
PML::PML(Mesh *mesh_, Array2D<double> length_)
|
||||
CartesianPML::CartesianPML(Mesh *mesh_, Array2D<double> length_)
|
||||
: mesh(mesh_), length(length_)
|
||||
{
|
||||
dim = mesh->Dimension();
|
||||
SetBoundaries();
|
||||
}
|
||||
|
||||
void PML::SetBoundaries()
|
||||
void CartesianPML::SetBoundaries()
|
||||
{
|
||||
comp_dom_bdr.SetSize(dim, 2);
|
||||
dom_bdr.SetSize(dim, 2);
|
||||
@@ -1026,7 +1030,7 @@ void PML::SetBoundaries()
|
||||
}
|
||||
}
|
||||
|
||||
void PML::SetAttributes(ParMesh *pmesh)
|
||||
void CartesianPML::SetAttributes(ParMesh *pmesh)
|
||||
{
|
||||
// Initialize bdr attributes
|
||||
for (int i = 0; i < pmesh->GetNBE(); ++i)
|
||||
@@ -1076,8 +1080,8 @@ void PML::SetAttributes(ParMesh *pmesh)
|
||||
pmesh->SetAttributes();
|
||||
}
|
||||
|
||||
void PML::StretchFunction(const Vector &x,
|
||||
vector<complex<double>> &dxs)
|
||||
void CartesianPML::StretchFunction(const Vector &x,
|
||||
vector<complex<double>> &dxs)
|
||||
{
|
||||
complex<double> zi = complex<double>(0., 1.);
|
||||
|
||||
|
||||
@@ -1,622 +0,0 @@
|
||||
// MFEM Example 34
|
||||
//
|
||||
// Compile with: make ex34
|
||||
//
|
||||
// Sample runs: ex34 -o 2
|
||||
// ex34 -o 2 -pa -hex
|
||||
//
|
||||
// Device sample runs:
|
||||
// ex34 -o 2 -pa -hex -d cuda
|
||||
// ex34 -o 2 -no-pa -d cuda
|
||||
//
|
||||
// Description: This example code solves a simple magnetostatic problem
|
||||
// curl curl A = J where the current density J is computed on a
|
||||
// subset of the domain as J = -sigma grad phi. We discretize the
|
||||
// vector potential with Nedelec finite elements, the scalar
|
||||
// potential with Lagrange finite elements, and the current
|
||||
// density with Raviart-Thomas finite elements.
|
||||
//
|
||||
// The example demonstrates the use of a SubMesh to compute the
|
||||
// scalar potential and its associated current density which is
|
||||
// then transferred to the original mesh and used as a source
|
||||
// function.
|
||||
//
|
||||
// Note that this example takes certain liberties with the
|
||||
// current density which is not necessarily divergence free
|
||||
// as it should be. This was done to focus on the use of the
|
||||
// SubMesh to transfer information between a full mesh and a
|
||||
// sub-domain. A more rigorous implementation might employ an
|
||||
// H(div) saddle point solver to obtain a divergence free J on
|
||||
// the SubMesh. It would then also need to ensure that the r.h.s.
|
||||
// of curl curl A = J does in fact lie in the range of the weak
|
||||
// curl operator by performing a divergence cleaning procedure
|
||||
// before the solve. After divergence cleaning the delta
|
||||
// parameter would probably not be needed.
|
||||
//
|
||||
// This example is designed to make use of a specific mesh which
|
||||
// has a known configuration of elements and boundary attributes.
|
||||
// Other meshes could be used but extra care would be required to
|
||||
// properly define the SubMesh and the necessary boundaries.
|
||||
//
|
||||
// We recommend viewing examples 1 and 3 before viewing this
|
||||
// example.
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
static bool pa_ = false;
|
||||
static bool algebraic_ceed_ = false;
|
||||
|
||||
void ComputeCurrentDensityOnSubMesh(int order,
|
||||
const Array<int> &phi0_attr,
|
||||
const Array<int> &phi1_attr,
|
||||
const Array<int> &jn_zero_attr,
|
||||
GridFunction &j_cond);
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
// 1. Parse command-line options.
|
||||
const char *mesh_file = "../data/fichera-mixed.mesh";
|
||||
Array<int> cond_attr;
|
||||
Array<int> submesh_elems;
|
||||
Array<int> sym_plane_attr;
|
||||
Array<int> phi0_attr;
|
||||
Array<int> phi1_attr;
|
||||
Array<int> jn_zero_attr;
|
||||
int ref_levels = 1;
|
||||
int order = 1;
|
||||
double delta_const = 1e-6;
|
||||
bool mixed = true;
|
||||
bool static_cond = false;
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = true;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&ref_levels, "-r", "--refine",
|
||||
"Number of times to refine the mesh uniformly.");
|
||||
args.AddOption(&order, "-o", "--order",
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&delta_const, "-mc", "--magnetic-cond",
|
||||
"Magnetic Conductivity");
|
||||
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
|
||||
"--no-static-condensation", "Enable static condensation.");
|
||||
args.AddOption(&mixed, "-mixed", "--mixed-mesh", "-hex",
|
||||
"--hex-mesh", "Mixed mesh of hexahedral mesh.");
|
||||
args.AddOption(&pa_, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
#ifdef MFEM_USE_CEED
|
||||
args.AddOption(&algebraic_ceed_, "-a", "--algebraic", "-no-a", "--no-algebraic",
|
||||
"Use algebraic Ceed solver");
|
||||
#endif
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
return 1;
|
||||
}
|
||||
args.PrintOptions(cout);
|
||||
|
||||
if (!mixed || pa_)
|
||||
{
|
||||
mesh_file = "../data/fichera.mesh";
|
||||
}
|
||||
|
||||
if (submesh_elems.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0)
|
||||
{
|
||||
submesh_elems.SetSize(5);
|
||||
submesh_elems[0] = 0;
|
||||
submesh_elems[1] = 2;
|
||||
submesh_elems[2] = 3;
|
||||
submesh_elems[3] = 4;
|
||||
submesh_elems[4] = 9;
|
||||
}
|
||||
else if (strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
submesh_elems.SetSize(7);
|
||||
submesh_elems[0] = 10;
|
||||
submesh_elems[1] = 14;
|
||||
submesh_elems[2] = 34;
|
||||
submesh_elems[3] = 36;
|
||||
submesh_elems[4] = 37;
|
||||
submesh_elems[5] = 38;
|
||||
submesh_elems[6] = 39;
|
||||
}
|
||||
}
|
||||
if (sym_plane_attr.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
sym_plane_attr.SetSize(8);
|
||||
sym_plane_attr[0] = 9;
|
||||
sym_plane_attr[1] = 10;
|
||||
sym_plane_attr[2] = 11;
|
||||
sym_plane_attr[3] = 12;
|
||||
sym_plane_attr[4] = 13;
|
||||
sym_plane_attr[5] = 14;
|
||||
sym_plane_attr[6] = 15;
|
||||
sym_plane_attr[7] = 16;
|
||||
}
|
||||
}
|
||||
if (phi0_attr.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
phi0_attr.Append(2);
|
||||
}
|
||||
}
|
||||
if (phi1_attr.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
phi1_attr.Append(23);
|
||||
}
|
||||
}
|
||||
if (jn_zero_attr.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
jn_zero_attr.Append(25);
|
||||
}
|
||||
for (int i=0; i<sym_plane_attr.Size(); i++)
|
||||
{
|
||||
jn_zero_attr.Append(sym_plane_attr[i]);
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
device.Print();
|
||||
|
||||
// 3. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
|
||||
// and volume meshes with the same code.
|
||||
Mesh mesh(mesh_file, 1, 1);
|
||||
int dim = mesh.Dimension();
|
||||
|
||||
if (!mixed || pa_)
|
||||
{
|
||||
mesh.UniformRefinement();
|
||||
|
||||
if (ref_levels > 0)
|
||||
{
|
||||
ref_levels--;
|
||||
}
|
||||
}
|
||||
|
||||
int submesh_attr = -1;
|
||||
if (cond_attr.Size() == 0 && submesh_elems.Size() > 0)
|
||||
{
|
||||
int max_attr = mesh.attributes.Max();
|
||||
submesh_attr = max_attr + 1;
|
||||
|
||||
for (int i=0; i<submesh_elems.Size(); i++)
|
||||
{
|
||||
mesh.SetAttribute(submesh_elems[i], submesh_attr);
|
||||
}
|
||||
mesh.SetAttributes();
|
||||
|
||||
if (cond_attr.Size() == 0)
|
||||
{
|
||||
cond_attr.Append(submesh_attr);
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Refine the serial mesh on all processors to increase the resolution. In
|
||||
// this example we do 'ref_levels' of uniform refinement.
|
||||
{
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh.UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
// 5b. Extract a submesh covering a portion of the domain
|
||||
SubMesh mesh_cond(SubMesh::CreateFromDomain(mesh, cond_attr));
|
||||
|
||||
// 6. Define a suitable finite element space on the SubMesh and compute
|
||||
// the current density as an H(div) field.
|
||||
RT_FECollection fec_cond_rt(order - 1, dim);
|
||||
FiniteElementSpace fes_cond_rt(&mesh_cond, &fec_cond_rt);
|
||||
GridFunction j_cond(&fes_cond_rt);
|
||||
|
||||
ComputeCurrentDensityOnSubMesh(order, phi0_attr, phi1_attr, jn_zero_attr,
|
||||
j_cond);
|
||||
|
||||
// 6a. Save the SubMesh and associated current density in parallel. This
|
||||
// output can be viewed later using GLVis:
|
||||
// "glvis -np <np> -m cond_mesh -g cond_j"
|
||||
{
|
||||
ostringstream mesh_name, cond_name;
|
||||
mesh_name << "cond.mesh";
|
||||
cond_name << "cond_j.gf";
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
mesh_cond.Print(mesh_ofs);
|
||||
|
||||
ofstream cond_ofs(cond_name.str().c_str());
|
||||
cond_ofs.precision(8);
|
||||
j_cond.Save(cond_ofs);
|
||||
}
|
||||
// 6b. Send the current density, computed on the SubMesh, to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream port_sock(vishost, visport);
|
||||
port_sock.precision(8);
|
||||
port_sock << "solution\n" << mesh_cond << j_cond
|
||||
<< "window_title 'Conductor J'"
|
||||
<< "window_geometry 400 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// 7. Define a parallel finite element space on the full mesh. Here we
|
||||
// use the H(curl) finite elements for the vector potential and H(div)
|
||||
// for the current density.
|
||||
ND_FECollection fec_nd(order, dim);
|
||||
RT_FECollection fec_rt(order - 1, dim);
|
||||
FiniteElementSpace fespace_nd(&mesh, &fec_nd);
|
||||
FiniteElementSpace fespace_rt(&mesh, &fec_rt);
|
||||
|
||||
GridFunction j_full(&fespace_rt);
|
||||
j_full = 0.0;
|
||||
mesh_cond.Transfer(j_cond, j_full);
|
||||
|
||||
// 7a. Send the transferred current density to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << j_full
|
||||
<< "window_title 'J Full'"
|
||||
<< "window_geometry 400 430 400 350" << flush;
|
||||
}
|
||||
|
||||
// 8. Determine the list of true (i.e. parallel conforming) essential
|
||||
// boundary dofs. In this example, the boundary conditions are defined
|
||||
// by marking all the boundary attributes except for those on a symmetry
|
||||
// plane as essential (Dirichlet) and converting them to a list of
|
||||
// true dofs.
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr;
|
||||
if (mesh.bdr_attributes.Size())
|
||||
{
|
||||
ess_bdr.SetSize(mesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
for (int i=0; i<sym_plane_attr.Size(); i++)
|
||||
{
|
||||
ess_bdr[sym_plane_attr[i]-1] = 0;
|
||||
}
|
||||
fespace_nd.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
// 9. Set up the parallel linear form b(.) which corresponds to the
|
||||
// right-hand side of the FEM linear system, which in this case is
|
||||
// (J,W_i) where J is given by the function H(div) field transferred
|
||||
// from the SubMesh and W_i are the basis functions in the finite
|
||||
// element fespace.
|
||||
VectorGridFunctionCoefficient jCoef(&j_full);
|
||||
LinearForm b(&fespace_nd);
|
||||
b.AddDomainIntegrator(new VectorFEDomainLFIntegrator(jCoef));
|
||||
b.Assemble();
|
||||
|
||||
// 10. Define the solution vector x as a parallel finite element grid
|
||||
// function corresponding to fespace. Initialize x to zero.
|
||||
GridFunction x(&fespace_nd);
|
||||
x = 0.0;
|
||||
|
||||
// 11. Set up the parallel bilinear form corresponding to the EM
|
||||
// diffusion operator curl muinv curl + delta I, by adding the
|
||||
// curl-curl and the mass domain integrators. For standard
|
||||
// magnetostatics equations choose delta << 1. Larger values of
|
||||
// delta should make the linear system easier to solve at the
|
||||
// expense of resembling a diffusive quasistatic magnetic field.
|
||||
// A reasonable balance must be found whenever the mesh or problem
|
||||
// setup is altered.
|
||||
ConstantCoefficient muinv(1.0);
|
||||
ConstantCoefficient delta(delta_const);
|
||||
BilinearForm a(&fespace_nd);
|
||||
if (pa_) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a.AddDomainIntegrator(new CurlCurlIntegrator(muinv));
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(delta));
|
||||
|
||||
// 12. Assemble the parallel bilinear form and the corresponding linear
|
||||
// system, applying any necessary transformations such as: parallel
|
||||
// assembly, eliminating boundary conditions, applying conforming
|
||||
// constraints for non-conforming AMR, static condensation, etc.
|
||||
if (static_cond) { a.EnableStaticCondensation(); }
|
||||
a.Assemble();
|
||||
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
|
||||
// 13. Solve the system AX=B
|
||||
if (pa_) // Jacobi preconditioning in partial assembly mode
|
||||
{
|
||||
cout << "\nSolving for magnetic vector potential "
|
||||
<< "using CG with a Jacobi preconditioner" << endl;
|
||||
|
||||
OperatorJacobiSmoother M(a, ess_tdof_list);
|
||||
PCG(*A, M, B, X, 1, 1000, 1e-12, 0.0);
|
||||
}
|
||||
else
|
||||
{
|
||||
#ifndef MFEM_USE_SUITESPARSE
|
||||
cout << "\nSolving for magnetic vector potential "
|
||||
<< "using CG with a Gauss-Seidel preconditioner" << endl;
|
||||
|
||||
// 13a. Define a simple symmetric Gauss-Seidel preconditioner and use
|
||||
// it to solve the system Ax=b with PCG.
|
||||
GSSmoother M((SparseMatrix&)(*A));
|
||||
PCG(*A, M, B, X, 1, 500, 1e-12, 0.0);
|
||||
#else
|
||||
cout << "\nSolving for magnetic vector potential "
|
||||
<< "using UMFPack" << endl;
|
||||
|
||||
// 13a. If MFEM was compiled with SuiteSparse, use UMFPACK to solve the
|
||||
// system.
|
||||
UMFPackSolver umf_solver;
|
||||
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
umf_solver.SetOperator(*A);
|
||||
umf_solver.Mult(B, X);
|
||||
#endif
|
||||
}
|
||||
|
||||
// 14. Recover the parallel grid function corresponding to X. This is the
|
||||
// local finite element solution on each processor.
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
|
||||
// 15. Save the refined mesh and the solution in parallel. This output can
|
||||
// be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
|
||||
{
|
||||
ostringstream mesh_name, sol_name;
|
||||
mesh_name << "refined.mesh";
|
||||
sol_name << "sol.gf";
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
mesh.Print(mesh_ofs);
|
||||
|
||||
ofstream sol_ofs(sol_name.str().c_str());
|
||||
sol_ofs.precision(8);
|
||||
x.Save(sol_ofs);
|
||||
}
|
||||
|
||||
// 16. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << x
|
||||
<< "window_title 'Vector Potential'"
|
||||
<< "window_geometry 800 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// 17. Compute the magnetic flux as the curl of the solution
|
||||
DiscreteLinearOperator curl(&fespace_nd, &fespace_rt);
|
||||
curl.AddDomainInterpolator(new CurlInterpolator);
|
||||
curl.Assemble();
|
||||
curl.Finalize();
|
||||
|
||||
GridFunction dx(&fespace_rt);
|
||||
curl.Mult(x, dx);
|
||||
|
||||
// 18. Save the curl of the solution in parallel. This output can
|
||||
// be viewed later using GLVis: "glvis -np <np> -m mesh -g dsol".
|
||||
{
|
||||
ostringstream dsol_name;
|
||||
dsol_name << "dsol.gf";
|
||||
|
||||
ofstream dsol_ofs(dsol_name.str().c_str());
|
||||
dsol_ofs.precision(8);
|
||||
dx.Save(dsol_ofs);
|
||||
}
|
||||
|
||||
// 19. Send the curl of the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << dx
|
||||
<< "window_title 'Magnetic Flux'"
|
||||
<< "window_geometry 1200 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// 20. Clean exit
|
||||
return 0;
|
||||
}
|
||||
|
||||
void ComputeCurrentDensityOnSubMesh(int order,
|
||||
const Array<int> &phi0_attr,
|
||||
const Array<int> &phi1_attr,
|
||||
const Array<int> &jn_zero_attr,
|
||||
GridFunction &j_cond)
|
||||
{
|
||||
// Exract the finite element space and mesh on which j_cond is defined
|
||||
FiniteElementSpace &fes_cond_rt = *j_cond.FESpace();
|
||||
Mesh &mesh_cond = *fes_cond_rt.GetMesh();
|
||||
int dim = mesh_cond.Dimension();
|
||||
|
||||
// Define a parallel finite element space on the SubMesh. Here we use the
|
||||
// H1 finite elements for the electrostatic potential.
|
||||
H1_FECollection fec_h1(order, dim);
|
||||
FiniteElementSpace fes_cond_h1(&mesh_cond, &fec_h1);
|
||||
|
||||
// Define the conductivity coefficient and the boundaries associated with
|
||||
// the fixed potentials phi0 and phi1 which will drive the current.
|
||||
ConstantCoefficient sigmaCoef(1.0);
|
||||
Array<int> ess_bdr_phi(mesh_cond.bdr_attributes.Max());
|
||||
Array<int> ess_bdr_j(mesh_cond.bdr_attributes.Max());
|
||||
Array<int> ess_bdr_tdof_phi;
|
||||
ess_bdr_phi = 0;
|
||||
ess_bdr_j = 0;
|
||||
for (int i=0; i<phi0_attr.Size(); i++)
|
||||
{
|
||||
ess_bdr_phi[phi0_attr[i]-1] = 1;
|
||||
}
|
||||
for (int i=0; i<phi1_attr.Size(); i++)
|
||||
{
|
||||
ess_bdr_phi[phi1_attr[i]-1] = 1;
|
||||
}
|
||||
for (int i=0; i<jn_zero_attr.Size(); i++)
|
||||
{
|
||||
ess_bdr_j[jn_zero_attr[i]-1] = 1;
|
||||
}
|
||||
fes_cond_h1.GetEssentialTrueDofs(ess_bdr_phi, ess_bdr_tdof_phi);
|
||||
|
||||
// Setup the bilinear form corresponding to -Div(sigma Grad phi)
|
||||
BilinearForm a_h1(&fes_cond_h1);
|
||||
a_h1.AddDomainIntegrator(new DiffusionIntegrator(sigmaCoef));
|
||||
a_h1.Assemble();
|
||||
|
||||
// Set the r.h.s. to zero
|
||||
LinearForm b_h1(&fes_cond_h1);
|
||||
b_h1 = 0.0;
|
||||
|
||||
// Setup the boundary conditions on phi
|
||||
ConstantCoefficient one(1.0);
|
||||
ConstantCoefficient zero(0.0);
|
||||
GridFunction phi_h1(&fes_cond_h1);
|
||||
phi_h1 = 0.0;
|
||||
|
||||
Array<int> bdr0(mesh_cond.bdr_attributes.Max()); bdr0 = 0;
|
||||
for (int i=0; i<phi0_attr.Size(); i++)
|
||||
{
|
||||
bdr0[phi0_attr[i]-1] = 1;
|
||||
}
|
||||
phi_h1.ProjectBdrCoefficient(zero, bdr0);
|
||||
|
||||
Array<int> bdr1(mesh_cond.bdr_attributes.Max()); bdr1 = 0;
|
||||
for (int i=0; i<phi1_attr.Size(); i++)
|
||||
{
|
||||
bdr1[phi1_attr[i]-1] = 1;
|
||||
}
|
||||
phi_h1.ProjectBdrCoefficient(one, bdr1);
|
||||
|
||||
{
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a_h1.FormLinearSystem(ess_bdr_tdof_phi, phi_h1, b_h1, A, X, B);
|
||||
|
||||
// Solve the linear system
|
||||
if (!pa_)
|
||||
{
|
||||
#ifndef MFEM_USE_SUITESPARSE
|
||||
cout << "\nSolving for electric potential using PCG "
|
||||
<< "with a Gauss-Seidel preconditioner" << endl;
|
||||
|
||||
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
|
||||
GSSmoother M((SparseMatrix&)(*A));
|
||||
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
|
||||
#else
|
||||
cout << "\nSolving for electric potential using UMFPack" << endl;
|
||||
|
||||
// If MFEM was compiled with SuiteSparse,
|
||||
// use UMFPACK to solve the system.
|
||||
UMFPackSolver umf_solver;
|
||||
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
umf_solver.SetOperator(*A);
|
||||
umf_solver.Mult(B, X);
|
||||
#endif
|
||||
}
|
||||
else
|
||||
{
|
||||
cout << "\nSolving for electric potential using CG" << endl;
|
||||
|
||||
if (UsesTensorBasis(fes_cond_h1))
|
||||
{
|
||||
if (algebraic_ceed_)
|
||||
{
|
||||
ceed::AlgebraicSolver M(a_h1, ess_bdr_tdof_phi);
|
||||
PCG(*A, M, B, X, 1, 400, 1e-12, 0.0);
|
||||
}
|
||||
else
|
||||
{
|
||||
OperatorJacobiSmoother M(a_h1, ess_bdr_tdof_phi);
|
||||
PCG(*A, M, B, X, 1, 400, 1e-12, 0.0);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
CG(*A, B, X, 1, 400, 1e-12, 0.0);
|
||||
}
|
||||
}
|
||||
a_h1.RecoverFEMSolution(X, b_h1, phi_h1);
|
||||
}
|
||||
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream port_sock(vishost, visport);
|
||||
port_sock.precision(8);
|
||||
port_sock << "solution\n" << mesh_cond << phi_h1
|
||||
<< "window_title 'Conductor Potential'"
|
||||
<< "window_geometry 0 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// Solve for the current density J = -sigma Grad phi with boundary
|
||||
// conditions J.n = 0 on the walls of the conductor but not on the
|
||||
// ports where phi=0 and phi=1.
|
||||
|
||||
// J will be computed in H(div) so we need an RT mass matrix
|
||||
BilinearForm m_rt(&fes_cond_rt);
|
||||
m_rt.AddDomainIntegrator(new VectorFEMassIntegrator);
|
||||
m_rt.Assemble();
|
||||
|
||||
// Assemble the (sigma Grad phi) operator
|
||||
MixedBilinearForm d_h1(&fes_cond_h1, &fes_cond_rt);
|
||||
d_h1.AddDomainIntegrator(new MixedVectorGradientIntegrator(sigmaCoef));
|
||||
d_h1.Assemble();
|
||||
|
||||
// Compute the r.h.s, b_rt = sigma E = -sigma Grad phi
|
||||
LinearForm b_rt(&fes_cond_rt);
|
||||
d_h1.Mult(phi_h1, b_rt);
|
||||
b_rt *= -1.0;
|
||||
|
||||
// Apply the necessary boundary conditions and solve for J in H(div)
|
||||
cout << "\nSolving for current density in H(Div) "
|
||||
<< "using diagonally scaled CG" << endl;
|
||||
cout << "Size of linear system: "
|
||||
<< fes_cond_rt.GetTrueVSize() << endl;
|
||||
|
||||
Array<int> ess_bdr_tdof_rt;
|
||||
OperatorPtr M;
|
||||
Vector B, X;
|
||||
|
||||
fes_cond_rt.GetEssentialTrueDofs(ess_bdr_j, ess_bdr_tdof_rt);
|
||||
|
||||
j_cond = 0.0;
|
||||
m_rt.FormLinearSystem(ess_bdr_tdof_rt, j_cond, b_rt, M, X, B);
|
||||
|
||||
CGSolver cg;
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetPrintLevel(1);
|
||||
cg.SetOperator(*M);
|
||||
cg.Mult(B, X);
|
||||
m_rt.RecoverFEMSolution(X, b_rt, j_cond);
|
||||
}
|
||||
@@ -1,649 +0,0 @@
|
||||
// MFEM Example 34 - Parallel Version
|
||||
//
|
||||
// Compile with: make ex34p
|
||||
//
|
||||
// Sample runs: mpirun -np 4 ex34p -o 2
|
||||
// mpirun -np 4 ex34p -o 2 -hex -pa
|
||||
//
|
||||
// Device sample runs:
|
||||
// mpirun -np 4 ex34p -o 2 -hex -pa -d cuda
|
||||
// mpirun -np 4 ex34p -o 2 -no-pa -d cuda
|
||||
//
|
||||
// Description: This example code solves a simple magnetostatic problem
|
||||
// curl curl A = J where the current density J is computed on a
|
||||
// subset of the domain as J = -sigma grad phi. We discretize the
|
||||
// vector potential with Nedelec finite elements, the scalar
|
||||
// potential with Lagrange finite elements, and the current
|
||||
// density with Raviart-Thomas finite elements.
|
||||
//
|
||||
// The example demonstrates the use of a SubMesh to compute the
|
||||
// scalar potential and its associated current density which is
|
||||
// then transferred to the original mesh and used as a source
|
||||
// function.
|
||||
//
|
||||
// Note that this example takes certain liberties with the
|
||||
// current density which is not necessarily divergence free
|
||||
// as it should be. This was done to focus on the use of the
|
||||
// SubMesh to transfer information between a full mesh and a
|
||||
// sub-domain. A more rigorous implementation might employ an
|
||||
// H(div) saddle point solver to obtain a divergence free J on
|
||||
// the SubMesh. It would then also need to ensure that the r.h.s.
|
||||
// of curl curl A = J does in fact lie in the range of the weak
|
||||
// curl operator by performing a divergence cleaning procedure
|
||||
// before the solve. After divergence cleaning the delta
|
||||
// parameter would probably not be needed.
|
||||
//
|
||||
// This example is designed to make use of a specific mesh which
|
||||
// has a known configuration of elements and boundary attributes.
|
||||
// Other meshes could be used but extra care would be required to
|
||||
// properly define the SubMesh and the necessary boundaries.
|
||||
//
|
||||
// We recommend viewing examples 1 and 3 before viewing this
|
||||
// example.
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
void ComputeCurrentDensityOnSubMesh(int order,
|
||||
const Array<int> &phi0_attr,
|
||||
const Array<int> &phi1_attr,
|
||||
const Array<int> &jn_zero_attr,
|
||||
ParGridFunction &j_cond);
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
// 1. Initialize MPI and HYPRE.
|
||||
Mpi::Init(argc, argv);
|
||||
int num_procs = Mpi::WorldSize();
|
||||
int myid = Mpi::WorldRank();
|
||||
Hypre::Init();
|
||||
|
||||
// 2. Parse command-line options.
|
||||
const char *mesh_file = "../data/fichera-mixed.mesh";
|
||||
Array<int> cond_attr;
|
||||
Array<int> submesh_elems;
|
||||
Array<int> sym_plane_attr;
|
||||
Array<int> phi0_attr;
|
||||
Array<int> phi1_attr;
|
||||
Array<int> jn_zero_attr;
|
||||
int ser_ref_levels = 1;
|
||||
int par_ref_levels = 1;
|
||||
int order = 1;
|
||||
double delta_const = 1e-6;
|
||||
bool mixed = true;
|
||||
bool static_cond = false;
|
||||
bool pa = false;
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = true;
|
||||
#ifdef MFEM_USE_AMGX
|
||||
bool useAmgX = false;
|
||||
#endif
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&ser_ref_levels, "-rs", "--refine-serial",
|
||||
"Number of times to refine the mesh uniformly in serial.");
|
||||
args.AddOption(&par_ref_levels, "-rp", "--refine-parallel",
|
||||
"Number of times to refine the mesh uniformly in parallel.");
|
||||
args.AddOption(&order, "-o", "--order",
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&delta_const, "-mc", "--magnetic-cond",
|
||||
"Magnetic Conductivity");
|
||||
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
|
||||
"--no-static-condensation", "Enable static condensation.");
|
||||
args.AddOption(&mixed, "-mixed", "--mixed-mesh", "-hex",
|
||||
"--hex-mesh", "Mixed mesh of hexahedral mesh.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
#ifdef MFEM_USE_AMGX
|
||||
args.AddOption(&useAmgX, "-amgx", "--useAmgX", "-no-amgx",
|
||||
"--no-useAmgX",
|
||||
"Enable or disable AmgX in MatrixFreeAMS.");
|
||||
#endif
|
||||
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
if (!mixed || pa)
|
||||
{
|
||||
mesh_file = "../data/fichera.mesh";
|
||||
}
|
||||
|
||||
if (submesh_elems.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0)
|
||||
{
|
||||
submesh_elems.SetSize(5);
|
||||
submesh_elems[0] = 0;
|
||||
submesh_elems[1] = 2;
|
||||
submesh_elems[2] = 3;
|
||||
submesh_elems[3] = 4;
|
||||
submesh_elems[4] = 9;
|
||||
}
|
||||
else if (strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
submesh_elems.SetSize(7);
|
||||
submesh_elems[0] = 10;
|
||||
submesh_elems[1] = 14;
|
||||
submesh_elems[2] = 34;
|
||||
submesh_elems[3] = 36;
|
||||
submesh_elems[4] = 37;
|
||||
submesh_elems[5] = 38;
|
||||
submesh_elems[6] = 39;
|
||||
}
|
||||
}
|
||||
if (sym_plane_attr.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
sym_plane_attr.SetSize(8);
|
||||
sym_plane_attr[0] = 9;
|
||||
sym_plane_attr[1] = 10;
|
||||
sym_plane_attr[2] = 11;
|
||||
sym_plane_attr[3] = 12;
|
||||
sym_plane_attr[4] = 13;
|
||||
sym_plane_attr[5] = 14;
|
||||
sym_plane_attr[6] = 15;
|
||||
sym_plane_attr[7] = 16;
|
||||
}
|
||||
}
|
||||
if (phi0_attr.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
phi0_attr.Append(2);
|
||||
}
|
||||
}
|
||||
if (phi1_attr.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
phi1_attr.Append(23);
|
||||
}
|
||||
}
|
||||
if (jn_zero_attr.Size() == 0)
|
||||
{
|
||||
if (strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0)
|
||||
{
|
||||
jn_zero_attr.Append(25);
|
||||
}
|
||||
for (int i=0; i<sym_plane_attr.Size(); i++)
|
||||
{
|
||||
jn_zero_attr.Append(sym_plane_attr[i]);
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
if (myid == 0) { device.Print(); }
|
||||
|
||||
// 4. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
|
||||
// and volume meshes with the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
if (!mixed || pa)
|
||||
{
|
||||
mesh->UniformRefinement();
|
||||
|
||||
if (ser_ref_levels > 0)
|
||||
{
|
||||
ser_ref_levels--;
|
||||
}
|
||||
else
|
||||
{
|
||||
par_ref_levels--;
|
||||
}
|
||||
}
|
||||
|
||||
int submesh_attr = -1;
|
||||
if (cond_attr.Size() == 0 && submesh_elems.Size() > 0)
|
||||
{
|
||||
int max_attr = mesh->attributes.Max();
|
||||
submesh_attr = max_attr + 1;
|
||||
|
||||
for (int i=0; i<submesh_elems.Size(); i++)
|
||||
{
|
||||
mesh->SetAttribute(submesh_elems[i], submesh_attr);
|
||||
}
|
||||
mesh->SetAttributes();
|
||||
|
||||
if (cond_attr.Size() == 0)
|
||||
{
|
||||
cond_attr.Append(submesh_attr);
|
||||
}
|
||||
}
|
||||
|
||||
// 5. Refine the serial mesh on all processors to increase the resolution. In
|
||||
// this example we do 'ref_levels' of uniform refinement.
|
||||
{
|
||||
int ref_levels = ser_ref_levels;
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh->UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// this mesh further in parallel to increase the resolution. Once the
|
||||
// parallel mesh is defined, the serial mesh can be deleted.
|
||||
ParMesh pmesh(MPI_COMM_WORLD, *mesh);
|
||||
delete mesh;
|
||||
{
|
||||
for (int l = 0; l < par_ref_levels; l++)
|
||||
{
|
||||
pmesh.UniformRefinement();
|
||||
}
|
||||
}
|
||||
|
||||
// 6b. Extract a submesh covering a portion of the domain
|
||||
ParSubMesh pmesh_cond(ParSubMesh::CreateFromDomain(pmesh, cond_attr));
|
||||
|
||||
// 7. Define a suitable finite element space on the SubMesh and compute
|
||||
// the current density as an H(div) field.
|
||||
RT_FECollection fec_cond_rt(order - 1, dim);
|
||||
ParFiniteElementSpace fes_cond_rt(&pmesh_cond, &fec_cond_rt);
|
||||
ParGridFunction j_cond(&fes_cond_rt);
|
||||
|
||||
ComputeCurrentDensityOnSubMesh(order, phi0_attr, phi1_attr, jn_zero_attr,
|
||||
j_cond);
|
||||
|
||||
// 7a. Save the SubMesh and associated current density in parallel. This
|
||||
// output can be viewed later using GLVis:
|
||||
// "glvis -np <np> -m cond_mesh -g cond_j"
|
||||
{
|
||||
ostringstream mesh_name, cond_name;
|
||||
mesh_name << "cond_mesh." << setfill('0') << setw(6) << myid;
|
||||
cond_name << "cond_j." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh_cond.Print(mesh_ofs);
|
||||
|
||||
ofstream cond_ofs(cond_name.str().c_str());
|
||||
cond_ofs.precision(8);
|
||||
j_cond.Save(cond_ofs);
|
||||
}
|
||||
// 7b. Send the current density, computed on the SubMesh, to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream port_sock(vishost, visport);
|
||||
port_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
port_sock.precision(8);
|
||||
port_sock << "solution\n" << pmesh_cond << j_cond
|
||||
<< "window_title 'Conductor J'"
|
||||
<< "window_geometry 400 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// 8. Define a parallel finite element space on the full mesh. Here we
|
||||
// use the H(curl) finite elements for the vector potential and H(div)
|
||||
// for the current density.
|
||||
ND_FECollection fec_nd(order, dim);
|
||||
RT_FECollection fec_rt(order - 1, dim);
|
||||
ParFiniteElementSpace fespace_nd(&pmesh, &fec_nd);
|
||||
ParFiniteElementSpace fespace_rt(&pmesh, &fec_rt);
|
||||
|
||||
ParGridFunction j_full(&fespace_rt);
|
||||
j_full = 0.0;
|
||||
pmesh_cond.Transfer(j_cond, j_full);
|
||||
|
||||
// 8a. Send the transferred current density to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << pmesh << j_full
|
||||
<< "window_title 'J Full'"
|
||||
<< "window_geometry 400 430 400 350" << flush;
|
||||
}
|
||||
|
||||
// 9. Determine the list of true (i.e. parallel conforming) essential
|
||||
// boundary dofs. In this example, the boundary conditions are defined
|
||||
// by marking all the boundary attributes except for those on a symmetry
|
||||
// plane as essential (Dirichlet) and converting them to a list of
|
||||
// true dofs.
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr;
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
{
|
||||
ess_bdr.SetSize(pmesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
for (int i=0; i<sym_plane_attr.Size(); i++)
|
||||
{
|
||||
ess_bdr[sym_plane_attr[i]-1] = 0;
|
||||
}
|
||||
fespace_nd.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
// 10. Set up the parallel linear form b(.) which corresponds to the
|
||||
// right-hand side of the FEM linear system, which in this case is
|
||||
// (J,W_i) where J is given by the function H(div) field transferred
|
||||
// from the SubMesh and W_i are the basis functions in the finite
|
||||
// element fespace.
|
||||
VectorGridFunctionCoefficient jCoef(&j_full);
|
||||
ParLinearForm b(&fespace_nd);
|
||||
b.AddDomainIntegrator(new VectorFEDomainLFIntegrator(jCoef));
|
||||
b.Assemble();
|
||||
|
||||
// 11. Define the solution vector x as a parallel finite element grid
|
||||
// function corresponding to fespace. Initialize x to zero.
|
||||
ParGridFunction x(&fespace_nd);
|
||||
x = 0.0;
|
||||
|
||||
// 12. Set up the parallel bilinear form corresponding to the EM
|
||||
// diffusion operator curl muinv curl + delta I, by adding the
|
||||
// curl-curl and the mass domain integrators. For standard
|
||||
// magnetostatics equations choose delta << 1. Larger values of
|
||||
// delta should make the linear system easier to solve at the
|
||||
// expense of resembling a diffusive quasistatic magnetic field.
|
||||
// A reasonable balance must be found whenever the mesh or problem
|
||||
// setup is altered.
|
||||
ConstantCoefficient muinv(1.0);
|
||||
ConstantCoefficient delta(delta_const);
|
||||
ParBilinearForm a(&fespace_nd);
|
||||
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
a.AddDomainIntegrator(new CurlCurlIntegrator(muinv));
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(delta));
|
||||
|
||||
// 13. Assemble the parallel bilinear form and the corresponding linear
|
||||
// system, applying any necessary transformations such as: parallel
|
||||
// assembly, eliminating boundary conditions, applying conforming
|
||||
// constraints for non-conforming AMR, static condensation, etc.
|
||||
if (static_cond) { a.EnableStaticCondensation(); }
|
||||
a.Assemble();
|
||||
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "\nSolving for magnetic vector potential "
|
||||
<< "using CG with AMS" << endl;
|
||||
}
|
||||
|
||||
// 14. Solve the system AX=B using PCG with an AMS preconditioner.
|
||||
if (pa)
|
||||
{
|
||||
#ifdef MFEM_USE_AMGX
|
||||
MatrixFreeAMS ams(a, *A, fespace_nd, &muinv, &delta, NULL, ess_bdr,
|
||||
useAmgX);
|
||||
#else
|
||||
MatrixFreeAMS ams(a, *A, fespace_nd, &muinv, &delta, NULL, ess_bdr);
|
||||
#endif
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(1000);
|
||||
cg.SetPrintLevel(1);
|
||||
cg.SetOperator(*A);
|
||||
cg.SetPreconditioner(ams);
|
||||
cg.Mult(B, X);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Size of linear system: "
|
||||
<< A.As<HypreParMatrix>()->GetGlobalNumRows() << endl;
|
||||
}
|
||||
|
||||
ParFiniteElementSpace *prec_fespace =
|
||||
(a.StaticCondensationIsEnabled() ? a.SCParFESpace() : &fespace_nd);
|
||||
HypreAMS ams(*A.As<HypreParMatrix>(), prec_fespace);
|
||||
HyprePCG pcg(*A.As<HypreParMatrix>());
|
||||
pcg.SetTol(1e-12);
|
||||
pcg.SetMaxIter(500);
|
||||
pcg.SetPrintLevel(2);
|
||||
pcg.SetPreconditioner(ams);
|
||||
pcg.Mult(B, X);
|
||||
}
|
||||
|
||||
// 15. Recover the parallel grid function corresponding to X. This is the
|
||||
// local finite element solution on each processor.
|
||||
a.RecoverFEMSolution(X, b, x);
|
||||
|
||||
// 16. Save the refined mesh and the solution in parallel. This output can
|
||||
// be viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
|
||||
{
|
||||
ostringstream mesh_name, sol_name;
|
||||
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
|
||||
sol_name << "sol." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh.Print(mesh_ofs);
|
||||
|
||||
ofstream sol_ofs(sol_name.str().c_str());
|
||||
sol_ofs.precision(8);
|
||||
x.Save(sol_ofs);
|
||||
}
|
||||
|
||||
// 17. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << pmesh << x
|
||||
<< "window_title 'Vector Potential'"
|
||||
<< "window_geometry 800 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// 18. Compute the magnetic flux as the curl of the solution
|
||||
ParDiscreteLinearOperator curl(&fespace_nd, &fespace_rt);
|
||||
curl.AddDomainInterpolator(new CurlInterpolator);
|
||||
curl.Assemble();
|
||||
curl.Finalize();
|
||||
|
||||
ParGridFunction dx(&fespace_rt);
|
||||
curl.Mult(x, dx);
|
||||
|
||||
// 19. Save the curl of the solution in parallel. This output can
|
||||
// be viewed later using GLVis: "glvis -np <np> -m mesh -g dsol".
|
||||
{
|
||||
ostringstream dsol_name;
|
||||
dsol_name << "dsol." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream dsol_ofs(dsol_name.str().c_str());
|
||||
dsol_ofs.precision(8);
|
||||
dx.Save(dsol_ofs);
|
||||
}
|
||||
|
||||
// 20. Send the curl of the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << pmesh << dx
|
||||
<< "window_title 'Magnetic Flux'"
|
||||
<< "window_geometry 1200 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// 21. Clean exit
|
||||
return 0;
|
||||
}
|
||||
|
||||
void ComputeCurrentDensityOnSubMesh(int order,
|
||||
const Array<int> &phi0_attr,
|
||||
const Array<int> &phi1_attr,
|
||||
const Array<int> &jn_zero_attr,
|
||||
ParGridFunction &j_cond)
|
||||
{
|
||||
// Exract the finite element space and mesh on which j_cond is defined
|
||||
ParFiniteElementSpace &fes_cond_rt = *j_cond.ParFESpace();
|
||||
ParMesh &pmesh_cond = *fes_cond_rt.GetParMesh();
|
||||
int myid = fes_cond_rt.GetMyRank();
|
||||
int dim = pmesh_cond.Dimension();
|
||||
|
||||
// Define a parallel finite element space on the SubMesh. Here we use the
|
||||
// H1 finite elements for the electrostatic potential.
|
||||
H1_FECollection fec_h1(order, dim);
|
||||
ParFiniteElementSpace fes_cond_h1(&pmesh_cond, &fec_h1);
|
||||
|
||||
// Define the conductivity coefficient and the boundaries associated with
|
||||
// the fixed potentials phi0 and phi1 which will drive the current.
|
||||
ConstantCoefficient sigmaCoef(1.0);
|
||||
Array<int> ess_bdr_phi(pmesh_cond.bdr_attributes.Max());
|
||||
Array<int> ess_bdr_j(pmesh_cond.bdr_attributes.Max());
|
||||
Array<int> ess_bdr_tdof_phi;
|
||||
ess_bdr_phi = 0;
|
||||
ess_bdr_j = 0;
|
||||
for (int i=0; i<phi0_attr.Size(); i++)
|
||||
{
|
||||
ess_bdr_phi[phi0_attr[i]-1] = 1;
|
||||
}
|
||||
for (int i=0; i<phi1_attr.Size(); i++)
|
||||
{
|
||||
ess_bdr_phi[phi1_attr[i]-1] = 1;
|
||||
}
|
||||
for (int i=0; i<jn_zero_attr.Size(); i++)
|
||||
{
|
||||
ess_bdr_j[jn_zero_attr[i]-1] = 1;
|
||||
}
|
||||
fes_cond_h1.GetEssentialTrueDofs(ess_bdr_phi, ess_bdr_tdof_phi);
|
||||
|
||||
// Setup the bilinear form corresponding to -Div(sigma Grad phi)
|
||||
ParBilinearForm a_h1(&fes_cond_h1);
|
||||
a_h1.AddDomainIntegrator(new DiffusionIntegrator(sigmaCoef));
|
||||
a_h1.Assemble();
|
||||
|
||||
// Set the r.h.s. to zero
|
||||
ParLinearForm b_h1(&fes_cond_h1);
|
||||
b_h1 = 0.0;
|
||||
|
||||
// Setup the boundary conditions on phi
|
||||
ConstantCoefficient one(1.0);
|
||||
ConstantCoefficient zero(0.0);
|
||||
ParGridFunction phi_h1(&fes_cond_h1);
|
||||
phi_h1 = 0.0;
|
||||
|
||||
Array<int> bdr0(pmesh_cond.bdr_attributes.Max()); bdr0 = 0;
|
||||
for (int i=0; i<phi0_attr.Size(); i++)
|
||||
{
|
||||
bdr0[phi0_attr[i]-1] = 1;
|
||||
}
|
||||
phi_h1.ProjectBdrCoefficient(zero, bdr0);
|
||||
|
||||
Array<int> bdr1(pmesh_cond.bdr_attributes.Max()); bdr1 = 0;
|
||||
for (int i=0; i<phi1_attr.Size(); i++)
|
||||
{
|
||||
bdr1[phi1_attr[i]-1] = 1;
|
||||
}
|
||||
phi_h1.ProjectBdrCoefficient(one, bdr1);
|
||||
|
||||
// Solve the linear system using algebraic multigrid
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "\nSolving for electric potential "
|
||||
<< "using CG with AMG" << endl;
|
||||
}
|
||||
OperatorPtr A;
|
||||
Vector B, X;
|
||||
a_h1.FormLinearSystem(ess_bdr_tdof_phi, phi_h1, b_h1, A, X, B);
|
||||
|
||||
HypreBoomerAMG prec;
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetPrintLevel(1);
|
||||
cg.SetPreconditioner(prec);
|
||||
cg.SetOperator(*A);
|
||||
cg.Mult(B, X);
|
||||
a_h1.RecoverFEMSolution(X, b_h1, phi_h1);
|
||||
}
|
||||
{
|
||||
int num_procs = fes_cond_h1.GetNRanks();
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream port_sock(vishost, visport);
|
||||
port_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
port_sock.precision(8);
|
||||
port_sock << "solution\n" << pmesh_cond << phi_h1
|
||||
<< "window_title 'Conductor Potential'"
|
||||
<< "window_geometry 0 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// Solve for the current density J = -sigma Grad phi with boundary
|
||||
// conditions J.n = 0 on the walls of the conductor but not on the
|
||||
// ports where phi=0 and phi=1.
|
||||
|
||||
// J will be computed in H(div) so we need an RT mass matrix
|
||||
ParBilinearForm m_rt(&fes_cond_rt);
|
||||
m_rt.AddDomainIntegrator(new VectorFEMassIntegrator);
|
||||
m_rt.Assemble();
|
||||
|
||||
// Assemble the (sigma Grad phi) operator
|
||||
ParMixedBilinearForm d_h1(&fes_cond_h1, &fes_cond_rt);
|
||||
d_h1.AddDomainIntegrator(new MixedVectorGradientIntegrator(sigmaCoef));
|
||||
d_h1.Assemble();
|
||||
|
||||
// Compute the r.h.s, b_rt = sigma E = -sigma Grad phi
|
||||
ParLinearForm b_rt(&fes_cond_rt);
|
||||
d_h1.Mult(phi_h1, b_rt);
|
||||
b_rt *= -1.0;
|
||||
|
||||
// Apply the necessary boundary conditions and solve for J in H(div)
|
||||
HYPRE_BigInt glb_size_rt = fes_cond_rt.GlobalTrueVSize();
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "\nSolving for current density in H(Div) "
|
||||
<< "using diagonally scaled CG" << endl;
|
||||
cout << "Size of linear system: "
|
||||
<< glb_size_rt << endl;
|
||||
}
|
||||
Array<int> ess_bdr_tdof_rt;
|
||||
OperatorPtr M;
|
||||
Vector B, X;
|
||||
|
||||
fes_cond_rt.GetEssentialTrueDofs(ess_bdr_j, ess_bdr_tdof_rt);
|
||||
|
||||
j_cond = 0.0;
|
||||
m_rt.FormLinearSystem(ess_bdr_tdof_rt, j_cond, b_rt, M, X, B);
|
||||
|
||||
HypreDiagScale prec;
|
||||
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetPrintLevel(1);
|
||||
cg.SetPreconditioner(prec);
|
||||
cg.SetOperator(*M);
|
||||
cg.Mult(B, X);
|
||||
m_rt.RecoverFEMSolution(X, b_rt, j_cond);
|
||||
}
|
||||
@@ -1,818 +0,0 @@
|
||||
// MFEM Example 35 - Parallel Version
|
||||
//
|
||||
// Compile with: make ex35p
|
||||
//
|
||||
// Sample runs: mpirun -np 4 ex35p -p 0 -o 2
|
||||
// mpirun -np 4 ex35p -p 0 -o 2 -pbc '22 23 24' -em 0
|
||||
// mpirun -np 4 ex35p -p 1 -o 1 -rp 2
|
||||
// mpirun -np 4 ex35p -p 1 -o 2
|
||||
// mpirun -np 4 ex35p -p 2 -o 1 -rp 2 -c 15
|
||||
//
|
||||
// Device sample runs:
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define and
|
||||
// solve simple complex-valued linear systems. It implements three
|
||||
// variants of a damped harmonic oscillator:
|
||||
//
|
||||
// 1) A scalar H1 field
|
||||
// -Div(a Grad u) - omega^2 b u + i omega c u = 0
|
||||
//
|
||||
// 2) A vector H(Curl) field
|
||||
// Curl(a Curl u) - omega^2 b u + i omega c u = 0
|
||||
//
|
||||
// 3) A vector H(Div) field
|
||||
// -Grad(a Div u) - omega^2 b u + i omega c u = 0
|
||||
//
|
||||
// In each case the field is driven by a forced oscillation, with
|
||||
// angular frequency omega, imposed at the boundary or a portion
|
||||
// of the boundary. The spatial variation of the boundary
|
||||
// condition is computed as an eigenmode of an appropriate
|
||||
// operator defined on a portion of the boundary i.e. a port
|
||||
// boundary condition.
|
||||
//
|
||||
// In electromagnetics the coefficients are typically named the
|
||||
// permeability, mu = 1/a, permittivity, epsilon = b, and
|
||||
// conductivity, sigma = c. The user can specify these constants
|
||||
// using either set of names.
|
||||
//
|
||||
// This example demonstrates how to transfer fields computed on
|
||||
// a boundary generated SubMesh to the full mesh and apply them
|
||||
// as boundary conditions. The default mesh and corresponding
|
||||
// boundary attriburtes were chosen to verify proper behavior on
|
||||
// both triangular and quadrilateral faces of tetrahedral,
|
||||
// wedge-shaped, and hexahedral elements.
|
||||
//
|
||||
// The example also demonstrates how to display a time-varying
|
||||
// solution as a sequence of fields sent to a single GLVis socket.
|
||||
//
|
||||
// We recommend viewing examples 11, 13, and 22 before viewing
|
||||
// this example.
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
static double mu_ = 1.0;
|
||||
static double epsilon_ = 1.0;
|
||||
static double sigma_ = 2.0;
|
||||
|
||||
void SetPortBC(int prob, int dim, int mode, ParGridFunction &port_bc);
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
// 1. Initialize MPI and HYPRE.
|
||||
Mpi::Init(argc, argv);
|
||||
int num_procs = Mpi::WorldSize();
|
||||
int myid = Mpi::WorldRank();
|
||||
Hypre::Init();
|
||||
|
||||
// 2. Parse command-line options.
|
||||
const char *mesh_file = "../data/fichera-mixed.mesh";
|
||||
int ser_ref_levels = 1;
|
||||
int par_ref_levels = 1;
|
||||
int order = 1;
|
||||
Array<int> port_bc_attr;
|
||||
int prob = 0;
|
||||
int mode = 1;
|
||||
double freq = -1.0;
|
||||
double omega = 2.0 * M_PI;
|
||||
double a_coef = 0.0;
|
||||
bool herm_conv = true;
|
||||
bool slu_solver = false;
|
||||
bool visualization = 1;
|
||||
bool mixed = true;
|
||||
bool pa = false;
|
||||
const char *device_config = "cpu";
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
"Mesh file to use.");
|
||||
args.AddOption(&ser_ref_levels, "-rs", "--refine-serial",
|
||||
"Number of times to refine the mesh uniformly in serial.");
|
||||
args.AddOption(&par_ref_levels, "-rp", "--refine-parallel",
|
||||
"Number of times to refine the mesh uniformly in parallel.");
|
||||
args.AddOption(&order, "-o", "--order",
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&prob, "-p", "--problem-type",
|
||||
"Choose between 0: H_1, 1: H(Curl), or 2: H(Div) "
|
||||
"damped harmonic oscillator.");
|
||||
args.AddOption(&mode, "-em", "--eigenmode",
|
||||
"Choose the index of the port eigenmode.");
|
||||
args.AddOption(&a_coef, "-a", "--stiffness-coef",
|
||||
"Stiffness coefficient (spring constant or 1/mu).");
|
||||
args.AddOption(&epsilon_, "-b", "--mass-coef",
|
||||
"Mass coefficient (or epsilon).");
|
||||
args.AddOption(&sigma_, "-c", "--damping-coef",
|
||||
"Damping coefficient (or sigma).");
|
||||
args.AddOption(&mu_, "-mu", "--permeability",
|
||||
"Permeability of free space (or 1/(spring constant)).");
|
||||
args.AddOption(&epsilon_, "-eps", "--permittivity",
|
||||
"Permittivity of free space (or mass constant).");
|
||||
args.AddOption(&sigma_, "-sigma", "--conductivity",
|
||||
"Conductivity (or damping constant).");
|
||||
args.AddOption(&freq, "-f", "--frequency",
|
||||
"Frequency (in Hz).");
|
||||
args.AddOption(&port_bc_attr, "-pbc", "--port-bc-attr",
|
||||
"Attributes of port boundary condition");
|
||||
args.AddOption(&herm_conv, "-herm", "--hermitian", "-no-herm",
|
||||
"--no-hermitian", "Use convention for Hermitian operators.");
|
||||
#ifdef MFEM_USE_SUPERLU
|
||||
args.AddOption(&slu_solver, "-slu", "--superlu", "-no-slu",
|
||||
"--no-superlu", "Use the SuperLU Solver.");
|
||||
#endif
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.AddOption(&mixed, "-mixed", "--mixed-mesh", "-hex",
|
||||
"--hex-mesh", "Mixed mesh of hexahedral mesh.");
|
||||
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
|
||||
"--no-partial-assembly", "Enable Partial Assembly.");
|
||||
args.AddOption(&device_config, "-d", "--device",
|
||||
"Device configuration string, see Device::Configure().");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
if (!mixed || pa)
|
||||
{
|
||||
mesh_file = "../data/fichera.mesh";
|
||||
}
|
||||
|
||||
if ( a_coef != 0.0 )
|
||||
{
|
||||
mu_ = 1.0 / a_coef;
|
||||
}
|
||||
if ( freq > 0.0 )
|
||||
{
|
||||
omega = 2.0 * M_PI * freq;
|
||||
}
|
||||
if (port_bc_attr.Size() == 0 &&
|
||||
(strcmp(mesh_file, "../data/fichera-mixed.mesh") == 0 ||
|
||||
strcmp(mesh_file, "../data/fichera.mesh") == 0))
|
||||
{
|
||||
port_bc_attr.SetSize(4);
|
||||
port_bc_attr[0] = 7;
|
||||
port_bc_attr[1] = 8;
|
||||
port_bc_attr[2] = 11;
|
||||
port_bc_attr[3] = 12;
|
||||
}
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
MFEM_VERIFY(prob >= 0 && prob <=2,
|
||||
"Unrecognized problem type: " << prob);
|
||||
|
||||
ComplexOperator::Convention conv =
|
||||
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
if (myid == 0) { device.Print(); }
|
||||
|
||||
// 4. Read the (serial) mesh from the given mesh file on all processors. We
|
||||
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
|
||||
// and volume meshes with the same code.
|
||||
Mesh *mesh = new Mesh(mesh_file, 1, 1);
|
||||
int dim = mesh->Dimension();
|
||||
|
||||
// 5. Refine the serial mesh on all processors to increase the resolution.
|
||||
for (int l = 0; l < ser_ref_levels; l++)
|
||||
{
|
||||
mesh->UniformRefinement();
|
||||
}
|
||||
|
||||
// 6a. Define a parallel mesh by a partitioning of the serial mesh. Refine
|
||||
// this mesh further in parallel to increase the resolution. Once the
|
||||
// parallel mesh is defined, the serial mesh can be deleted.
|
||||
ParMesh pmesh(MPI_COMM_WORLD, *mesh);
|
||||
delete mesh;
|
||||
for (int l = 0; l < par_ref_levels; l++)
|
||||
{
|
||||
pmesh.UniformRefinement();
|
||||
}
|
||||
|
||||
// 6b. Extract a submesh covering a portion of the boundary
|
||||
ParSubMesh pmesh_port(ParSubMesh::CreateFromBoundary(pmesh, port_bc_attr));
|
||||
|
||||
// 7a. Define a parallel finite element space on the parallel mesh. Here we
|
||||
// use continuous Lagrange, Nedelec, or Raviart-Thomas finite elements
|
||||
// of the specified order.
|
||||
if (dim == 1 && prob != 0 )
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Switching to problem type 0, H1 basis functions, "
|
||||
<< "for 1 dimensional mesh." << endl;
|
||||
}
|
||||
prob = 0;
|
||||
}
|
||||
|
||||
FiniteElementCollection *fec = NULL;
|
||||
switch (prob)
|
||||
{
|
||||
case 0: fec = new H1_FECollection(order, dim); break;
|
||||
case 1: fec = new ND_FECollection(order, dim); break;
|
||||
case 2: fec = new RT_FECollection(order - 1, dim); break;
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
ParFiniteElementSpace fespace(&pmesh, fec);
|
||||
HYPRE_BigInt size = fespace.GlobalTrueVSize();
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Number of finite element unknowns: " << size << endl;
|
||||
}
|
||||
|
||||
// 7b. Define a parallel finite element space on the sub-mesh. Here we
|
||||
// use continuous Lagrange, Nedelec, or L2 finite elements of
|
||||
// the specified order.
|
||||
FiniteElementCollection *fec_port = NULL;
|
||||
switch (prob)
|
||||
{
|
||||
case 0: fec_port = new H1_FECollection(order, dim-1); break;
|
||||
case 1:
|
||||
if (dim == 3)
|
||||
{
|
||||
fec_port = new ND_FECollection(order, dim-1);
|
||||
}
|
||||
else
|
||||
{
|
||||
fec_port = new L2_FECollection(order - 1, dim-1,
|
||||
BasisType::GaussLegendre,
|
||||
FiniteElement::INTEGRAL);
|
||||
}
|
||||
break;
|
||||
case 2: fec_port = new L2_FECollection(order - 1, dim-1,
|
||||
BasisType::GaussLegendre,
|
||||
FiniteElement::INTEGRAL); break;
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
ParFiniteElementSpace fespace_port(&pmesh_port, fec_port);
|
||||
HYPRE_BigInt size_port = fespace_port.GlobalTrueVSize();
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Number of finite element port BC unknowns: " << size_port
|
||||
<< endl;
|
||||
}
|
||||
|
||||
// 8a. Define a parallel grid function on the SubMesh which will contain
|
||||
// the field to be applied as a port boundary condition.
|
||||
ParGridFunction port_bc(&fespace_port);
|
||||
port_bc = 0.0;
|
||||
|
||||
SetPortBC(prob, dim, mode, port_bc);
|
||||
|
||||
// 8b. Save the SubMesh and associated port boundary condition in parallel.
|
||||
// This output can be viewed later using GLVis:
|
||||
// "glvis -np <np> -m port_mesh -g port_mode"
|
||||
{
|
||||
ostringstream mesh_name, port_name;
|
||||
mesh_name << "port_mesh." << setfill('0') << setw(6) << myid;
|
||||
port_name << "port_mode." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh_port.Print(mesh_ofs);
|
||||
|
||||
ofstream port_ofs(port_name.str().c_str());
|
||||
port_ofs.precision(8);
|
||||
port_bc.Save(port_ofs);
|
||||
}
|
||||
// 8c. Send the port bc, computed on the SubMesh, to a GLVis server.
|
||||
if (visualization && dim == 3)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream port_sock(vishost, visport);
|
||||
port_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
port_sock.precision(8);
|
||||
port_sock << "solution\n" << pmesh_port << port_bc
|
||||
<< "window_title 'Port BC'"
|
||||
<< "window_geometry 0 0 400 350" << flush;
|
||||
}
|
||||
|
||||
// 9. Determine the list of true (i.e. parallel conforming) essential
|
||||
// boundary dofs. In this example, the boundary conditions are defined
|
||||
// using an eigenmode of the appropriate type computed on the SubMesh.
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr;
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
{
|
||||
ess_bdr.SetSize(pmesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
// 10. Set up the parallel linear form b(.) which corresponds to the
|
||||
// right-hand side of the FEM linear system.
|
||||
ParComplexLinearForm b(&fespace, conv);
|
||||
b.Vector::operator=(0.0);
|
||||
|
||||
// 11a. Define the solution vector u as a parallel complex finite element
|
||||
// grid function corresponding to fespace. Initialize u to equal zero.
|
||||
ParComplexGridFunction u(&fespace);
|
||||
u = 0.0;
|
||||
pmesh_port.Transfer(port_bc, u.real());
|
||||
|
||||
// 11b. Send the transferred port bc field to a GLVis server.
|
||||
{
|
||||
ParGridFunction full_bc(&fespace);
|
||||
ParTransferMap port_to_full(port_bc, full_bc);
|
||||
|
||||
full_bc = 0.0;
|
||||
port_to_full.Transfer(port_bc, full_bc);
|
||||
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream full_sock(vishost, visport);
|
||||
full_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
full_sock.precision(8);
|
||||
full_sock << "solution\n" << pmesh << full_bc
|
||||
<< "window_title 'Transferred BC'"
|
||||
<< "window_geometry 400 0 400 350"<< flush;
|
||||
}
|
||||
}
|
||||
|
||||
// 12. Set up the parallel sesquilinear form a(.,.) on the finite element
|
||||
// space corresponding to the damped harmonic oscillator operator of the
|
||||
// appropriate type:
|
||||
//
|
||||
// 0) A scalar H1 field
|
||||
// -Div(a Grad) - omega^2 b + i omega c
|
||||
//
|
||||
// 1) A vector H(Curl) field
|
||||
// Curl(a Curl) - omega^2 b + i omega c
|
||||
//
|
||||
// 2) A vector H(Div) field
|
||||
// -Grad(a Div) - omega^2 b + i omega c
|
||||
//
|
||||
ConstantCoefficient stiffnessCoef(1.0/mu_);
|
||||
ConstantCoefficient massCoef(-omega * omega * epsilon_);
|
||||
ConstantCoefficient lossCoef(omega * sigma_);
|
||||
ConstantCoefficient negMassCoef(omega * omega * epsilon_);
|
||||
|
||||
ParSesquilinearForm a(&fespace, conv);
|
||||
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef),
|
||||
NULL);
|
||||
a.AddDomainIntegrator(new MassIntegrator(massCoef),
|
||||
new MassIntegrator(lossCoef));
|
||||
break;
|
||||
case 1:
|
||||
a.AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef),
|
||||
NULL);
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
|
||||
new VectorFEMassIntegrator(lossCoef));
|
||||
break;
|
||||
case 2:
|
||||
a.AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef),
|
||||
NULL);
|
||||
a.AddDomainIntegrator(new VectorFEMassIntegrator(massCoef),
|
||||
new VectorFEMassIntegrator(lossCoef));
|
||||
break;
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
|
||||
// 13. Assemble the parallel bilinear form and the corresponding linear
|
||||
// system, applying any necessary transformations such as: parallel
|
||||
// assembly, eliminating boundary conditions, applying conforming
|
||||
// constraints for non-conforming AMR, etc.
|
||||
a.Assemble();
|
||||
|
||||
OperatorHandle A;
|
||||
Vector B, U;
|
||||
|
||||
a.FormLinearSystem(ess_tdof_list, u, b, A, U, B);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Size of linear system: "
|
||||
<< 2 * size << endl << endl;
|
||||
}
|
||||
|
||||
if (!slu_solver)
|
||||
{
|
||||
// 14a. Set up the parallel bilinear form for the preconditioner
|
||||
// corresponding to the appropriate operator
|
||||
//
|
||||
// 0) A scalar H1 field
|
||||
// -Div(a Grad) - omega^2 b + i omega c
|
||||
//
|
||||
// 1) A vector H(Curl) field
|
||||
// Curl(a Curl) + omega^2 b + i omega c
|
||||
//
|
||||
// 2) A vector H(Div) field
|
||||
// -Grad(a Div) - omega^2 b + i omega c
|
||||
//
|
||||
ParBilinearForm pcOp(&fespace);
|
||||
if (pa) { pcOp.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
pcOp.AddDomainIntegrator(new DiffusionIntegrator(stiffnessCoef));
|
||||
pcOp.AddDomainIntegrator(new MassIntegrator(massCoef));
|
||||
pcOp.AddDomainIntegrator(new MassIntegrator(lossCoef));
|
||||
break;
|
||||
case 1:
|
||||
pcOp.AddDomainIntegrator(new CurlCurlIntegrator(stiffnessCoef));
|
||||
pcOp.AddDomainIntegrator(new VectorFEMassIntegrator(negMassCoef));
|
||||
pcOp.AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
|
||||
break;
|
||||
case 2:
|
||||
pcOp.AddDomainIntegrator(new DivDivIntegrator(stiffnessCoef));
|
||||
pcOp.AddDomainIntegrator(new VectorFEMassIntegrator(massCoef));
|
||||
pcOp.AddDomainIntegrator(new VectorFEMassIntegrator(lossCoef));
|
||||
break;
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
pcOp.Assemble();
|
||||
|
||||
// 14b. Define and apply a parallel FGMRES solver for AU=B with a block
|
||||
// diagonal preconditioner based on the appropriate multigrid
|
||||
// preconditioner from hypre.
|
||||
Array<int> blockTrueOffsets;
|
||||
blockTrueOffsets.SetSize(3);
|
||||
blockTrueOffsets[0] = 0;
|
||||
blockTrueOffsets[1] = A->Height() / 2;
|
||||
blockTrueOffsets[2] = A->Height() / 2;
|
||||
blockTrueOffsets.PartialSum();
|
||||
|
||||
BlockDiagonalPreconditioner BDP(blockTrueOffsets);
|
||||
|
||||
Operator * pc_r = NULL;
|
||||
Operator * pc_i = NULL;
|
||||
|
||||
if (pa)
|
||||
{
|
||||
pc_r = new OperatorJacobiSmoother(pcOp, ess_tdof_list);
|
||||
}
|
||||
else
|
||||
{
|
||||
OperatorHandle PCOp;
|
||||
pcOp.FormSystemMatrix(ess_tdof_list, PCOp);
|
||||
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
pc_r = new HypreBoomerAMG(*PCOp.As<HypreParMatrix>());
|
||||
break;
|
||||
case 1:
|
||||
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), &fespace);
|
||||
break;
|
||||
case 2:
|
||||
if (dim == 2 )
|
||||
{
|
||||
pc_r = new HypreAMS(*PCOp.As<HypreParMatrix>(), &fespace);
|
||||
}
|
||||
else
|
||||
{
|
||||
pc_r = new HypreADS(*PCOp.As<HypreParMatrix>(), &fespace);
|
||||
}
|
||||
break;
|
||||
default: break; // This should be unreachable
|
||||
}
|
||||
}
|
||||
pc_i = new ScaledOperator(pc_r,
|
||||
(conv == ComplexOperator::HERMITIAN) ?
|
||||
-1.0:1.0);
|
||||
|
||||
BDP.SetDiagonalBlock(0, pc_r);
|
||||
BDP.SetDiagonalBlock(1, pc_i);
|
||||
BDP.owns_blocks = 1;
|
||||
|
||||
FGMRESSolver fgmres(MPI_COMM_WORLD);
|
||||
fgmres.SetPreconditioner(BDP);
|
||||
fgmres.SetOperator(*A.Ptr());
|
||||
fgmres.SetRelTol(1e-6);
|
||||
fgmres.SetMaxIter(1000);
|
||||
fgmres.SetPrintLevel(1);
|
||||
fgmres.Mult(B, U);
|
||||
}
|
||||
#ifdef MFEM_USE_SUPERLU
|
||||
else
|
||||
{
|
||||
// 14. Solve using a direct solver
|
||||
// Transform to monolithic HypreParMatrix
|
||||
HypreParMatrix *A_hyp = A.As<ComplexHypreParMatrix>()->GetSystemMatrix();
|
||||
SuperLURowLocMatrix SA(*A_hyp);
|
||||
SuperLUSolver superlu(MPI_COMM_WORLD);
|
||||
superlu.SetPrintStatistics(true);
|
||||
superlu.SetSymmetricPattern(false);
|
||||
superlu.SetColumnPermutation(superlu::PARMETIS);
|
||||
superlu.SetOperator(SA);
|
||||
superlu.Mult(B, U);
|
||||
delete A_hyp;
|
||||
}
|
||||
#endif
|
||||
|
||||
// 15. Recover the parallel grid function corresponding to U. This is the
|
||||
// local finite element solution on each processor.
|
||||
a.RecoverFEMSolution(U, b, u);
|
||||
|
||||
// 16. Save the refined mesh and the solution in parallel. This output can be
|
||||
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol_r" or
|
||||
// "glvis -np <np> -m mesh -g sol_i".
|
||||
{
|
||||
ostringstream mesh_name, sol_r_name, sol_i_name;
|
||||
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
|
||||
sol_r_name << "sol_r." << setfill('0') << setw(6) << myid;
|
||||
sol_i_name << "sol_i." << setfill('0') << setw(6) << myid;
|
||||
|
||||
ofstream mesh_ofs(mesh_name.str().c_str());
|
||||
mesh_ofs.precision(8);
|
||||
pmesh.Print(mesh_ofs);
|
||||
|
||||
ofstream sol_r_ofs(sol_r_name.str().c_str());
|
||||
ofstream sol_i_ofs(sol_i_name.str().c_str());
|
||||
sol_r_ofs.precision(8);
|
||||
sol_i_ofs.precision(8);
|
||||
u.real().Save(sol_r_ofs);
|
||||
u.imag().Save(sol_i_ofs);
|
||||
}
|
||||
|
||||
// 17. Send the solution by socket to a GLVis server.
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock_r(vishost, visport);
|
||||
sol_sock_r << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_r.precision(8);
|
||||
sol_sock_r << "solution\n" << pmesh << u.real()
|
||||
<< "window_title 'Solution: Real Part'"
|
||||
<< "window_geometry 800 0 400 350" << flush;
|
||||
|
||||
MPI_Barrier(MPI_COMM_WORLD);
|
||||
|
||||
socketstream sol_sock_i(vishost, visport);
|
||||
sol_sock_i << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock_i.precision(8);
|
||||
sol_sock_i << "solution\n" << pmesh << u.imag()
|
||||
<< "window_title 'Solution: Imaginary Part'"
|
||||
<< "window_geometry 1200 0 400 350" << flush;
|
||||
}
|
||||
if (visualization)
|
||||
{
|
||||
ParGridFunction u_t(&fespace);
|
||||
u_t = u.real();
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << pmesh << u_t
|
||||
<< "window_title 'Harmonic Solution (t = 0.0 T)'"
|
||||
<< "window_geometry 0 432 600 450"
|
||||
<< "pause\n" << flush;
|
||||
if (myid == 0)
|
||||
cout << "GLVis visualization paused."
|
||||
<< " Press space (in the GLVis window) to resume it.\n";
|
||||
int num_frames = 32;
|
||||
int i = 0;
|
||||
while (sol_sock)
|
||||
{
|
||||
double t = (double)(i % num_frames) / num_frames;
|
||||
ostringstream oss;
|
||||
oss << "Harmonic Solution (t = " << t << " T)";
|
||||
|
||||
add(cos( 2.0 * M_PI * t), u.real(),
|
||||
sin(-2.0 * M_PI * t), u.imag(), u_t);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock << "solution\n" << pmesh << u_t
|
||||
<< "window_title '" << oss.str() << "'" << flush;
|
||||
i++;
|
||||
}
|
||||
}
|
||||
|
||||
// 18. Free the used memory.
|
||||
delete fec_port;
|
||||
delete fec;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/**
|
||||
Solves the eigenvalue problem -Div(Grad x) = lambda x with
|
||||
homogeneous Dirichlet boundary conditions on the boundary of the
|
||||
domain. Returns mode number "mode" (counting from zero) in the
|
||||
ParGridFunction "x".
|
||||
*/
|
||||
void ScalarWaveGuide(int mode, ParGridFunction &x)
|
||||
{
|
||||
int nev = std::max(mode + 2, 5);
|
||||
int seed = 75;
|
||||
|
||||
ParFiniteElementSpace &fespace = *x.ParFESpace();
|
||||
ParMesh &pmesh = *fespace.GetParMesh();
|
||||
|
||||
Array<int> ess_bdr;
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
{
|
||||
ess_bdr.SetSize(pmesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
}
|
||||
|
||||
ParBilinearForm a(&fespace);
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator);
|
||||
a.Assemble();
|
||||
a.EliminateEssentialBCDiag(ess_bdr, 1.0);
|
||||
a.Finalize();
|
||||
|
||||
ParBilinearForm m(&fespace);
|
||||
m.AddDomainIntegrator(new MassIntegrator);
|
||||
m.Assemble();
|
||||
// shift the eigenvalue corresponding to eliminated dofs to a large value
|
||||
m.EliminateEssentialBCDiag(ess_bdr, numeric_limits<double>::min());
|
||||
m.Finalize();
|
||||
|
||||
HypreParMatrix *A = a.ParallelAssemble();
|
||||
HypreParMatrix *M = m.ParallelAssemble();
|
||||
|
||||
HypreBoomerAMG amg(*A);
|
||||
amg.SetPrintLevel(0);
|
||||
|
||||
HypreLOBPCG lobpcg(MPI_COMM_WORLD);
|
||||
lobpcg.SetNumModes(nev);
|
||||
lobpcg.SetRandomSeed(seed);
|
||||
lobpcg.SetPreconditioner(amg);
|
||||
lobpcg.SetMaxIter(200);
|
||||
lobpcg.SetTol(1e-8);
|
||||
lobpcg.SetPrecondUsageMode(1);
|
||||
lobpcg.SetPrintLevel(1);
|
||||
lobpcg.SetMassMatrix(*M);
|
||||
lobpcg.SetOperator(*A);
|
||||
lobpcg.Solve();
|
||||
|
||||
x = lobpcg.GetEigenvector(mode);
|
||||
|
||||
delete A;
|
||||
delete M;
|
||||
}
|
||||
|
||||
/**
|
||||
Solves the eigenvalue problem -Curl(Curl x) = lambda x with
|
||||
homogeneous Dirichlet boundary conditions, on the tangential
|
||||
component of x, on the boundary of the domain. Returns mode number
|
||||
"mode" (counting from zero) in the ParGridFunction "x".
|
||||
*/
|
||||
void VectorWaveGuide(int mode, ParGridFunction &x)
|
||||
{
|
||||
int nev = std::max(mode + 2, 5);
|
||||
|
||||
ParFiniteElementSpace &fespace = *x.ParFESpace();
|
||||
ParMesh &pmesh = *fespace.GetParMesh();
|
||||
|
||||
Array<int> ess_bdr;
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
{
|
||||
ess_bdr.SetSize(pmesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
}
|
||||
|
||||
ParBilinearForm a(&fespace);
|
||||
a.AddDomainIntegrator(new CurlCurlIntegrator);
|
||||
a.Assemble();
|
||||
a.EliminateEssentialBCDiag(ess_bdr, 1.0);
|
||||
a.Finalize();
|
||||
|
||||
ParBilinearForm m(&fespace);
|
||||
m.AddDomainIntegrator(new VectorFEMassIntegrator);
|
||||
m.Assemble();
|
||||
// shift the eigenvalue corresponding to eliminated dofs to a large value
|
||||
m.EliminateEssentialBCDiag(ess_bdr, numeric_limits<double>::min());
|
||||
m.Finalize();
|
||||
|
||||
HypreParMatrix *A = a.ParallelAssemble();
|
||||
HypreParMatrix *M = m.ParallelAssemble();
|
||||
|
||||
HypreAMS ams(*A,&fespace);
|
||||
ams.SetPrintLevel(0);
|
||||
ams.SetSingularProblem();
|
||||
|
||||
HypreAME ame(MPI_COMM_WORLD);
|
||||
ame.SetNumModes(nev);
|
||||
ame.SetPreconditioner(ams);
|
||||
ame.SetMaxIter(100);
|
||||
ame.SetTol(1e-8);
|
||||
ame.SetPrintLevel(1);
|
||||
ame.SetMassMatrix(*M);
|
||||
ame.SetOperator(*A);
|
||||
ame.Solve();
|
||||
|
||||
x = ame.GetEigenvector(mode);
|
||||
|
||||
delete A;
|
||||
delete M;
|
||||
}
|
||||
|
||||
/**
|
||||
Solves the eigenvalue problem -Div(Grad x) = lambda x with
|
||||
homogeneous Neumann boundary conditions on the boundary of the
|
||||
domain. Returns mode number "mode" (counting from zero) in the
|
||||
ParGridFunction "x_l2". Note that mode 0 is a constant field so
|
||||
higher mode numbers are often more interesting. The eigenmode is
|
||||
solved using continuous H1 basis of the appropriate order and then
|
||||
projected onto the L2 basis and returned.
|
||||
*/
|
||||
void PseudoScalarWaveGuide(int mode, ParGridFunction &x_l2)
|
||||
{
|
||||
int nev = std::max(mode + 2, 5);
|
||||
int seed = 75;
|
||||
|
||||
ParFiniteElementSpace &fespace_l2 = *x_l2.ParFESpace();
|
||||
ParMesh &pmesh = *fespace_l2.GetParMesh();
|
||||
int order_l2 = fespace_l2.FEColl()->GetOrder();
|
||||
|
||||
H1_FECollection fec(order_l2+1, pmesh.Dimension());
|
||||
ParFiniteElementSpace fespace(&pmesh, &fec);
|
||||
ParGridFunction x(&fespace);
|
||||
x = 0.0;
|
||||
|
||||
GridFunctionCoefficient xCoef(&x);
|
||||
|
||||
if (mode == 0)
|
||||
{
|
||||
x = 1.0;
|
||||
x_l2.ProjectCoefficient(xCoef);
|
||||
return;
|
||||
}
|
||||
|
||||
ParBilinearForm a(&fespace);
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator);
|
||||
a.AddDomainIntegrator(new MassIntegrator); // Shift eigenvalues by 1
|
||||
a.Assemble();
|
||||
a.Finalize();
|
||||
|
||||
ParBilinearForm m(&fespace);
|
||||
m.AddDomainIntegrator(new MassIntegrator);
|
||||
m.Assemble();
|
||||
m.Finalize();
|
||||
|
||||
HypreParMatrix *A = a.ParallelAssemble();
|
||||
HypreParMatrix *M = m.ParallelAssemble();
|
||||
|
||||
HypreBoomerAMG amg(*A);
|
||||
amg.SetPrintLevel(0);
|
||||
|
||||
HypreLOBPCG lobpcg(MPI_COMM_WORLD);
|
||||
lobpcg.SetNumModes(nev);
|
||||
lobpcg.SetRandomSeed(seed);
|
||||
lobpcg.SetPreconditioner(amg);
|
||||
lobpcg.SetMaxIter(200);
|
||||
lobpcg.SetTol(1e-8);
|
||||
lobpcg.SetPrecondUsageMode(1);
|
||||
lobpcg.SetPrintLevel(1);
|
||||
lobpcg.SetMassMatrix(*M);
|
||||
lobpcg.SetOperator(*A);
|
||||
lobpcg.Solve();
|
||||
|
||||
x = lobpcg.GetEigenvector(mode);
|
||||
|
||||
x_l2.ProjectCoefficient(xCoef);
|
||||
|
||||
delete A;
|
||||
delete M;
|
||||
}
|
||||
|
||||
// Compute eigenmode "mode" of either a Dirichlet or Neumann Laplacian
|
||||
// or of a Dirichlet curl curl operator based on the problem type and
|
||||
// dimension of the domain.
|
||||
void SetPortBC(int prob, int dim, int mode, ParGridFunction &port_bc)
|
||||
{
|
||||
switch (prob)
|
||||
{
|
||||
case 0:
|
||||
ScalarWaveGuide(mode, port_bc);
|
||||
break;
|
||||
case 1:
|
||||
if (dim == 3)
|
||||
{
|
||||
VectorWaveGuide(mode, port_bc);
|
||||
}
|
||||
else
|
||||
{
|
||||
PseudoScalarWaveGuide(mode, port_bc);
|
||||
}
|
||||
break;
|
||||
case 2:
|
||||
PseudoScalarWaveGuide(mode, port_bc);
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -1,463 +0,0 @@
|
||||
// MFEM Example 36
|
||||
//
|
||||
//
|
||||
// Compile with: make ex36
|
||||
//
|
||||
// Sample runs: ex36 -o 2
|
||||
// ex36 -o 2 -r 4
|
||||
//
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to solve the
|
||||
// bound-constrained energy minimization problem
|
||||
//
|
||||
// minimize ||∇u||² subject to u ≥ ϕ in H¹₀.
|
||||
//
|
||||
// This is known as the obstacle problem, and it is a simple
|
||||
// mathematical model for contact mechanics.
|
||||
//
|
||||
// In this example, the obstacle ϕ is a half-sphere centered
|
||||
// at the origin of a circular domain Ω. After solving to a
|
||||
// specified tolerance, the numerical solution is compared to
|
||||
// a closed-form exact solution to assess accuracy.
|
||||
//
|
||||
// The problem is discretized and solved using the proximal
|
||||
// Galerkin finite element method, introduced by Keith and
|
||||
// Surowiec [1].
|
||||
//
|
||||
// This example highlights the ability of MFEM to deliver high-
|
||||
// order solutions to variation inequality problems and
|
||||
// showcases how to set up and solve nonlinear mixed methods.
|
||||
//
|
||||
//
|
||||
// [1] Keith, B. and Surowiec, T. (2023) Proximal Galerkin: A structure-
|
||||
// preserving finite element method for pointwise bound constraints.
|
||||
// arXiv:2307.12444 [math.NA]
|
||||
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
double spherical_obstacle(const Vector &pt);
|
||||
double exact_solution_obstacle(const Vector &pt);
|
||||
void exact_solution_gradient_obstacle(const Vector &pt, Vector &grad);
|
||||
|
||||
class LogarithmGridFunctionCoefficient : public Coefficient
|
||||
{
|
||||
protected:
|
||||
GridFunction *u; // grid function
|
||||
Coefficient *obstacle;
|
||||
double min_val;
|
||||
|
||||
public:
|
||||
LogarithmGridFunctionCoefficient(GridFunction &u_, Coefficient &obst_,
|
||||
double min_val_=-36)
|
||||
: u(&u_), obstacle(&obst_), min_val(min_val_) { }
|
||||
|
||||
virtual double Eval(ElementTransformation &T, const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
class ExponentialGridFunctionCoefficient : public Coefficient
|
||||
{
|
||||
protected:
|
||||
GridFunction *u; // grid function
|
||||
Coefficient *obstacle;
|
||||
double min_val;
|
||||
double max_val;
|
||||
|
||||
public:
|
||||
ExponentialGridFunctionCoefficient(GridFunction &u_, Coefficient &obst_,
|
||||
double min_val_=0.0, double max_val_=1e6)
|
||||
: u(&u_), obstacle(&obst_), min_val(min_val_), max_val(max_val_) { }
|
||||
|
||||
virtual double Eval(ElementTransformation &T, const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
// 1. Parse command-line options.
|
||||
int order = 1;
|
||||
int max_it = 10;
|
||||
int ref_levels = 3;
|
||||
double alpha = 1.0;
|
||||
double tol = 1e-5;
|
||||
bool visualization = true;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&order, "-o", "--order",
|
||||
"Finite element order (polynomial degree)");
|
||||
args.AddOption(&ref_levels, "-r", "--refs",
|
||||
"Number of h-refinements.");
|
||||
args.AddOption(&max_it, "-mi", "--max-it",
|
||||
"Maximum number of iterations");
|
||||
args.AddOption(&tol, "-tol", "--tol",
|
||||
"Stopping criteria based on the difference between"
|
||||
"successive solution updates");
|
||||
args.AddOption(&alpha, "-step", "--step",
|
||||
"Step size alpha");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
return 1;
|
||||
}
|
||||
args.PrintOptions(cout);
|
||||
|
||||
// 2. Read the mesh from the mesh file.
|
||||
const char *mesh_file = "../data/disc-nurbs.mesh";
|
||||
Mesh mesh(mesh_file, 1, 1);
|
||||
int dim = mesh.Dimension();
|
||||
|
||||
// 3. Postprocess the mesh.
|
||||
// 3A. Refine the mesh to increase the resolution.
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh.UniformRefinement();
|
||||
}
|
||||
|
||||
// 3B. Interpolate the geometry after refinement to control geometry error.
|
||||
// NOTE: Minimum second-order interpolation is used to improve the accuracy.
|
||||
int curvature_order = max(order,2);
|
||||
mesh.SetCurvature(curvature_order);
|
||||
|
||||
// 3C. Rescale the domain to a unit circle (radius = 1).
|
||||
GridFunction *nodes = mesh.GetNodes();
|
||||
double scale = 2*sqrt(2);
|
||||
*nodes /= scale;
|
||||
|
||||
// 4. Define the necessary finite element spaces on the mesh.
|
||||
H1_FECollection H1fec(order+1, dim);
|
||||
FiniteElementSpace H1fes(&mesh, &H1fec);
|
||||
|
||||
L2_FECollection L2fec(order-1, dim);
|
||||
FiniteElementSpace L2fes(&mesh, &L2fec);
|
||||
|
||||
cout << "Number of H1 finite element unknowns: "
|
||||
<< H1fes.GetTrueVSize() << endl;
|
||||
cout << "Number of L2 finite element unknowns: "
|
||||
<< L2fes.GetTrueVSize() << endl;
|
||||
|
||||
Array<int> offsets(3);
|
||||
offsets[0] = 0;
|
||||
offsets[1] = H1fes.GetVSize();
|
||||
offsets[2] = L2fes.GetVSize();
|
||||
offsets.PartialSum();
|
||||
|
||||
BlockVector x(offsets), rhs(offsets);
|
||||
x = 0.0; rhs = 0.0;
|
||||
|
||||
// 5. Determine the list of true (i.e. conforming) essential boundary dofs.
|
||||
Array<int> ess_bdr;
|
||||
if (mesh.bdr_attributes.Size())
|
||||
{
|
||||
ess_bdr.SetSize(mesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
}
|
||||
|
||||
// 6. Define an initial guess for the solution.
|
||||
auto IC_func = [](const Vector &x)
|
||||
{
|
||||
double r0 = 1.0;
|
||||
double rr = 0.0;
|
||||
for (int i=0; i<x.Size(); i++)
|
||||
{
|
||||
rr += x(i)*x(i);
|
||||
}
|
||||
return r0*r0 - rr;
|
||||
};
|
||||
ConstantCoefficient one(1.0);
|
||||
ConstantCoefficient zero(0.0);
|
||||
|
||||
// 7. Define the solution vectors as a finite element grid functions
|
||||
// corresponding to the fespaces.
|
||||
GridFunction u_gf, delta_psi_gf;
|
||||
|
||||
u_gf.MakeRef(&H1fes,x,offsets[0]);
|
||||
delta_psi_gf.MakeRef(&L2fes,x,offsets[1]);
|
||||
delta_psi_gf = 0.0;
|
||||
|
||||
GridFunction u_old_gf(&H1fes);
|
||||
GridFunction psi_old_gf(&L2fes);
|
||||
GridFunction psi_gf(&L2fes);
|
||||
u_old_gf = 0.0;
|
||||
psi_old_gf = 0.0;
|
||||
|
||||
// 8. Define the function coefficients for the solution and use them to
|
||||
// initialize the initial guess
|
||||
FunctionCoefficient exact_coef(exact_solution_obstacle);
|
||||
VectorFunctionCoefficient exact_grad_coef(dim,exact_solution_gradient_obstacle);
|
||||
FunctionCoefficient IC_coef(IC_func);
|
||||
ConstantCoefficient f(0.0);
|
||||
FunctionCoefficient obstacle(spherical_obstacle);
|
||||
u_gf.ProjectCoefficient(IC_coef);
|
||||
u_old_gf = u_gf;
|
||||
|
||||
// 9. Initialize the slack variable ψₕ = exp(uₕ)
|
||||
LogarithmGridFunctionCoefficient ln_u(u_gf, obstacle);
|
||||
psi_gf.ProjectCoefficient(ln_u);
|
||||
psi_old_gf = psi_gf;
|
||||
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock;
|
||||
if (visualization)
|
||||
{
|
||||
sol_sock.open(vishost,visport);
|
||||
sol_sock.precision(8);
|
||||
}
|
||||
|
||||
// 10. Iterate
|
||||
int k;
|
||||
int total_iterations = 0;
|
||||
double increment_u = 0.1;
|
||||
for (k = 0; k < max_it; k++)
|
||||
{
|
||||
GridFunction u_tmp(&H1fes);
|
||||
u_tmp = u_old_gf;
|
||||
|
||||
mfem::out << "\nOUTER ITERATION " << k+1 << endl;
|
||||
|
||||
int j;
|
||||
for ( j = 0; j < 10; j++)
|
||||
{
|
||||
total_iterations++;
|
||||
|
||||
ConstantCoefficient alpha_cf(alpha);
|
||||
|
||||
LinearForm b0,b1;
|
||||
b0.Update(&H1fes,rhs.GetBlock(0),0);
|
||||
b1.Update(&L2fes,rhs.GetBlock(1),0);
|
||||
|
||||
ExponentialGridFunctionCoefficient exp_psi(psi_gf, zero);
|
||||
ProductCoefficient neg_exp_psi(-1.0,exp_psi);
|
||||
GradientGridFunctionCoefficient grad_u_old(&u_old_gf);
|
||||
ProductCoefficient alpha_f(alpha, f);
|
||||
GridFunctionCoefficient psi_cf(&psi_gf);
|
||||
GridFunctionCoefficient psi_old_cf(&psi_old_gf);
|
||||
SumCoefficient psi_old_minus_psi(psi_old_cf, psi_cf, 1.0, -1.0);
|
||||
|
||||
b0.AddDomainIntegrator(new DomainLFIntegrator(alpha_f));
|
||||
b0.AddDomainIntegrator(new DomainLFIntegrator(psi_old_minus_psi));
|
||||
b0.Assemble();
|
||||
|
||||
b1.AddDomainIntegrator(new DomainLFIntegrator(exp_psi));
|
||||
b1.AddDomainIntegrator(new DomainLFIntegrator(obstacle));
|
||||
b1.Assemble();
|
||||
|
||||
BilinearForm a00(&H1fes);
|
||||
a00.SetDiagonalPolicy(mfem::Operator::DIAG_ONE);
|
||||
a00.AddDomainIntegrator(new DiffusionIntegrator(alpha_cf));
|
||||
a00.Assemble();
|
||||
a00.EliminateEssentialBC(ess_bdr,x.GetBlock(0),rhs.GetBlock(0),
|
||||
mfem::Operator::DIAG_ONE);
|
||||
a00.Finalize();
|
||||
SparseMatrix &A00 = a00.SpMat();
|
||||
|
||||
MixedBilinearForm a10(&H1fes,&L2fes);
|
||||
a10.AddDomainIntegrator(new MixedScalarMassIntegrator());
|
||||
a10.Assemble();
|
||||
a10.EliminateTrialDofs(ess_bdr, x.GetBlock(0), rhs.GetBlock(1));
|
||||
a10.Finalize();
|
||||
SparseMatrix &A10 = a10.SpMat();
|
||||
|
||||
SparseMatrix *A01 = Transpose(A10);
|
||||
|
||||
BilinearForm a11(&L2fes);
|
||||
a11.AddDomainIntegrator(new MassIntegrator(neg_exp_psi));
|
||||
// NOTE: Shift the spectrum of the Hessian matrix for additional
|
||||
// stability (Quasi-Newton).
|
||||
ConstantCoefficient eps_cf(-1e-6);
|
||||
if (order == 1)
|
||||
{
|
||||
// NOTE: ∇ₕuₕ = 0 for constant functions.
|
||||
// Therefore, we use the mass matrix to shift the spectrum
|
||||
a11.AddDomainIntegrator(new MassIntegrator(eps_cf));
|
||||
}
|
||||
else
|
||||
{
|
||||
a11.AddDomainIntegrator(new DiffusionIntegrator(eps_cf));
|
||||
}
|
||||
a11.Assemble();
|
||||
a11.Finalize();
|
||||
SparseMatrix &A11 = a11.SpMat();
|
||||
|
||||
BlockOperator A(offsets);
|
||||
A.SetBlock(0,0,&A00);
|
||||
A.SetBlock(1,0,&A10);
|
||||
A.SetBlock(0,1,A01);
|
||||
A.SetBlock(1,1,&A11);
|
||||
|
||||
BlockDiagonalPreconditioner prec(offsets);
|
||||
prec.SetDiagonalBlock(0,new GSSmoother(A00));
|
||||
prec.SetDiagonalBlock(1,new GSSmoother(A11));
|
||||
prec.owns_blocks = 1;
|
||||
|
||||
GMRES(A,prec,rhs,x,0,10000,500,1e-12,0.0);
|
||||
|
||||
u_gf.MakeRef(&H1fes, x.GetBlock(0), 0);
|
||||
delta_psi_gf.MakeRef(&L2fes, x.GetBlock(1), 0);
|
||||
|
||||
u_tmp -= u_gf;
|
||||
double Newton_update_size = u_tmp.ComputeL2Error(zero);
|
||||
u_tmp = u_gf;
|
||||
|
||||
double gamma = 1.0;
|
||||
delta_psi_gf *= gamma;
|
||||
psi_gf += delta_psi_gf;
|
||||
|
||||
if (visualization)
|
||||
{
|
||||
sol_sock << "solution\n" << mesh << u_gf << "window_title 'Discrete solution'"
|
||||
<< flush;
|
||||
mfem::out << "Newton_update_size = " << Newton_update_size << endl;
|
||||
}
|
||||
|
||||
delete A01;
|
||||
|
||||
if (Newton_update_size < increment_u)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
u_tmp = u_gf;
|
||||
u_tmp -= u_old_gf;
|
||||
increment_u = u_tmp.ComputeL2Error(zero);
|
||||
|
||||
mfem::out << "Number of Newton iterations = " << j+1 << endl;
|
||||
mfem::out << "Increment (|| uₕ - uₕ_prvs||) = " << increment_u << endl;
|
||||
|
||||
u_old_gf = u_gf;
|
||||
psi_old_gf = psi_gf;
|
||||
|
||||
if (increment_u < tol || k == max_it-1)
|
||||
{
|
||||
break;
|
||||
}
|
||||
|
||||
double H1_error = u_gf.ComputeH1Error(&exact_coef,&exact_grad_coef);
|
||||
mfem::out << "H1-error (|| u - uₕᵏ||) = " << H1_error << endl;
|
||||
|
||||
}
|
||||
|
||||
mfem::out << "\n Outer iterations: " << k+1
|
||||
<< "\n Total iterations: " << total_iterations
|
||||
<< "\n Total dofs: " << H1fes.GetTrueVSize() + L2fes.GetTrueVSize()
|
||||
<< endl;
|
||||
|
||||
// 11. Exact solution.
|
||||
if (visualization)
|
||||
{
|
||||
socketstream err_sock(vishost, visport);
|
||||
err_sock.precision(8);
|
||||
|
||||
GridFunction error_gf(&H1fes);
|
||||
error_gf.ProjectCoefficient(exact_coef);
|
||||
error_gf -= u_gf;
|
||||
|
||||
err_sock << "solution\n" << mesh << error_gf << "window_title 'Error'" <<
|
||||
flush;
|
||||
}
|
||||
|
||||
{
|
||||
double L2_error = u_gf.ComputeL2Error(exact_coef);
|
||||
double H1_error = u_gf.ComputeH1Error(&exact_coef,&exact_grad_coef);
|
||||
|
||||
ExponentialGridFunctionCoefficient u_alt_cf(psi_gf,obstacle);
|
||||
GridFunction u_alt_gf(&L2fes);
|
||||
u_alt_gf.ProjectCoefficient(u_alt_cf);
|
||||
double L2_error_alt = u_alt_gf.ComputeL2Error(exact_coef);
|
||||
|
||||
mfem::out << "\n Final L2-error (|| u - uₕ||) = " << L2_error <<
|
||||
endl;
|
||||
mfem::out << " Final H1-error (|| u - uₕ||) = " << H1_error << endl;
|
||||
mfem::out << " Final L2-error (|| u - ϕ - exp(ψₕ)||) = " << L2_error_alt <<
|
||||
endl;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
double LogarithmGridFunctionCoefficient::Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
MFEM_ASSERT(u != NULL, "grid function is not set");
|
||||
|
||||
double val = u->GetValue(T, ip) - obstacle->Eval(T, ip);
|
||||
return max(min_val, log(val));
|
||||
}
|
||||
|
||||
double ExponentialGridFunctionCoefficient::Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
MFEM_ASSERT(u != NULL, "grid function is not set");
|
||||
|
||||
double val = u->GetValue(T, ip);
|
||||
return min(max_val, max(min_val, exp(val) + obstacle->Eval(T, ip)));
|
||||
}
|
||||
|
||||
double spherical_obstacle(const Vector &pt)
|
||||
{
|
||||
double x = pt(0), y = pt(1);
|
||||
double r = sqrt(x*x + y*y);
|
||||
double r0 = 0.5;
|
||||
double beta = 0.9;
|
||||
|
||||
double b = r0*beta;
|
||||
double tmp = sqrt(r0*r0 - b*b);
|
||||
double B = tmp + b*b/tmp;
|
||||
double C = -b/tmp;
|
||||
|
||||
if (r > b)
|
||||
{
|
||||
return B + r * C;
|
||||
}
|
||||
else
|
||||
{
|
||||
return sqrt(r0*r0 - r*r);
|
||||
}
|
||||
}
|
||||
|
||||
double exact_solution_obstacle(const Vector &pt)
|
||||
{
|
||||
double x = pt(0), y = pt(1);
|
||||
double r = sqrt(x*x + y*y);
|
||||
double r0 = 0.5;
|
||||
double a = 0.348982574111686;
|
||||
double A = -0.340129705945858;
|
||||
|
||||
if (r > a)
|
||||
{
|
||||
return A * log(r);
|
||||
}
|
||||
else
|
||||
{
|
||||
return sqrt(r0*r0-r*r);
|
||||
}
|
||||
}
|
||||
|
||||
void exact_solution_gradient_obstacle(const Vector &pt, Vector &grad)
|
||||
{
|
||||
double x = pt(0), y = pt(1);
|
||||
double r = sqrt(x*x + y*y);
|
||||
double r0 = 0.5;
|
||||
double a = 0.348982574111686;
|
||||
double A = -0.340129705945858;
|
||||
|
||||
if (r > a)
|
||||
{
|
||||
grad(0) = A * x / (r*r);
|
||||
grad(1) = A * y / (r*r);
|
||||
}
|
||||
else
|
||||
{
|
||||
grad(0) = - x / sqrt( r0*r0 - r*r );
|
||||
grad(1) = - y / sqrt( r0*r0 - r*r );
|
||||
}
|
||||
}
|
||||
@@ -1,528 +0,0 @@
|
||||
// MFEM Example 36 - Parallel Version
|
||||
//
|
||||
//
|
||||
// Compile with: make ex36p
|
||||
//
|
||||
// Sample runs: mpirun -np 4 ex36p -o 2
|
||||
// mpirun -np 4 ex36p -o 2 -r 4
|
||||
//
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to solve the
|
||||
// bound-constrained energy minimization problem
|
||||
//
|
||||
// minimize ||∇u||² subject to u ≥ ϕ in H¹₀.
|
||||
//
|
||||
// This is known as the obstacle problem, and it is a simple
|
||||
// mathematical model for contact mechanics.
|
||||
//
|
||||
// In this example, the obstacle ϕ is a half-sphere centered
|
||||
// at the origin of a circular domain Ω. After solving to a
|
||||
// specified tolerance, the numerical solution is compared to
|
||||
// a closed-form exact solution to assess accuracy.
|
||||
//
|
||||
// The problem is discretized and solved using the proximal
|
||||
// Galerkin finite element method, introduced by Keith and
|
||||
// Surowiec [1].
|
||||
//
|
||||
// This example highlights the ability of MFEM to deliver high-
|
||||
// order solutions to variation inequality problems and
|
||||
// showcases how to set up and solve nonlinear mixed methods.
|
||||
//
|
||||
//
|
||||
// [1] Keith, B. and Surowiec, T. (2023) Proximal Galerkin: A structure-
|
||||
// preserving finite element method for pointwise bound constraints.
|
||||
// arXiv:2307.12444 [math.NA]
|
||||
|
||||
|
||||
#include "mfem.hpp"
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
using namespace std;
|
||||
using namespace mfem;
|
||||
|
||||
double spherical_obstacle(const Vector &pt);
|
||||
double exact_solution_obstacle(const Vector &pt);
|
||||
void exact_solution_gradient_obstacle(const Vector &pt, Vector &grad);
|
||||
|
||||
class LogarithmGridFunctionCoefficient : public Coefficient
|
||||
{
|
||||
protected:
|
||||
GridFunction *u; // grid function
|
||||
Coefficient *obstacle;
|
||||
double min_val;
|
||||
|
||||
public:
|
||||
LogarithmGridFunctionCoefficient(GridFunction &u_, Coefficient &obst_,
|
||||
double min_val_=-36)
|
||||
: u(&u_), obstacle(&obst_), min_val(min_val_) { }
|
||||
|
||||
virtual double Eval(ElementTransformation &T, const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
class ExponentialGridFunctionCoefficient : public Coefficient
|
||||
{
|
||||
protected:
|
||||
GridFunction *u; // grid function
|
||||
Coefficient *obstacle;
|
||||
double min_val;
|
||||
double max_val;
|
||||
|
||||
public:
|
||||
ExponentialGridFunctionCoefficient(GridFunction &u_, Coefficient &obst_,
|
||||
double min_val_=0.0, double max_val_=1e6)
|
||||
: u(&u_), obstacle(&obst_), min_val(min_val_), max_val(max_val_) { }
|
||||
|
||||
virtual double Eval(ElementTransformation &T, const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
// 0. Initialize MPI and HYPRE.
|
||||
Mpi::Init();
|
||||
int num_procs = Mpi::WorldSize();
|
||||
int myid = Mpi::WorldRank();
|
||||
Hypre::Init();
|
||||
|
||||
// 1. Parse command-line options.
|
||||
int order = 1;
|
||||
int max_it = 10;
|
||||
int ref_levels = 3;
|
||||
double alpha = 1.0;
|
||||
double tol = 1e-5;
|
||||
bool visualization = true;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&order, "-o", "--order",
|
||||
"Finite element order (polynomial degree).");
|
||||
args.AddOption(&ref_levels, "-r", "--refs",
|
||||
"Number of h-refinements.");
|
||||
args.AddOption(&max_it, "-mi", "--max-it",
|
||||
"Maximum number of iterations");
|
||||
args.AddOption(&tol, "-tol", "--tol",
|
||||
"Stopping criteria based on the difference between"
|
||||
"successive solution updates");
|
||||
args.AddOption(&alpha, "-step", "--step",
|
||||
"Step size alpha");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
// 2. Read the mesh from the mesh file.
|
||||
const char *mesh_file = "../data/disc-nurbs.mesh";
|
||||
Mesh mesh(mesh_file, 1, 1);
|
||||
int dim = mesh.Dimension();
|
||||
|
||||
// 3. Postprocess the mesh.
|
||||
// 3A. Refine the mesh to increase the resolution.
|
||||
for (int l = 0; l < ref_levels; l++)
|
||||
{
|
||||
mesh.UniformRefinement();
|
||||
}
|
||||
|
||||
// 3B. Interpolate the geometry after refinement to control geometry error.
|
||||
// NOTE: Minimum second-order interpolation is used to improve the accuracy.
|
||||
int curvature_order = max(order,2);
|
||||
mesh.SetCurvature(curvature_order);
|
||||
|
||||
// 3C. Rescale the domain to a unit circle (radius = 1).
|
||||
GridFunction *nodes = mesh.GetNodes();
|
||||
double scale = 2*sqrt(2);
|
||||
*nodes /= scale;
|
||||
|
||||
ParMesh pmesh(MPI_COMM_WORLD, mesh);
|
||||
mesh.Clear();
|
||||
|
||||
// 4. Define the necessary finite element spaces on the mesh.
|
||||
H1_FECollection H1fec(order+1, dim);
|
||||
ParFiniteElementSpace H1fes(&pmesh, &H1fec);
|
||||
|
||||
L2_FECollection L2fec(order-1, dim);
|
||||
ParFiniteElementSpace L2fes(&pmesh, &L2fec);
|
||||
|
||||
int num_dofs_H1 = H1fes.GetTrueVSize();
|
||||
MPI_Allreduce(MPI_IN_PLACE, &num_dofs_H1, 1, MPI_INT, MPI_SUM, MPI_COMM_WORLD);
|
||||
int num_dofs_L2 = L2fes.GetTrueVSize();
|
||||
MPI_Allreduce(MPI_IN_PLACE, &num_dofs_L2, 1, MPI_INT, MPI_SUM, MPI_COMM_WORLD);
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Number of H1 finite element unknowns: "
|
||||
<< num_dofs_H1 << endl;
|
||||
cout << "Number of L2 finite element unknowns: "
|
||||
<< num_dofs_L2 << endl;
|
||||
}
|
||||
|
||||
Array<int> offsets(3);
|
||||
offsets[0] = 0;
|
||||
offsets[1] = H1fes.GetVSize();
|
||||
offsets[2] = L2fes.GetVSize();
|
||||
offsets.PartialSum();
|
||||
|
||||
Array<int> toffsets(3);
|
||||
toffsets[0] = 0;
|
||||
toffsets[1] = H1fes.GetTrueVSize();
|
||||
toffsets[2] = L2fes.GetTrueVSize();
|
||||
toffsets.PartialSum();
|
||||
|
||||
BlockVector x(offsets), rhs(offsets);
|
||||
x = 0.0; rhs = 0.0;
|
||||
|
||||
BlockVector tx(toffsets), trhs(toffsets);
|
||||
tx = 0.0; trhs = 0.0;
|
||||
|
||||
// 5. Determine the list of true (i.e. conforming) essential boundary dofs.
|
||||
Array<int> empty;
|
||||
Array<int> ess_tdof_list;
|
||||
if (pmesh.bdr_attributes.Size())
|
||||
{
|
||||
Array<int> ess_bdr(pmesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
H1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
}
|
||||
|
||||
// 6. Define an initial guess for the solution.
|
||||
auto IC_func = [](const Vector &x)
|
||||
{
|
||||
double r0 = 1.0;
|
||||
double rr = 0.0;
|
||||
for (int i=0; i<x.Size(); i++)
|
||||
{
|
||||
rr += x(i)*x(i);
|
||||
}
|
||||
return r0*r0 - rr;
|
||||
};
|
||||
ConstantCoefficient one(1.0);
|
||||
ConstantCoefficient zero(0.0);
|
||||
|
||||
// 7. Define the solution vectors as a finite element grid functions
|
||||
// corresponding to the fespaces.
|
||||
ParGridFunction u_gf, delta_psi_gf;
|
||||
u_gf.MakeRef(&H1fes,x,offsets[0]);
|
||||
delta_psi_gf.MakeRef(&L2fes,x,offsets[1]);
|
||||
delta_psi_gf = 0.0;
|
||||
|
||||
ParGridFunction u_old_gf(&H1fes);
|
||||
ParGridFunction psi_old_gf(&L2fes);
|
||||
ParGridFunction psi_gf(&L2fes);
|
||||
u_old_gf = 0.0;
|
||||
psi_old_gf = 0.0;
|
||||
|
||||
|
||||
// 8. Define the function coefficients for the solution and use them to
|
||||
// initialize the initial guess
|
||||
FunctionCoefficient exact_coef(exact_solution_obstacle);
|
||||
VectorFunctionCoefficient exact_grad_coef(dim,exact_solution_gradient_obstacle);
|
||||
FunctionCoefficient IC_coef(IC_func);
|
||||
ConstantCoefficient f(0.0);
|
||||
FunctionCoefficient obstacle(spherical_obstacle);
|
||||
u_gf.ProjectCoefficient(IC_coef);
|
||||
u_old_gf = u_gf;
|
||||
|
||||
// 9. Initialize the slack variable ψₕ = exp(uₕ)
|
||||
LogarithmGridFunctionCoefficient ln_u(u_gf, obstacle);
|
||||
psi_gf.ProjectCoefficient(ln_u);
|
||||
psi_old_gf = psi_gf;
|
||||
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock;
|
||||
if (visualization)
|
||||
{
|
||||
sol_sock.open(vishost,visport);
|
||||
sol_sock.precision(8);
|
||||
}
|
||||
|
||||
// 10. Iterate
|
||||
int k;
|
||||
int total_iterations = 0;
|
||||
double increment_u = 0.1;
|
||||
for (k = 0; k < max_it; k++)
|
||||
{
|
||||
ParGridFunction u_tmp(&H1fes);
|
||||
u_tmp = u_old_gf;
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
mfem::out << "\nOUTER ITERATION " << k+1 << endl;
|
||||
}
|
||||
|
||||
int j;
|
||||
for ( j = 0; j < 10; j++)
|
||||
{
|
||||
total_iterations++;
|
||||
|
||||
ConstantCoefficient alpha_cf(alpha);
|
||||
|
||||
ParLinearForm b0,b1;
|
||||
b0.Update(&H1fes,rhs.GetBlock(0),0);
|
||||
b1.Update(&L2fes,rhs.GetBlock(1),0);
|
||||
|
||||
ExponentialGridFunctionCoefficient exp_psi(psi_gf, zero);
|
||||
ProductCoefficient neg_exp_psi(-1.0,exp_psi);
|
||||
GradientGridFunctionCoefficient grad_u_old(&u_old_gf);
|
||||
ProductCoefficient alpha_f(alpha, f);
|
||||
GridFunctionCoefficient psi_cf(&psi_gf);
|
||||
GridFunctionCoefficient psi_old_cf(&psi_old_gf);
|
||||
SumCoefficient psi_old_minus_psi(psi_old_cf, psi_cf, 1.0, -1.0);
|
||||
|
||||
b0.AddDomainIntegrator(new DomainLFIntegrator(alpha_f));
|
||||
b0.AddDomainIntegrator(new DomainLFIntegrator(psi_old_minus_psi));
|
||||
b0.Assemble();
|
||||
|
||||
b1.AddDomainIntegrator(new DomainLFIntegrator(exp_psi));
|
||||
b1.AddDomainIntegrator(new DomainLFIntegrator(obstacle));
|
||||
b1.Assemble();
|
||||
|
||||
ParBilinearForm a00(&H1fes);
|
||||
a00.SetDiagonalPolicy(mfem::Operator::DIAG_ONE);
|
||||
a00.AddDomainIntegrator(new DiffusionIntegrator(alpha_cf));
|
||||
a00.Assemble();
|
||||
HypreParMatrix A00;
|
||||
a00.FormLinearSystem(ess_tdof_list, x.GetBlock(0), rhs.GetBlock(0),
|
||||
A00, tx.GetBlock(0), trhs.GetBlock(0));
|
||||
|
||||
|
||||
ParMixedBilinearForm a10(&H1fes,&L2fes);
|
||||
a10.AddDomainIntegrator(new MixedScalarMassIntegrator());
|
||||
a10.Assemble();
|
||||
HypreParMatrix A10;
|
||||
a10.FormRectangularLinearSystem(ess_tdof_list, empty, x.GetBlock(0),
|
||||
rhs.GetBlock(1),
|
||||
A10, tx.GetBlock(0), trhs.GetBlock(1));
|
||||
|
||||
HypreParMatrix *A01 = A10.Transpose();
|
||||
|
||||
ParBilinearForm a11(&L2fes);
|
||||
a11.AddDomainIntegrator(new MassIntegrator(neg_exp_psi));
|
||||
// NOTE: Shift the spectrum of the Hessian matrix for additional
|
||||
// stability (Quasi-Newton).
|
||||
ConstantCoefficient eps_cf(-1e-6);
|
||||
if (order == 1)
|
||||
{
|
||||
// NOTE: ∇ₕuₕ = 0 for constant functions.
|
||||
// Therefore, we use the mass matrix to shift the spectrum
|
||||
a11.AddDomainIntegrator(new MassIntegrator(eps_cf));
|
||||
}
|
||||
else
|
||||
{
|
||||
a11.AddDomainIntegrator(new DiffusionIntegrator(eps_cf));
|
||||
}
|
||||
a11.Assemble();
|
||||
a11.Finalize();
|
||||
HypreParMatrix A11;
|
||||
a11.FormSystemMatrix(empty, A11);
|
||||
|
||||
BlockOperator A(toffsets);
|
||||
A.SetBlock(0,0,&A00);
|
||||
A.SetBlock(1,0,&A10);
|
||||
A.SetBlock(0,1,A01);
|
||||
A.SetBlock(1,1,&A11);
|
||||
|
||||
BlockDiagonalPreconditioner prec(toffsets);
|
||||
HypreBoomerAMG P00(A00);
|
||||
P00.SetPrintLevel(0);
|
||||
HypreSmoother P11(A11);
|
||||
prec.SetDiagonalBlock(0,&P00);
|
||||
prec.SetDiagonalBlock(1,&P11);
|
||||
|
||||
GMRESSolver gmres(MPI_COMM_WORLD);
|
||||
gmres.SetPrintLevel(-1);
|
||||
gmres.SetRelTol(1e-8);
|
||||
gmres.SetMaxIter(20000);
|
||||
gmres.SetKDim(500);
|
||||
gmres.SetOperator(A);
|
||||
gmres.SetPreconditioner(prec);
|
||||
gmres.Mult(trhs,tx);
|
||||
|
||||
u_gf.SetFromTrueDofs(tx.GetBlock(0));
|
||||
delta_psi_gf.SetFromTrueDofs(tx.GetBlock(1));
|
||||
|
||||
u_tmp -= u_gf;
|
||||
double Newton_update_size = u_tmp.ComputeL2Error(zero);
|
||||
u_tmp = u_gf;
|
||||
|
||||
double gamma = 1.0;
|
||||
delta_psi_gf *= gamma;
|
||||
psi_gf += delta_psi_gf;
|
||||
|
||||
if (visualization)
|
||||
{
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock << "solution\n" << pmesh << u_gf << "window_title 'Discrete solution'"
|
||||
<< flush;
|
||||
}
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
mfem::out << "Newton_update_size = " << Newton_update_size << endl;
|
||||
}
|
||||
|
||||
delete A01;
|
||||
|
||||
if (Newton_update_size < increment_u)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
u_tmp = u_gf;
|
||||
u_tmp -= u_old_gf;
|
||||
increment_u = u_tmp.ComputeL2Error(zero);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
mfem::out << "Number of Newton iterations = " << j+1 << endl;
|
||||
mfem::out << "Increment (|| uₕ - uₕ_prvs||) = " << increment_u << endl;
|
||||
}
|
||||
|
||||
u_old_gf = u_gf;
|
||||
psi_old_gf = psi_gf;
|
||||
|
||||
if (increment_u < tol || k == max_it-1)
|
||||
{
|
||||
break;
|
||||
}
|
||||
|
||||
double H1_error = u_gf.ComputeH1Error(&exact_coef,&exact_grad_coef);
|
||||
if (myid == 0)
|
||||
{
|
||||
mfem::out << "H1-error (|| u - uₕᵏ||) = " << H1_error << endl;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
mfem::out << "\n Outer iterations: " << k+1
|
||||
<< "\n Total iterations: " << total_iterations
|
||||
<< "\n Total dofs: " << num_dofs_H1 + num_dofs_L2
|
||||
<< endl;
|
||||
}
|
||||
|
||||
// 11. Exact solution.
|
||||
if (visualization)
|
||||
{
|
||||
socketstream err_sock(vishost, visport);
|
||||
err_sock.precision(8);
|
||||
|
||||
ParGridFunction error_gf(&H1fes);
|
||||
error_gf.ProjectCoefficient(exact_coef);
|
||||
error_gf -= u_gf;
|
||||
|
||||
err_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
err_sock << "solution\n" << pmesh << error_gf << "window_title 'Error'" <<
|
||||
flush;
|
||||
}
|
||||
|
||||
{
|
||||
double L2_error = u_gf.ComputeL2Error(exact_coef);
|
||||
double H1_error = u_gf.ComputeH1Error(&exact_coef,&exact_grad_coef);
|
||||
|
||||
ExponentialGridFunctionCoefficient u_alt_cf(psi_gf,obstacle);
|
||||
ParGridFunction u_alt_gf(&L2fes);
|
||||
u_alt_gf.ProjectCoefficient(u_alt_cf);
|
||||
double L2_error_alt = u_alt_gf.ComputeL2Error(exact_coef);
|
||||
|
||||
if (myid == 0)
|
||||
{
|
||||
mfem::out << "\n Final L2-error (|| u - uₕ||) = " << L2_error <<
|
||||
endl;
|
||||
mfem::out << " Final H1-error (|| u - uₕ||) = " << H1_error << endl;
|
||||
mfem::out << " Final L2-error (|| u - ϕ - exp(ψₕ)||) = " << L2_error_alt <<
|
||||
endl;
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
double LogarithmGridFunctionCoefficient::Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
MFEM_ASSERT(u != NULL, "grid function is not set");
|
||||
|
||||
double val = u->GetValue(T, ip) - obstacle->Eval(T, ip);
|
||||
return max(min_val, log(val));
|
||||
}
|
||||
|
||||
double ExponentialGridFunctionCoefficient::Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
MFEM_ASSERT(u != NULL, "grid function is not set");
|
||||
|
||||
double val = u->GetValue(T, ip);
|
||||
return min(max_val, max(min_val, exp(val) + obstacle->Eval(T, ip)));
|
||||
}
|
||||
|
||||
double spherical_obstacle(const Vector &pt)
|
||||
{
|
||||
double x = pt(0), y = pt(1);
|
||||
double r = sqrt(x*x + y*y);
|
||||
double r0 = 0.5;
|
||||
double beta = 0.9;
|
||||
|
||||
double b = r0*beta;
|
||||
double tmp = sqrt(r0*r0 - b*b);
|
||||
double B = tmp + b*b/tmp;
|
||||
double C = -b/tmp;
|
||||
|
||||
if (r > b)
|
||||
{
|
||||
return B + r * C;
|
||||
}
|
||||
else
|
||||
{
|
||||
return sqrt(r0*r0 - r*r);
|
||||
}
|
||||
}
|
||||
|
||||
double exact_solution_obstacle(const Vector &pt)
|
||||
{
|
||||
double x = pt(0), y = pt(1);
|
||||
double r = sqrt(x*x + y*y);
|
||||
double r0 = 0.5;
|
||||
double a = 0.348982574111686;
|
||||
double A = -0.340129705945858;
|
||||
|
||||
if (r > a)
|
||||
{
|
||||
return A * log(r);
|
||||
}
|
||||
else
|
||||
{
|
||||
return sqrt(r0*r0-r*r);
|
||||
}
|
||||
}
|
||||
|
||||
void exact_solution_gradient_obstacle(const Vector &pt, Vector &grad)
|
||||
{
|
||||
double x = pt(0), y = pt(1);
|
||||
double r = sqrt(x*x + y*y);
|
||||
double r0 = 0.5;
|
||||
double a = 0.348982574111686;
|
||||
double A = -0.340129705945858;
|
||||
|
||||
if (r > a)
|
||||
{
|
||||
grad(0) = A * x / (r*r);
|
||||
grad(1) = A * y / (r*r);
|
||||
}
|
||||
else
|
||||
{
|
||||
grad(0) = - x / sqrt( r0*r0 - r*r );
|
||||
grad(1) = - y / sqrt( r0*r0 - r*r );
|
||||
}
|
||||
}
|
||||
@@ -536,10 +536,8 @@ int main(int argc, char *argv[])
|
||||
if (!sout)
|
||||
{
|
||||
if (Mpi::Root())
|
||||
{
|
||||
cout << "Unable to connect to GLVis server at "
|
||||
<< vishost << ':' << visport << endl;
|
||||
}
|
||||
visualization = false;
|
||||
if (Mpi::Root())
|
||||
{
|
||||
@@ -554,10 +552,8 @@ int main(int argc, char *argv[])
|
||||
sout << "pause\n";
|
||||
sout << flush;
|
||||
if (Mpi::Root())
|
||||
{
|
||||
cout << "GLVis visualization paused."
|
||||
<< " Press space (in the GLVis window) to resume it.\n";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+4
-5
@@ -23,13 +23,13 @@ MFEM_LIB_FILE = mfem_is_not_built
|
||||
|
||||
SEQ_EXAMPLES = ex0 ex1 ex2 ex3 ex4 ex5 ex6 ex7 ex8 ex9 ex10 ex14 ex15 ex16 \
|
||||
ex17 ex18 ex19 ex20 ex21 ex22 ex23 ex24 ex25 ex26 ex27 ex28 ex29 ex30 \
|
||||
ex31 ex33 ex34 ex36
|
||||
ex31 ex33
|
||||
PAR_EXAMPLES = ex0p ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex8p ex9p ex10p ex11p \
|
||||
ex12p ex13p ex14p ex15p ex16p ex17p ex18p ex19p ex20p ex21p ex22p ex24p \
|
||||
ex25p ex26p ex27p ex28p ex29p ex30p ex31p ex32p ex33p ex34p ex35p ex36p
|
||||
SEQ_DEVICE_EXAMPLES = ex1 ex3 ex4 ex5 ex6 ex9 ex22 ex24 ex25 ex26 ex34
|
||||
ex25p ex26p ex27p ex28p ex29p ex30p ex31p ex32p ex33p
|
||||
SEQ_DEVICE_EXAMPLES = ex1 ex3 ex4 ex5 ex6 ex9 ex22 ex24 ex25 ex26
|
||||
PAR_DEVICE_EXAMPLES = ex1p ex2p ex3p ex4p ex5p ex6p ex7p ex9p ex13p ex22p \
|
||||
ex24p ex25p ex26p ex34p ex35p
|
||||
ex24p ex25p ex26p
|
||||
|
||||
ifeq ($(MFEM_USE_MPI),NO)
|
||||
EXAMPLES = $(SEQ_EXAMPLES)
|
||||
@@ -183,4 +183,3 @@ clean-exec:
|
||||
@rm -f ex23.mesh ex23-*.gf
|
||||
@rm -f ex25.mesh ex25-*.gf ex25p-*.*
|
||||
@rm -rf ex28_* ex28p_*
|
||||
@rm -rf cond.* cond_mesh.* cond_j.* dsol.* port_mesh.* port_mode.*
|
||||
|
||||
@@ -68,43 +68,11 @@ if (MFEM_ENABLE_TESTING)
|
||||
add_test(NAME ${TEST_NAME}_ser
|
||||
COMMAND ${TEST_NAME} ${THIS_TEST_OPTIONS})
|
||||
else()
|
||||
add_test(NAME ${TEST_NAME}_np=${MFEM_MPI_NP}
|
||||
add_test(NAME ${TEST_NAME}_np=4
|
||||
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
|
||||
${MPIEXEC_PREFLAGS}
|
||||
$<TARGET_FILE:${TEST_NAME}> ${THIS_TEST_OPTIONS}
|
||||
${MPIEXEC_POSTFLAGS})
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
# Add CUDA/HIP tests.
|
||||
set(DEVICE_EXAMPLES
|
||||
# serial examples with device support:
|
||||
ex9
|
||||
# parallel examples with device support:
|
||||
ex9p)
|
||||
set(MFEM_TEST_DEVICE)
|
||||
if (MFEM_USE_CUDA)
|
||||
set(MFEM_TEST_DEVICE "cuda")
|
||||
elseif (MFEM_USE_HIP)
|
||||
set(MFEM_TEST_DEVICE "hip")
|
||||
endif()
|
||||
if (MFEM_TEST_DEVICE)
|
||||
foreach(TEST_NAME ${DEVICE_EXAMPLES})
|
||||
string(TOUPPER ${TEST_NAME} UP_TEST_NAME)
|
||||
|
||||
set(THIS_TEST_OPTIONS "-no-vis" "-d" "${MFEM_TEST_DEVICE}")
|
||||
list(APPEND THIS_TEST_OPTIONS ${${UP_TEST_NAME}_TEST_OPTS})
|
||||
|
||||
if (NOT (${TEST_NAME} MATCHES ".*p$"))
|
||||
add_test(NAME ${PFX}${TEST_NAME}_${MFEM_TEST_DEVICE}_ser
|
||||
COMMAND ${PFX}${TEST_NAME} ${THIS_TEST_OPTIONS})
|
||||
else()
|
||||
add_test(NAME ${PFX}${TEST_NAME}_${MFEM_TEST_DEVICE}_np=${MFEM_MPI_NP}
|
||||
COMMAND ${MPIEXEC} ${MPIEXEC_NUMPROC_FLAG} ${MFEM_MPI_NP}
|
||||
${MPIEXEC_PREFLAGS}
|
||||
$<TARGET_FILE:${PFX}${TEST_NAME}> ${THIS_TEST_OPTIONS}
|
||||
${MPIEXEC_POSTFLAGS})
|
||||
endif()
|
||||
endforeach()
|
||||
endif(MFEM_TEST_DEVICE)
|
||||
endif(MFEM_ENABLE_TESTING)
|
||||
endif()
|
||||
|
||||
@@ -12,7 +12,8 @@ use of MFEM features based on the SUNDIALS suite of time integration and
|
||||
non-linear solvers.
|
||||
|
||||
To build these examples, make sure that MFEM is configured with the option
|
||||
"MFEM_USE_SUNDIALS = YES", see the top-level INSTALL file for details.
|
||||
"MFEM_USE_SUNDIALS = YES", see the top-level INSTALL file for details (version
|
||||
2.7 or higher of SUNDIALS is required).
|
||||
|
||||
We recommend comparing the original example codes with the corresponding files
|
||||
in the current directory.
|
||||
|
||||
@@ -280,16 +280,15 @@ int main(int argc, char *argv[])
|
||||
k.SetAssemblyLevel(AssemblyLevel::FULL);
|
||||
}
|
||||
m.AddDomainIntegrator(new MassIntegrator);
|
||||
constexpr double alpha = -1.0;
|
||||
k.AddDomainIntegrator(new ConvectionIntegrator(velocity, alpha));
|
||||
k.AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
|
||||
k.AddInteriorFaceIntegrator(
|
||||
new NonconservativeDGTraceIntegrator(velocity, alpha));
|
||||
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
|
||||
k.AddBdrFaceIntegrator(
|
||||
new NonconservativeDGTraceIntegrator(velocity, alpha));
|
||||
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
|
||||
|
||||
LinearForm b(&fes);
|
||||
b.AddBdrFaceIntegrator(
|
||||
new BoundaryFlowIntegrator(inflow, velocity, alpha));
|
||||
new BoundaryFlowIntegrator(inflow, velocity, -1.0, -0.5));
|
||||
|
||||
m.Assemble();
|
||||
int skip_zeros = 0;
|
||||
|
||||
+22
-114
@@ -63,66 +63,6 @@ double inflow_function(const Vector &x);
|
||||
// Mesh bounding box
|
||||
Vector bb_min, bb_max;
|
||||
|
||||
// Type of preconditioner for implicit time integrator
|
||||
enum class PrecType : int
|
||||
{
|
||||
ILU = 0,
|
||||
AIR = 1
|
||||
};
|
||||
|
||||
#if MFEM_HYPRE_VERSION >= 21800
|
||||
// Algebraic multigrid preconditioner for advective problems based on
|
||||
// approximate ideal restriction (AIR). Most effective when matrix is
|
||||
// first scaled by DG block inverse, and AIR applied to scaled matrix.
|
||||
// See https://doi.org/10.1137/17M1144350.
|
||||
class AIR_prec : public Solver
|
||||
{
|
||||
private:
|
||||
const HypreParMatrix *A;
|
||||
// Copy of A scaled by block-diagonal inverse
|
||||
HypreParMatrix A_s;
|
||||
|
||||
HypreBoomerAMG *AIR_solver;
|
||||
int blocksize;
|
||||
|
||||
public:
|
||||
AIR_prec(int blocksize_) : AIR_solver(NULL), blocksize(blocksize_) { }
|
||||
|
||||
void SetOperator(const Operator &op)
|
||||
{
|
||||
width = op.Width();
|
||||
height = op.Height();
|
||||
|
||||
A = dynamic_cast<const HypreParMatrix *>(&op);
|
||||
MFEM_VERIFY(A != NULL, "AIR_prec requires a HypreParMatrix.")
|
||||
|
||||
// Scale A by block-diagonal inverse
|
||||
BlockInverseScale(A, &A_s, NULL, NULL, blocksize,
|
||||
BlockInverseScaleJob::MATRIX_ONLY);
|
||||
delete AIR_solver;
|
||||
AIR_solver = new HypreBoomerAMG(A_s);
|
||||
AIR_solver->SetAdvectiveOptions(1, "", "FA");
|
||||
AIR_solver->SetPrintLevel(0);
|
||||
AIR_solver->SetMaxLevels(50);
|
||||
}
|
||||
|
||||
virtual void Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
// Scale the rhs by block inverse and solve system
|
||||
HypreParVector z_s;
|
||||
BlockInverseScale(A, NULL, &x, &z_s, blocksize,
|
||||
BlockInverseScaleJob::RHS_ONLY);
|
||||
AIR_solver->Mult(z_s, y);
|
||||
}
|
||||
|
||||
~AIR_prec()
|
||||
{
|
||||
delete AIR_solver;
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
|
||||
class DG_Solver : public Solver
|
||||
{
|
||||
private:
|
||||
@@ -130,37 +70,24 @@ private:
|
||||
SparseMatrix M_diag;
|
||||
HypreParMatrix *A;
|
||||
GMRESSolver linear_solver;
|
||||
Solver *prec;
|
||||
BlockILU prec;
|
||||
double dt;
|
||||
public:
|
||||
DG_Solver(HypreParMatrix &M_, HypreParMatrix &K_, const FiniteElementSpace &fes,
|
||||
PrecType prec_type)
|
||||
DG_Solver(HypreParMatrix &M_, HypreParMatrix &K_, const FiniteElementSpace &fes)
|
||||
: M(M_),
|
||||
K(K_),
|
||||
A(NULL),
|
||||
linear_solver(M.GetComm()),
|
||||
prec(fes.GetFE(0)->GetDof(),
|
||||
BlockILU::Reordering::MINIMUM_DISCARDED_FILL),
|
||||
dt(-1.0)
|
||||
{
|
||||
int block_size = fes.GetFE(0)->GetDof();
|
||||
if (prec_type == PrecType::ILU)
|
||||
{
|
||||
prec = new BlockILU(block_size,
|
||||
BlockILU::Reordering::MINIMUM_DISCARDED_FILL);
|
||||
}
|
||||
else if (prec_type == PrecType::AIR)
|
||||
{
|
||||
#if MFEM_HYPRE_VERSION >= 21800
|
||||
prec = new AIR_prec(block_size);
|
||||
#else
|
||||
MFEM_ABORT("Must have MFEM_HYPRE_VERSION >= 21800 to use AIR.\n");
|
||||
#endif
|
||||
}
|
||||
linear_solver.iterative_mode = false;
|
||||
linear_solver.SetRelTol(1e-9);
|
||||
linear_solver.SetAbsTol(0.0);
|
||||
linear_solver.SetMaxIter(100);
|
||||
linear_solver.SetPrintLevel(0);
|
||||
linear_solver.SetPreconditioner(*prec);
|
||||
linear_solver.SetPreconditioner(prec);
|
||||
|
||||
M.GetDiag(M_diag);
|
||||
}
|
||||
@@ -193,12 +120,10 @@ public:
|
||||
|
||||
~DG_Solver()
|
||||
{
|
||||
delete prec;
|
||||
delete A;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
/** A time-dependent operator for the right-hand side of the ODE. The DG weak
|
||||
form of du/dt = -v.grad(u) is M du/dt = K u + b, where M and K are the mass
|
||||
and advection matrices, and b describes the flow on the boundary. This can
|
||||
@@ -216,8 +141,7 @@ private:
|
||||
mutable Vector z;
|
||||
|
||||
public:
|
||||
FE_Evolution(ParBilinearForm &M_, ParBilinearForm &K_, const Vector &b_,
|
||||
PrecType prec_type);
|
||||
FE_Evolution(ParBilinearForm &M_, ParBilinearForm &K_, const Vector &b_);
|
||||
|
||||
virtual void Mult(const Vector &x, Vector &y) const;
|
||||
virtual void ImplicitSolve(const double dt, const Vector &x, Vector &k);
|
||||
@@ -254,11 +178,6 @@ int main(int argc, char *argv[])
|
||||
bool adios2 = false;
|
||||
bool binary = false;
|
||||
int vis_steps = 5;
|
||||
#if MFEM_HYPRE_VERSION >= 21800
|
||||
PrecType prec_type = PrecType::AIR;
|
||||
#else
|
||||
PrecType prec_type = PrecType::ILU;
|
||||
#endif
|
||||
|
||||
// Relative and absolute tolerances for CVODE and ARKODE.
|
||||
const double reltol = 1e-2, abstol = 1e-2;
|
||||
@@ -299,8 +218,6 @@ int main(int argc, char *argv[])
|
||||
"Final time; start time is 0.");
|
||||
args.AddOption(&dt, "-dt", "--time-step",
|
||||
"Time step.");
|
||||
args.AddOption((int *)&prec_type, "-pt", "--prec-type", "Preconditioner for "
|
||||
"implicit solves. 0 for ILU, 1 for pAIR-AMG.");
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
"Enable or disable GLVis visualization.");
|
||||
@@ -321,13 +238,13 @@ int main(int argc, char *argv[])
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
{
|
||||
if (Mpi::Root())
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintUsage(cout);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
if (Mpi::Root())
|
||||
if (myid == 0)
|
||||
{
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
@@ -335,7 +252,7 @@ int main(int argc, char *argv[])
|
||||
// check for valid ODE solver option
|
||||
if (ode_solver_type < 1 || ode_solver_type > 9)
|
||||
{
|
||||
if (Mpi::Root())
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Unknown ODE solver type: " << ode_solver_type << '\n';
|
||||
}
|
||||
@@ -343,7 +260,7 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
Device device(device_config);
|
||||
if (Mpi::Root()) { device.Print(); }
|
||||
if (myid == 0) { device.Print(); }
|
||||
|
||||
// 3. Read the serial mesh from the given mesh file on all processors. We can
|
||||
// handle geometrically periodic meshes in this code.
|
||||
@@ -380,7 +297,7 @@ int main(int argc, char *argv[])
|
||||
ParFiniteElementSpace *fes = new ParFiniteElementSpace(pmesh, &fec);
|
||||
|
||||
HYPRE_BigInt global_vSize = fes->GlobalTrueVSize();
|
||||
if (Mpi::Root())
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Number of unknowns: " << global_vSize << endl;
|
||||
}
|
||||
@@ -411,16 +328,15 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
m->AddDomainIntegrator(new MassIntegrator);
|
||||
constexpr double alpha = -1.0;
|
||||
k->AddDomainIntegrator(new ConvectionIntegrator(velocity, alpha));
|
||||
k->AddDomainIntegrator(new ConvectionIntegrator(velocity, -1.0));
|
||||
k->AddInteriorFaceIntegrator(
|
||||
new NonconservativeDGTraceIntegrator(velocity, alpha));
|
||||
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
|
||||
k->AddBdrFaceIntegrator(
|
||||
new NonconservativeDGTraceIntegrator(velocity, alpha));
|
||||
new TransposeIntegrator(new DGTraceIntegrator(velocity, 1.0, -0.5)));
|
||||
|
||||
ParLinearForm *b = new ParLinearForm(fes);
|
||||
b->AddBdrFaceIntegrator(
|
||||
new BoundaryFlowIntegrator(inflow, velocity, alpha));
|
||||
new BoundaryFlowIntegrator(inflow, velocity, -1.0, -0.5));
|
||||
|
||||
int skip_zeros = 0;
|
||||
m->Assemble();
|
||||
@@ -519,13 +435,11 @@ int main(int argc, char *argv[])
|
||||
sout.open(vishost, visport);
|
||||
if (!sout)
|
||||
{
|
||||
if (Mpi::Root())
|
||||
{
|
||||
if (myid == 0)
|
||||
cout << "Unable to connect to GLVis server at "
|
||||
<< vishost << ':' << visport << endl;
|
||||
}
|
||||
visualization = false;
|
||||
if (Mpi::Root())
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "GLVis visualization disabled.\n";
|
||||
}
|
||||
@@ -537,17 +451,15 @@ int main(int argc, char *argv[])
|
||||
sout << "solution\n" << *pmesh << *u;
|
||||
sout << "pause\n";
|
||||
sout << flush;
|
||||
if (Mpi::Root())
|
||||
{
|
||||
if (myid == 0)
|
||||
cout << "GLVis visualization paused."
|
||||
<< " Press space (in the GLVis window) to resume it.\n";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 9. Define the time-dependent evolution operator describing the ODE
|
||||
// right-hand side, and define the ODE solver used for time integration.
|
||||
FE_Evolution adv(*m, *k, *B, prec_type);
|
||||
FE_Evolution adv(*m, *k, *B);
|
||||
|
||||
double t = 0.0;
|
||||
adv.SetTime(t);
|
||||
@@ -599,7 +511,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
if (done || ti % vis_steps == 0)
|
||||
{
|
||||
if (Mpi::Root())
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "time step: " << ti << ", time: " << t << endl;
|
||||
if (cvode) { cvode->PrintInfo(); }
|
||||
@@ -678,7 +590,7 @@ int main(int argc, char *argv[])
|
||||
|
||||
// Implementation of class FE_Evolution
|
||||
FE_Evolution::FE_Evolution(ParBilinearForm &M_, ParBilinearForm &K_,
|
||||
const Vector &b_, PrecType prec_type)
|
||||
const Vector &b_)
|
||||
: TimeDependentOperator(M_.Height()),
|
||||
b(b_),
|
||||
M_solver(M_.ParFESpace()->GetComm()),
|
||||
@@ -705,7 +617,7 @@ FE_Evolution::FE_Evolution(ParBilinearForm &M_, ParBilinearForm &K_,
|
||||
HypreSmoother *hypre_prec = new HypreSmoother(M_mat, HypreSmoother::Jacobi);
|
||||
M_prec = hypre_prec;
|
||||
|
||||
dg_solver = new DG_Solver(M_mat, K_mat, *M_.FESpace(), prec_type);
|
||||
dg_solver = new DG_Solver(M_mat, K_mat, *M_.FESpace());
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -721,10 +633,6 @@ FE_Evolution::FE_Evolution(ParBilinearForm &M_, ParBilinearForm &K_,
|
||||
M_solver.SetPrintLevel(0);
|
||||
}
|
||||
|
||||
// Solve the equation:
|
||||
// u_t = M^{-1}(Ku + b),
|
||||
// by solving associated linear system
|
||||
// (M - dt*K) d = K*u + b
|
||||
void FE_Evolution::ImplicitSolve(const double dt, const Vector &x, Vector &k)
|
||||
{
|
||||
K->Mult(x, z);
|
||||
|
||||
@@ -23,8 +23,6 @@ MFEM_LIB_FILE = mfem_is_not_built
|
||||
|
||||
SEQ_EXAMPLES = ex9 ex10 ex16
|
||||
PAR_EXAMPLES = ex9p ex10p ex16p
|
||||
SEQ_DEVICE_EXAMPLES = ex9
|
||||
PAR_DEVICE_EXAMPLES = ex9p
|
||||
ifeq ($(MFEM_USE_MPI),NO)
|
||||
EXAMPLES = $(SEQ_EXAMPLES)
|
||||
else
|
||||
@@ -56,22 +54,10 @@ include $(MFEM_TEST_MK)
|
||||
RUN_MPI = $(MFEM_MPIEXEC) $(MFEM_MPIEXEC_NP) $(MFEM_MPI_NP)
|
||||
SERIAL_NAME := Serial SUNDIALS example
|
||||
PARALLEL_NAME := Parallel SUNDIALS example
|
||||
SERIAL_CUDA_NAME := Serial SUNDIALS CUDA example
|
||||
PARALLEL_CUDA_NAME := Parallel SUNDIALS CUDA example
|
||||
SERIAL_HIP_NAME := Serial SUNDIALS HIP example
|
||||
PARALLEL_HIP_NAME := Parallel SUNDIALS HIP example
|
||||
%-test-par: %
|
||||
@$(call mfem-test,$<, $(RUN_MPI), $(PARALLEL_NAME))
|
||||
%-test-seq: %
|
||||
@$(call mfem-test,$<,, $(SERIAL_NAME))
|
||||
%-test-par-cuda: %
|
||||
@$(call mfem-test,$<, $(RUN_MPI), $(PARALLEL_CUDA_NAME),-d cuda)
|
||||
%-test-seq-cuda: %
|
||||
@$(call mfem-test,$<,, $(SERIAL_CUDA_NAME),-d cuda)
|
||||
%-test-par-hip: %
|
||||
@$(call mfem-test,$<, $(RUN_MPI), $(PARALLEL_HIP_NAME),-d hip)
|
||||
%-test-seq-hip: %
|
||||
@$(call mfem-test,$<,, $(SERIAL_HIP_NAME),-d hip)
|
||||
|
||||
# Testing: Specific execution options:
|
||||
# Example 9: test CVODE with CV_ADAMS (non-stiff implicit) time stepping
|
||||
@@ -82,16 +68,6 @@ ex9-test-seq: ex9
|
||||
@$(call mfem-test,$<,, $(SERIAL_NAME),$(EX9_ARGS))
|
||||
ex9p-test-par: ex9p
|
||||
@$(call mfem-test,$<, $(RUN_MPI), $(PARALLEL_NAME),$(EX9P_ARGS))
|
||||
ex9-test-seq-cuda: ex9
|
||||
@$(call mfem-test,$<,, $(SERIAL_CUDA_NAME),-d cuda $(EX9_ARGS))
|
||||
ex9p-test-par-cuda: ex9p
|
||||
@$(call mfem-test,$<, $(RUN_MPI), $(PARALLEL_CUDA_NAME),-d cuda \
|
||||
$(EX9P_ARGS))
|
||||
ex9-test-seq-hip: ex9
|
||||
@$(call mfem-test,$<,, $(SERIAL_HIP_NAME),-d hip $(EX9_ARGS))
|
||||
ex9p-test-par-hip: ex9p
|
||||
@$(call mfem-test,$<, $(RUN_MPI), $(PARALLEL_HIP_NAME),-d hip \
|
||||
$(EX9P_ARGS))
|
||||
# Example 10: test CVODE with CV_BDF (stiff implicit) time stepping
|
||||
EX10_COMMON_ARGS := -m ../../data/beam-quad.mesh -o 2 -s 5 -dt 0.15 -tf 6 -vs 10
|
||||
EX10_ARGS := $(EX10_COMMON_ARGS) -r 2
|
||||
|
||||
@@ -67,7 +67,6 @@ int main(int argc, char *argv[])
|
||||
int slu_colperm = 4;
|
||||
int slu_rowperm = 1;
|
||||
int slu_iterref = 2;
|
||||
int slu_npdep = 1;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
@@ -86,11 +85,9 @@ int main(int argc, char *argv[])
|
||||
"6-ZOLTAN");
|
||||
args.AddOption(&slu_rowperm, "-rp", "--rowperm",
|
||||
"SuperLU Row Permutation Method: 0-NOROWPERM, 1-LargeDiag");
|
||||
args.AddOption(&slu_iterref, "-ir", "--iterref",
|
||||
args.AddOption(&slu_iterref, "-rp", "--rowperm",
|
||||
"SuperLU Iterative Refinement: 0-NOREFINE, 1-Single, "
|
||||
"2-Double, 3-Extra");
|
||||
args.AddOption(&slu_npdep, "-npdep", "--npdepth",
|
||||
"Depth of 3D parition for SuperLU (>= 7.2.0)");
|
||||
|
||||
args.Parse();
|
||||
if (!args.Good())
|
||||
@@ -217,7 +214,7 @@ int main(int argc, char *argv[])
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
|
||||
// 13. Solve the linear system A X = B utilizing SuperLU.
|
||||
SuperLUSolver *superlu = new SuperLUSolver(MPI_COMM_WORLD, slu_npdep);
|
||||
SuperLUSolver *superlu = new SuperLUSolver(MPI_COMM_WORLD);
|
||||
Operator *SLU_A = new SuperLURowLocMatrix(*A.As<HypreParMatrix>());
|
||||
superlu->SetPrintStatistics(true);
|
||||
superlu->SetSymmetricPattern(false);
|
||||
@@ -284,9 +281,10 @@ int main(int argc, char *argv[])
|
||||
superlu->SetOperator(*SLU_A);
|
||||
superlu->SetPrintStatistics(true);
|
||||
superlu->Mult(B, X);
|
||||
superlu->DismantleGrid();
|
||||
|
||||
delete superlu;
|
||||
delete SLU_A;
|
||||
delete superlu;
|
||||
|
||||
// 14. Recover the parallel grid function corresponding to X. This is the
|
||||
// local finite element solution on each processor.
|
||||
|
||||
+28
-45
@@ -13,44 +13,28 @@ set(SRCS
|
||||
bilinearform.cpp
|
||||
bilinearform_ext.cpp
|
||||
bilininteg.cpp
|
||||
integ/bilininteg_br2.cpp
|
||||
integ/bilininteg_convection_mf.cpp
|
||||
integ/bilininteg_convection_pa.cpp
|
||||
integ/bilininteg_convection_ea.cpp
|
||||
integ/bilininteg_curlcurl_pa.cpp
|
||||
integ/bilininteg_dgtrace_pa.cpp
|
||||
integ/bilininteg_dgtrace_ea.cpp
|
||||
integ/bilininteg_diffusion_mf.cpp
|
||||
integ/bilininteg_diffusion_pa.cpp
|
||||
integ/bilininteg_diffusion_ea.cpp
|
||||
integ/bilininteg_divdiv_pa.cpp
|
||||
integ/bilininteg_gradient_pa.cpp
|
||||
integ/bilininteg_interp_pa.cpp
|
||||
integ/bilininteg_mass_mf.cpp
|
||||
integ/bilininteg_mass_pa.cpp
|
||||
integ/bilininteg_mass_ea.cpp
|
||||
integ/bilininteg_mixedcurl_pa.cpp
|
||||
integ/bilininteg_mixedvecgrad_pa.cpp
|
||||
integ/bilininteg_transpose_ea.cpp
|
||||
integ/bilininteg_vecdiffusion_mf.cpp
|
||||
integ/bilininteg_vecdiffusion_pa.cpp
|
||||
integ/bilininteg_vecdiv_pa.cpp
|
||||
integ/bilininteg_vecmass_mf.cpp
|
||||
integ/bilininteg_vecmass_pa.cpp
|
||||
integ/bilininteg_vectorfediv_pa.cpp
|
||||
integ/bilininteg_vectorfemass_pa.cpp
|
||||
integ/bilininteg_diffusion_kernels.cpp
|
||||
integ/bilininteg_hcurl_kernels.cpp
|
||||
integ/bilininteg_hdiv_kernels.cpp
|
||||
integ/bilininteg_hcurlhdiv_kernels.cpp
|
||||
integ/bilininteg_mass_kernels.cpp
|
||||
integ/lininteg_boundary.cpp
|
||||
integ/lininteg_boundary_flux.cpp
|
||||
integ/lininteg_domain.cpp
|
||||
integ/lininteg_domain_grad.cpp
|
||||
integ/lininteg_domain_vectorfe.cpp
|
||||
integ/nonlininteg_vecconvection_pa.cpp
|
||||
integ/nonlininteg_vecconvection_mf.cpp
|
||||
bilininteg_br2.cpp
|
||||
bilininteg_convection_mf.cpp
|
||||
bilininteg_convection_pa.cpp
|
||||
bilininteg_convection_ea.cpp
|
||||
bilininteg_dgtrace_pa.cpp
|
||||
bilininteg_dgtrace_ea.cpp
|
||||
bilininteg_diffusion_mf.cpp
|
||||
bilininteg_diffusion_pa.cpp
|
||||
bilininteg_diffusion_ea.cpp
|
||||
bilininteg_divergence.cpp
|
||||
bilininteg_hcurl.cpp
|
||||
bilininteg_hdiv.cpp
|
||||
bilininteg_vectorfe.cpp
|
||||
bilininteg_gradient.cpp
|
||||
bilininteg_mass_mf.cpp
|
||||
bilininteg_mass_pa.cpp
|
||||
bilininteg_mass_ea.cpp
|
||||
bilininteg_transpose_ea.cpp
|
||||
bilininteg_vecdiffusion.cpp
|
||||
bilininteg_vecdiffusion_mf.cpp
|
||||
bilininteg_vecmass.cpp
|
||||
bilininteg_vecmass_mf.cpp
|
||||
coefficient.cpp
|
||||
complex_fem.cpp
|
||||
convergence.cpp
|
||||
@@ -60,7 +44,6 @@ set(SRCS
|
||||
eltrans.cpp
|
||||
estimators.cpp
|
||||
fe.cpp
|
||||
fe/face_map_utils.cpp
|
||||
fe/fe_base.cpp
|
||||
fe/fe_fixed_order.cpp
|
||||
fe/fe_h1.cpp
|
||||
@@ -90,6 +73,9 @@ set(SRCS
|
||||
linearform.cpp
|
||||
linearform_ext.cpp
|
||||
lininteg.cpp
|
||||
lininteg_boundary.cpp
|
||||
lininteg_domain.cpp
|
||||
lininteg_domain_grad.cpp
|
||||
lor/lor.cpp
|
||||
lor/lor_ads.cpp
|
||||
lor/lor_ams.cpp
|
||||
@@ -102,6 +88,8 @@ set(SRCS
|
||||
nonlinearform_ext.cpp
|
||||
nonlininteg.cpp
|
||||
fespacehierarchy.cpp
|
||||
nonlininteg_vectorconvection.cpp
|
||||
nonlininteg_vectorconvection_mf.cpp
|
||||
qfunction.cpp
|
||||
qinterp/det.cpp
|
||||
qinterp/eval_by_nodes.cpp
|
||||
@@ -152,11 +140,7 @@ set(HDRS
|
||||
bilinearform.hpp
|
||||
bilinearform_ext.hpp
|
||||
bilininteg.hpp
|
||||
integ/bilininteg_diffusion_kernels.hpp
|
||||
integ/bilininteg_hcurl_kernels.hpp
|
||||
integ/bilininteg_hdiv_kernels.hpp
|
||||
integ/bilininteg_hcurlhdiv_kernels.hpp
|
||||
integ/bilininteg_mass_kernels.hpp
|
||||
bilininteg_mass_pa.hpp
|
||||
coefficient.hpp
|
||||
complex_fem.hpp
|
||||
convergence.hpp
|
||||
@@ -167,7 +151,6 @@ set(HDRS
|
||||
eltrans.hpp
|
||||
estimators.hpp
|
||||
fe.hpp
|
||||
fe/face_map_utils.hpp
|
||||
fe/fe_base.hpp
|
||||
fe/fe_fixed_order.hpp
|
||||
fe/fe_h1.hpp
|
||||
|
||||
+48
-86
@@ -56,9 +56,6 @@ void MFBilinearFormExtension::Assemble()
|
||||
{
|
||||
integrators[i]->AssembleMF(*a->FESpace());
|
||||
}
|
||||
|
||||
MFEM_VERIFY(a->GetBBFI()->Size() == 0, "AddBoundaryIntegrator is not "
|
||||
"currently supported in MFBilinearFormExtension");
|
||||
}
|
||||
|
||||
void MFBilinearFormExtension::AssembleDiagonal(Vector &y) const
|
||||
@@ -278,9 +275,7 @@ void PABilinearFormExtension::SetupRestrictionOperators(const L2FaceValues m)
|
||||
int_face_Y.UseDevice(true); // ensure 'int_face_Y = 0.0' is done on device
|
||||
}
|
||||
|
||||
const bool has_bdr_integs = (a->GetBFBFI()->Size() > 0 ||
|
||||
a->GetBBFI()->Size() > 0);
|
||||
if (bdr_face_restrict_lex == NULL && has_bdr_integs)
|
||||
if (bdr_face_restrict_lex == NULL && a->GetBFBFI()->Size() > 0)
|
||||
{
|
||||
bdr_face_restrict_lex = trial_fes->GetFaceRestriction(
|
||||
ElementDofOrdering::LEXICOGRAPHIC,
|
||||
@@ -297,27 +292,27 @@ void PABilinearFormExtension::Assemble()
|
||||
SetupRestrictionOperators(L2FaceValues::DoubleValued);
|
||||
|
||||
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
|
||||
for (BilinearFormIntegrator *integ : integrators)
|
||||
const int integratorCount = integrators.Size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integ->AssemblePA(*a->FESpace());
|
||||
integrators[i]->AssemblePA(*a->FESpace());
|
||||
}
|
||||
|
||||
Array<BilinearFormIntegrator*> &bdr_integrators = *a->GetBBFI();
|
||||
for (BilinearFormIntegrator *integ : bdr_integrators)
|
||||
{
|
||||
integ->AssemblePABoundary(*a->FESpace());
|
||||
}
|
||||
MFEM_VERIFY(a->GetBBFI()->Size() == 0,
|
||||
"Partial assembly does not support AddBoundaryIntegrator yet.");
|
||||
|
||||
Array<BilinearFormIntegrator*> &intFaceIntegrators = *a->GetFBFI();
|
||||
for (BilinearFormIntegrator *integ : intFaceIntegrators)
|
||||
const int intFaceIntegratorCount = intFaceIntegrators.Size();
|
||||
for (int i = 0; i < intFaceIntegratorCount; ++i)
|
||||
{
|
||||
integ->AssemblePAInteriorFaces(*a->FESpace());
|
||||
intFaceIntegrators[i]->AssemblePAInteriorFaces(*a->FESpace());
|
||||
}
|
||||
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
for (BilinearFormIntegrator *integ : bdrFaceIntegrators)
|
||||
const int boundFaceIntegratorCount = bdrFaceIntegrators.Size();
|
||||
for (int i = 0; i < boundFaceIntegratorCount; ++i)
|
||||
{
|
||||
integ->AssemblePABoundaryFaces(*a->FESpace());
|
||||
bdrFaceIntegrators[i]->AssemblePABoundaryFaces(*a->FESpace());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -328,27 +323,20 @@ void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
|
||||
const int iSz = integrators.Size();
|
||||
if (elem_restrict && !DeviceCanUseCeed())
|
||||
{
|
||||
if (iSz > 0)
|
||||
localY = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
{
|
||||
localY = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
{
|
||||
integrators[i]->AssembleDiagonalPA(localY);
|
||||
}
|
||||
const ElementRestriction* H1elem_restrict =
|
||||
dynamic_cast<const ElementRestriction*>(elem_restrict);
|
||||
if (H1elem_restrict)
|
||||
{
|
||||
H1elem_restrict->MultTransposeUnsigned(localY, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
elem_restrict->MultTranspose(localY, y);
|
||||
}
|
||||
integrators[i]->AssembleDiagonalPA(localY);
|
||||
}
|
||||
const ElementRestriction* H1elem_restrict =
|
||||
dynamic_cast<const ElementRestriction*>(elem_restrict);
|
||||
if (H1elem_restrict)
|
||||
{
|
||||
H1elem_restrict->MultTransposeUnsigned(localY, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
y = 0.0;
|
||||
elem_restrict->MultTranspose(localY, y);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -360,18 +348,6 @@ void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
|
||||
integrators[i]->AssembleDiagonalPA(y);
|
||||
}
|
||||
}
|
||||
|
||||
Array<BilinearFormIntegrator*> &bdr_integs = *a->GetBBFI();
|
||||
const int n_bdr_integs = bdr_integs.Size();
|
||||
if (bdr_face_restrict_lex && n_bdr_integs > 0)
|
||||
{
|
||||
bdr_face_Y = 0.0;
|
||||
for (int i = 0; i < n_bdr_integs; ++i)
|
||||
{
|
||||
bdr_integs[i]->AssembleDiagonalPA(bdr_face_Y);
|
||||
}
|
||||
bdr_face_restrict_lex->AddMultTransposeUnsigned(bdr_face_Y, y);
|
||||
}
|
||||
}
|
||||
|
||||
void PABilinearFormExtension::Update()
|
||||
@@ -421,20 +397,13 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
}
|
||||
else
|
||||
{
|
||||
if (iSz)
|
||||
elem_restrict->Mult(x, localX);
|
||||
localY = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
{
|
||||
elem_restrict->Mult(x, localX);
|
||||
localY = 0.0;
|
||||
for (int i = 0; i < iSz; ++i)
|
||||
{
|
||||
integrators[i]->AddMultPA(localX, localY);
|
||||
}
|
||||
elem_restrict->MultTranspose(localY, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
y = 0.0;
|
||||
integrators[i]->AddMultPA(localX, localY);
|
||||
}
|
||||
elem_restrict->MultTranspose(localY, y);
|
||||
}
|
||||
|
||||
Array<BilinearFormIntegrator*> &intFaceIntegrators = *a->GetFBFI();
|
||||
@@ -453,24 +422,17 @@ void PABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
Array<BilinearFormIntegrator*> &bdr_integs = *a->GetBBFI();
|
||||
Array<BilinearFormIntegrator*> &bdr_face_integs = *a->GetBFBFI();
|
||||
const int n_bdr_integs = bdr_integs.Size();
|
||||
const int n_bdr_face_integs = bdr_face_integs.Size();
|
||||
const bool has_bdr_integs = (n_bdr_face_integs > 0 || n_bdr_integs > 0);
|
||||
if (bdr_face_restrict_lex && has_bdr_integs)
|
||||
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
|
||||
const int bFISz = bdrFaceIntegrators.Size();
|
||||
if (bdr_face_restrict_lex && bFISz>0)
|
||||
{
|
||||
bdr_face_restrict_lex->Mult(x, bdr_face_X);
|
||||
if (bdr_face_X.Size()>0)
|
||||
{
|
||||
bdr_face_Y = 0.0;
|
||||
for (int i = 0; i < n_bdr_integs; ++i)
|
||||
for (int i = 0; i < bFISz; ++i)
|
||||
{
|
||||
bdr_integs[i]->AddMultPA(bdr_face_X, bdr_face_Y);
|
||||
}
|
||||
for (int i = 0; i < n_bdr_face_integs; ++i)
|
||||
{
|
||||
bdr_face_integs[i]->AddMultPA(bdr_face_X, bdr_face_Y);
|
||||
bdrFaceIntegrators[i]->AddMultPA(bdr_face_X, bdr_face_Y);
|
||||
}
|
||||
bdr_face_restrict_lex->AddMultTransposeInPlace(bdr_face_Y, y);
|
||||
}
|
||||
@@ -634,7 +596,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
auto X = Reshape(useRestrict?localX.Read():x.Read(), NDOFS, ne);
|
||||
auto Y = Reshape(useRestrict?localY.ReadWrite():y.ReadWrite(), NDOFS, ne);
|
||||
auto A = Reshape(ea_data.Read(), NDOFS, NDOFS, ne);
|
||||
mfem::forall(ne*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
MFEM_FORALL(glob_j, ne*NDOFS,
|
||||
{
|
||||
const int e = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
@@ -669,7 +631,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
if (!factorize_face_terms)
|
||||
{
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
@@ -688,7 +650,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
});
|
||||
}
|
||||
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
@@ -725,7 +687,7 @@ void EABilinearFormExtension::Mult(const Vector &x, Vector &y) const
|
||||
auto X = Reshape(bdr_face_X.Read(), NDOFS, nf_bdr);
|
||||
auto Y = Reshape(bdr_face_Y.ReadWrite(), NDOFS, nf_bdr);
|
||||
auto A = Reshape(ea_data_bdr.Read(), NDOFS, NDOFS, nf_bdr);
|
||||
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
MFEM_FORALL(glob_j, nf_bdr*NDOFS,
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
@@ -762,7 +724,7 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
auto X = Reshape(useRestrict?localX.Read():x.Read(), NDOFS, ne);
|
||||
auto Y = Reshape(useRestrict?localY.ReadWrite():y.ReadWrite(), NDOFS, ne);
|
||||
auto A = Reshape(ea_data.Read(), NDOFS, NDOFS, ne);
|
||||
mfem::forall(ne*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
MFEM_FORALL(glob_j, ne*NDOFS,
|
||||
{
|
||||
const int e = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
@@ -797,7 +759,7 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
if (!factorize_face_terms)
|
||||
{
|
||||
auto A_int = Reshape(ea_data_int.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
@@ -816,7 +778,7 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
});
|
||||
}
|
||||
auto A_ext = Reshape(ea_data_ext.Read(), NDOFS, NDOFS, 2, nf_int);
|
||||
mfem::forall(nf_int*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
MFEM_FORALL(glob_j, nf_int*NDOFS,
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
@@ -853,7 +815,7 @@ void EABilinearFormExtension::MultTranspose(const Vector &x, Vector &y) const
|
||||
auto X = Reshape(bdr_face_X.Read(), NDOFS, nf_bdr);
|
||||
auto Y = Reshape(bdr_face_Y.ReadWrite(), NDOFS, nf_bdr);
|
||||
auto A = Reshape(ea_data_bdr.Read(), NDOFS, NDOFS, nf_bdr);
|
||||
mfem::forall(nf_bdr*NDOFS, [=] MFEM_HOST_DEVICE (int glob_j)
|
||||
MFEM_FORALL(glob_j, nf_bdr*NDOFS,
|
||||
{
|
||||
const int f = glob_j/NDOFS;
|
||||
const int j = glob_j%NDOFS;
|
||||
@@ -1068,13 +1030,13 @@ void FABilinearFormExtension::DGMult(const Vector &x, Vector &y) const
|
||||
const int local_size = a->FESpace()->GetVSize();
|
||||
auto dg_x_ptr = dg_x.Write();
|
||||
auto x_ptr = x.Read();
|
||||
mfem::forall(local_size, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i,local_size,
|
||||
{
|
||||
dg_x_ptr[i] = x_ptr[i];
|
||||
});
|
||||
const int shared_size = shared_x.Size();
|
||||
auto shared_x_ptr = shared_x.Read();
|
||||
mfem::forall(shared_size, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i,shared_size,
|
||||
{
|
||||
dg_x_ptr[local_size+i] = shared_x_ptr[i];
|
||||
});
|
||||
@@ -1085,7 +1047,7 @@ void FABilinearFormExtension::DGMult(const Vector &x, Vector &y) const
|
||||
// DG Restriction
|
||||
auto dg_y_ptr = dg_y.Read();
|
||||
auto y_ptr = y.ReadWrite();
|
||||
mfem::forall(local_size, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i,local_size,
|
||||
{
|
||||
y_ptr[i] += dg_y_ptr[i];
|
||||
});
|
||||
@@ -1129,13 +1091,13 @@ void FABilinearFormExtension::DGMultTranspose(const Vector &x, Vector &y) const
|
||||
const int local_size = a->FESpace()->GetVSize();
|
||||
auto dg_x_ptr = dg_x.Write();
|
||||
auto x_ptr = x.Read();
|
||||
mfem::forall(local_size, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i,local_size,
|
||||
{
|
||||
dg_x_ptr[i] = x_ptr[i];
|
||||
});
|
||||
const int shared_size = shared_x.Size();
|
||||
auto shared_x_ptr = shared_x.Read();
|
||||
mfem::forall(shared_size, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i,shared_size,
|
||||
{
|
||||
dg_x_ptr[local_size+i] = shared_x_ptr[i];
|
||||
});
|
||||
@@ -1146,7 +1108,7 @@ void FABilinearFormExtension::DGMultTranspose(const Vector &x, Vector &y) const
|
||||
// DG Restriction
|
||||
auto dg_y_ptr = dg_y.Read();
|
||||
auto y_ptr = y.ReadWrite();
|
||||
mfem::forall(local_size, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i,local_size,
|
||||
{
|
||||
y_ptr[i] += dg_y_ptr[i];
|
||||
});
|
||||
@@ -1484,7 +1446,7 @@ void PADiscreteLinearOperatorExtension::Assemble()
|
||||
}
|
||||
|
||||
auto tm = test_multiplicity.ReadWrite();
|
||||
mfem::forall(test_multiplicity.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i, test_multiplicity.Size(),
|
||||
{
|
||||
tm[i] = 1.0 / tm[i];
|
||||
});
|
||||
@@ -1536,7 +1498,7 @@ void PADiscreteLinearOperatorExtension::AddMultTranspose(
|
||||
MFEM_VERIFY(x.Size() == test_multiplicity.Size(), "Input vector of wrong size");
|
||||
auto xs = xscaled.ReadWrite();
|
||||
auto tm = test_multiplicity.Read();
|
||||
mfem::forall(x.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i, x.Size(),
|
||||
{
|
||||
xs[i] *= tm[i];
|
||||
});
|
||||
|
||||
+35
-261
@@ -22,47 +22,41 @@ namespace mfem
|
||||
|
||||
void BilinearFormIntegrator::AssemblePA(const FiniteElementSpace&)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssemblePA(fes)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssemblePA(fes)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssemblePA(const FiniteElementSpace&,
|
||||
const FiniteElementSpace&)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssemblePA(fes, fes)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssemblePABoundary(const FiniteElementSpace&)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssemblePABoundary(fes)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssemblePA(fes, fes)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssemblePAInteriorFaces(const FiniteElementSpace&)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssemblePAInteriorFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssemblePAInteriorFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssemblePABoundaryFaces(const FiniteElementSpace&)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssemblePABoundaryFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssemblePABoundaryFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleDiagonalPA(Vector &)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleDiagonalPA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleDiagonalPA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleEA(const FiniteElementSpace &fes,
|
||||
Vector &emat,
|
||||
const bool add)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleEA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleEA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace
|
||||
@@ -71,8 +65,8 @@ void BilinearFormIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace
|
||||
Vector &ea_data_ext,
|
||||
const bool add)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleEAInteriorFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleEAInteriorFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace
|
||||
@@ -80,8 +74,8 @@ void BilinearFormIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace
|
||||
Vector &ea_data_bdr,
|
||||
const bool add)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleEABoundaryFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleEABoundaryFaces(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleDiagonalPA_ADAt(const Vector &, Vector &)
|
||||
@@ -92,62 +86,62 @@ void BilinearFormIntegrator::AssembleDiagonalPA_ADAt(const Vector &, Vector &)
|
||||
|
||||
void BilinearFormIntegrator::AddMultPA(const Vector &, Vector &) const
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::MultAssembled(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::MultAssembled(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AddMultTransposePA(const Vector &, Vector &) const
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AddMultTransposePA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AddMultTransposePA(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleMF(const FiniteElementSpace &fes)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleMF(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleMF(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AddMultMF(const Vector &, Vector &) const
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AddMultMF(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AddMultMF(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AddMultTransposeMF(const Vector &, Vector &) const
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AddMultTransposeMF(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AddMultTransposeMF(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleDiagonalMF(Vector &)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleDiagonalMF(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleDiagonalMF(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleElementMatrix (
|
||||
const FiniteElement &el, ElementTransformation &Trans,
|
||||
DenseMatrix &elmat )
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleElementMatrix(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleElementMatrix(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleElementMatrix2 (
|
||||
const FiniteElement &el1, const FiniteElement &el2,
|
||||
ElementTransformation &Trans, DenseMatrix &elmat )
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleElementMatrix2(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleElementMatrix2(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleFaceMatrix (
|
||||
const FiniteElement &el1, const FiniteElement &el2,
|
||||
FaceElementTransformations &Trans, DenseMatrix &elmat)
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator::AssembleFaceMatrix(...)\n"
|
||||
" is not implemented for this class.");
|
||||
mfem_error ("BilinearFormIntegrator::AssembleFaceMatrix(...)\n"
|
||||
" is not implemented for this class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleFaceMatrix(
|
||||
@@ -159,16 +153,6 @@ void BilinearFormIntegrator::AssembleFaceMatrix(
|
||||
" Integrator class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleTraceFaceMatrix (int elem,
|
||||
const FiniteElement &trial_face_fe,
|
||||
const FiniteElement &test_fe1,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat)
|
||||
{
|
||||
MFEM_ABORT("AssembleTraceFaceMatrix (DPG form) is not implemented for this"
|
||||
" Integrator class.");
|
||||
}
|
||||
|
||||
void BilinearFormIntegrator::AssembleElementVector(
|
||||
const FiniteElement &el, ElementTransformation &Tr, const Vector &elfun,
|
||||
Vector &elvect)
|
||||
@@ -2649,7 +2633,7 @@ void VectorFEMassIntegrator::AssembleElementMatrix2(
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("VectorFEMassIntegrator::AssembleElementMatrix2(...)\n"
|
||||
mfem_error("VectorFEMassIntegrator::AssembleElementMatrix2(...)\n"
|
||||
" is not implemented for given trial and test bases.");
|
||||
}
|
||||
}
|
||||
@@ -4013,216 +3997,6 @@ void NormalTraceJumpIntegrator::AssembleFaceMatrix(
|
||||
}
|
||||
}
|
||||
|
||||
void TraceIntegrator::AssembleTraceFaceMatrix(int elem,
|
||||
const FiniteElement &trial_face_fe,
|
||||
const FiniteElement &test_fe,
|
||||
FaceElementTransformations & Trans,
|
||||
DenseMatrix &elmat)
|
||||
{
|
||||
MFEM_VERIFY(test_fe.GetMapType() == FiniteElement::VALUE,
|
||||
"TraceIntegrator::AssembleTraceFaceMatrix: Test space should be H1");
|
||||
MFEM_VERIFY(trial_face_fe.GetMapType() == FiniteElement::INTEGRAL,
|
||||
"TraceIntegrator::AssembleTraceFaceMatrix: Trial space should be RT trace");
|
||||
|
||||
int i, j, face_ndof, ndof;
|
||||
int order;
|
||||
|
||||
face_ndof = trial_face_fe.GetDof();
|
||||
ndof = test_fe.GetDof();
|
||||
|
||||
face_shape.SetSize(face_ndof);
|
||||
shape.SetSize(ndof);
|
||||
|
||||
elmat.SetSize(ndof, face_ndof);
|
||||
elmat = 0.0;
|
||||
|
||||
const IntegrationRule *ir = IntRule;
|
||||
if (ir == NULL)
|
||||
{
|
||||
order = test_fe.GetOrder();
|
||||
order += trial_face_fe.GetOrder();
|
||||
ir = &IntRules.Get(Trans.GetGeometryType(), order);
|
||||
}
|
||||
|
||||
int iel = Trans.Elem1->ElementNo;
|
||||
if (iel != elem)
|
||||
{
|
||||
MFEM_VERIFY(elem == Trans.Elem2->ElementNo, "Elem != Trans.Elem2->ElementNo");
|
||||
}
|
||||
|
||||
double scale = 1.0;
|
||||
if (iel != elem) { scale = -1.; }
|
||||
for (int p = 0; p < ir->GetNPoints(); p++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(p);
|
||||
|
||||
// Set the integration point in the face and the neighboring elements
|
||||
Trans.SetAllIntPoints(&ip);
|
||||
// Trace finite element shape function
|
||||
trial_face_fe.CalcPhysShape(Trans,face_shape);
|
||||
|
||||
// Finite element shape function
|
||||
ElementTransformation * eltrans = (iel == elem) ? Trans.Elem1 : Trans.Elem2;
|
||||
test_fe.CalcPhysShape(*eltrans, shape);
|
||||
|
||||
face_shape *= Trans.Weight()*ip.weight*scale;
|
||||
for (i = 0; i < ndof; i++)
|
||||
{
|
||||
for (j = 0; j < face_ndof; j++)
|
||||
{
|
||||
elmat(i, j) += shape(i) * face_shape(j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void NormalTraceIntegrator::AssembleTraceFaceMatrix(int elem,
|
||||
const FiniteElement &trial_face_fe,
|
||||
const FiniteElement &test_fe,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat)
|
||||
{
|
||||
int i, j, face_ndof, ndof, dim;
|
||||
int order;
|
||||
|
||||
MFEM_VERIFY(test_fe.GetMapType() == FiniteElement::H_DIV,
|
||||
"NormalTraceIntegrator::AssembleTraceFaceMatrix: Test space should be RT");
|
||||
MFEM_VERIFY(trial_face_fe.GetMapType() == FiniteElement::VALUE,
|
||||
"NormalTraceIntegrator::AssembleTraceFaceMatrix: Trial space should be H1 (trace)");
|
||||
|
||||
face_ndof = trial_face_fe.GetDof();
|
||||
ndof = test_fe.GetDof();
|
||||
dim = test_fe.GetDim();
|
||||
|
||||
face_shape.SetSize(face_ndof);
|
||||
normal.SetSize(dim);
|
||||
shape.SetSize(ndof,dim);
|
||||
shape_n.SetSize(ndof);
|
||||
|
||||
elmat.SetSize(ndof, face_ndof);
|
||||
elmat = 0.0;
|
||||
|
||||
const IntegrationRule *ir = IntRule;
|
||||
if (ir == NULL)
|
||||
{
|
||||
order = test_fe.GetOrder();
|
||||
order += trial_face_fe.GetOrder();
|
||||
ir = &IntRules.Get(Trans.GetGeometryType(), order);
|
||||
}
|
||||
|
||||
int iel = Trans.Elem1->ElementNo;
|
||||
if (iel != elem)
|
||||
{
|
||||
MFEM_VERIFY(elem == Trans.Elem2->ElementNo, "Elem != Trans.Elem2->ElementNo");
|
||||
}
|
||||
|
||||
double scale = 1.0;
|
||||
if (iel != elem) { scale = -1.; }
|
||||
|
||||
for (int p = 0; p < ir->GetNPoints(); p++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(p);
|
||||
Trans.SetAllIntPoints(&ip);
|
||||
trial_face_fe.CalcPhysShape(Trans, face_shape);
|
||||
CalcOrtho(Trans.Jacobian(),normal);
|
||||
ElementTransformation * etrans = (iel == elem) ? Trans.Elem1 : Trans.Elem2;
|
||||
test_fe.CalcVShape(*etrans, shape);
|
||||
shape.Mult(normal, shape_n);
|
||||
face_shape *= ip.weight*scale;
|
||||
|
||||
for (i = 0; i < ndof; i++)
|
||||
{
|
||||
for (j = 0; j < face_ndof; j++)
|
||||
{
|
||||
elmat(i, j) += shape_n(i) * face_shape(j);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void TangentTraceIntegrator::AssembleTraceFaceMatrix(int elem,
|
||||
const FiniteElement &trial_face_fe,
|
||||
const FiniteElement &test_fe,
|
||||
FaceElementTransformations & Trans,
|
||||
DenseMatrix &elmat)
|
||||
{
|
||||
|
||||
MFEM_VERIFY(test_fe.GetMapType() == FiniteElement::H_CURL,
|
||||
"TangentTraceIntegrator::AssembleTraceFaceMatrix: Test space should be ND");
|
||||
|
||||
int face_ndof, ndof, dim;
|
||||
int order;
|
||||
dim = test_fe.GetDim();
|
||||
if (dim == 3)
|
||||
{
|
||||
std::string msg =
|
||||
"Trial space should be ND face trace and test space should be a ND vector field in 3D ";
|
||||
MFEM_VERIFY(trial_face_fe.GetMapType() == FiniteElement::H_CURL &&
|
||||
trial_face_fe.GetDim() == 2 && test_fe.GetDim() == 3, msg);
|
||||
}
|
||||
else
|
||||
{
|
||||
std::string msg =
|
||||
"Trial space should be H1 edge trace and test space should be a ND vector field in 2D";
|
||||
MFEM_VERIFY(trial_face_fe.GetMapType() == FiniteElement::VALUE &&
|
||||
trial_face_fe.GetDim() == 1 && test_fe.GetDim() == 2, msg);
|
||||
}
|
||||
face_ndof = trial_face_fe.GetDof();
|
||||
ndof = test_fe.GetDof();
|
||||
|
||||
int dimc = (dim == 3) ? 3 : 1;
|
||||
|
||||
face_shape.SetSize(face_ndof,dimc);
|
||||
shape_n.SetSize(ndof,dimc);
|
||||
shape.SetSize(ndof,dim);
|
||||
normal.SetSize(dim);
|
||||
DenseMatrix face_shape_n(face_ndof,dimc);
|
||||
|
||||
elmat.SetSize(ndof, face_ndof);
|
||||
elmat = 0.0;
|
||||
|
||||
const IntegrationRule *ir = IntRule;
|
||||
if (ir == NULL)
|
||||
{
|
||||
order = test_fe.GetOrder();
|
||||
order += trial_face_fe.GetOrder();
|
||||
ir = &IntRules.Get(Trans.GetGeometryType(), order);
|
||||
}
|
||||
|
||||
int iel = Trans.Elem1->ElementNo;
|
||||
if (iel != elem)
|
||||
{
|
||||
MFEM_VERIFY(elem == Trans.Elem2->ElementNo, "Elem != Trans.Elem2->ElementNo");
|
||||
}
|
||||
|
||||
double scale = 1.0;
|
||||
if (iel != elem) { scale = -1.; }
|
||||
for (int p = 0; p < ir->GetNPoints(); p++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir->IntPoint(p);
|
||||
// Set the integration point in the face and the neighboring elements
|
||||
Trans.SetAllIntPoints(&ip);
|
||||
// Trace finite element shape function
|
||||
if (dim == 3)
|
||||
{
|
||||
trial_face_fe.CalcVShape(Trans,face_shape);
|
||||
}
|
||||
else
|
||||
{
|
||||
face_shape.GetColumnReference(0,temp);
|
||||
trial_face_fe.CalcPhysShape(Trans,temp);
|
||||
}
|
||||
CalcOrtho(Trans.Jacobian(),normal);
|
||||
ElementTransformation * eltrans = (iel == elem) ? Trans.Elem1 : Trans.Elem2;
|
||||
test_fe.CalcVShape(*eltrans, shape);
|
||||
|
||||
// rotate
|
||||
cross_product(normal, shape, shape_n);
|
||||
|
||||
const double w = scale*ip.weight;
|
||||
AddMult_a_ABt(w,shape_n, face_shape, elmat);
|
||||
}
|
||||
}
|
||||
|
||||
void NormalInterpolator::AssembleElementMatrix2(
|
||||
const FiniteElement &dom_fe, const FiniteElement &ran_fe,
|
||||
|
||||
+16
-109
@@ -61,8 +61,6 @@ public:
|
||||
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes);
|
||||
|
||||
virtual void AssemblePABoundary(const FiniteElementSpace &fes);
|
||||
|
||||
virtual void AssemblePAInteriorFaces(const FiniteElementSpace &fes);
|
||||
|
||||
virtual void AssemblePABoundaryFaces(const FiniteElementSpace &fes);
|
||||
@@ -161,15 +159,6 @@ public:
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat);
|
||||
|
||||
/** Abstract method used for assembling TraceFaceIntegrators for
|
||||
DPG weak formulations. */
|
||||
virtual void AssembleTraceFaceMatrix(int elem,
|
||||
const FiniteElement &trial_face_fe,
|
||||
const FiniteElement &test_fe,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat);
|
||||
|
||||
|
||||
/// @brief Perform the local action of the BilinearFormIntegrator.
|
||||
/// Note that the default implementation in the base class is general but not
|
||||
/// efficient.
|
||||
@@ -303,12 +292,6 @@ public:
|
||||
bfi->AssemblePA(fes);
|
||||
}
|
||||
|
||||
virtual void AssemblePA(const FiniteElementSpace &trial_fes,
|
||||
const FiniteElementSpace &test_fes)
|
||||
{
|
||||
bfi->AssemblePA(test_fes, trial_fes); // Reverse test and trial
|
||||
}
|
||||
|
||||
virtual void AssemblePAInteriorFaces(const FiniteElementSpace &fes)
|
||||
{
|
||||
bfi->AssemblePAInteriorFaces(fes);
|
||||
@@ -2200,9 +2183,8 @@ protected:
|
||||
// PA extension
|
||||
const FiniteElementSpace *fespace;
|
||||
Vector pa_data;
|
||||
const DofToQuad *maps; ///< Not owned
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
const FaceGeometricFactors *face_geom; ///< Not owned
|
||||
const DofToQuad *maps; ///< Not owned
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
int dim, ne, nq, dofs1D, quad1D;
|
||||
|
||||
public:
|
||||
@@ -2229,8 +2211,6 @@ public:
|
||||
|
||||
virtual void AssemblePA(const FiniteElementSpace &fes);
|
||||
|
||||
virtual void AssemblePABoundary(const FiniteElementSpace &fes);
|
||||
|
||||
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
|
||||
const bool add);
|
||||
|
||||
@@ -3321,87 +3301,6 @@ public:
|
||||
DenseMatrix &elmat);
|
||||
};
|
||||
|
||||
/** Integrator for the DPG form: < v, w > over a face (the interface) where
|
||||
the trial variable v is defined on the interface
|
||||
(H^-1/2 i.e., v:=u⋅n normal trace of H(div))
|
||||
and the test variable w is in an H1-conforming space. */
|
||||
class TraceIntegrator : public BilinearFormIntegrator
|
||||
{
|
||||
private:
|
||||
Vector face_shape, shape;
|
||||
public:
|
||||
TraceIntegrator() { }
|
||||
void AssembleTraceFaceMatrix(int elem,
|
||||
const FiniteElement &trial_face_fe,
|
||||
const FiniteElement &test_fe,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat);
|
||||
};
|
||||
|
||||
/** Integrator for the form: < v, w.n > over a face (the interface) where
|
||||
the trial variable v is defined on the interface (H^1/2, i.e., trace of H1)
|
||||
and the test variable w is in an H(div)-conforming space. */
|
||||
class NormalTraceIntegrator : public BilinearFormIntegrator
|
||||
{
|
||||
private:
|
||||
Vector face_shape, normal, shape_n;
|
||||
DenseMatrix shape;
|
||||
|
||||
public:
|
||||
NormalTraceIntegrator() { }
|
||||
virtual void AssembleTraceFaceMatrix(int ielem,
|
||||
const FiniteElement &trial_face_fe,
|
||||
const FiniteElement &test_fe,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat);
|
||||
};
|
||||
|
||||
|
||||
/** Integrator for the form: < v, w × n > over a face (the interface)
|
||||
* In 3D the trial variable v is defined on the interface (H^-1/2(curl), trace of H(curl))
|
||||
* In 2D it's defined on the interface (H^1/2, trace of H1)
|
||||
* The test variable w is in an H(curl)-conforming space. */
|
||||
class TangentTraceIntegrator : public BilinearFormIntegrator
|
||||
{
|
||||
private:
|
||||
DenseMatrix face_shape, shape, shape_n;
|
||||
Vector normal;
|
||||
Vector temp;
|
||||
|
||||
void cross_product(const Vector & x, const DenseMatrix & Y, DenseMatrix & Z)
|
||||
{
|
||||
int dim = x.Size();
|
||||
MFEM_VERIFY(Y.Width() == dim, "Size missmatch");
|
||||
int dimc = dim == 3 ? dim : 1;
|
||||
int h = Y.Height();
|
||||
Z.SetSize(h,dimc);
|
||||
if (dim == 3)
|
||||
{
|
||||
for (int i = 0; i<h; i++)
|
||||
{
|
||||
Z(i,0) = x(2) * Y(i,1) - x(1) * Y(i,2);
|
||||
Z(i,1) = x(0) * Y(i,2) - x(2) * Y(i,0);
|
||||
Z(i,2) = x(1) * Y(i,0) - x(0) * Y(i,1);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i<h; i++)
|
||||
{
|
||||
Z(i,0) = x(1) * Y(i,0) - x(0) * Y(i,1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
TangentTraceIntegrator() { }
|
||||
void AssembleTraceFaceMatrix(int elem,
|
||||
const FiniteElement &trial_face_fe,
|
||||
const FiniteElement &test_fe,
|
||||
FaceElementTransformations &Trans,
|
||||
DenseMatrix &elmat);
|
||||
};
|
||||
|
||||
/** Abstract class to serve as a base for local interpolators to be used in the
|
||||
DiscreteLinearOperator class. */
|
||||
class DiscreteInterpolator : public BilinearFormIntegrator { };
|
||||
@@ -3437,7 +3336,7 @@ public:
|
||||
|
||||
private:
|
||||
/// 1D finite element that generates and owns the 1D DofToQuad maps below
|
||||
FiniteElement *dofquad_fe;
|
||||
FiniteElement * dofquad_fe;
|
||||
|
||||
bool B_id; // is the B basis operator (maps_C_C) the identity?
|
||||
const DofToQuad *maps_C_C; // one-d map with Lobatto rows, Lobatto columns
|
||||
@@ -3452,8 +3351,6 @@ private:
|
||||
class IdentityInterpolator : public DiscreteInterpolator
|
||||
{
|
||||
public:
|
||||
IdentityInterpolator(): dofquad_fe(NULL) { }
|
||||
|
||||
virtual void AssembleElementMatrix2(const FiniteElement &dom_fe,
|
||||
const FiniteElement &ran_fe,
|
||||
ElementTransformation &Trans,
|
||||
@@ -3468,11 +3365,9 @@ public:
|
||||
virtual void AddMultPA(const Vector &x, Vector &y) const;
|
||||
virtual void AddMultTransposePA(const Vector &x, Vector &y) const;
|
||||
|
||||
virtual ~IdentityInterpolator() { delete dofquad_fe; }
|
||||
|
||||
private:
|
||||
/// 1D finite element that generates and owns the 1D DofToQuad maps below
|
||||
FiniteElement *dofquad_fe;
|
||||
FiniteElement * dofquad_fe;
|
||||
|
||||
const DofToQuad *maps_C_C; // one-d map with Lobatto rows, Lobatto columns
|
||||
const DofToQuad *maps_O_C; // one-d map with Legendre rows, Lobatto columns
|
||||
@@ -3627,5 +3522,17 @@ protected:
|
||||
VectorCoefficient *VQ;
|
||||
};
|
||||
|
||||
|
||||
|
||||
// PA Diffusion Assemble 2D kernel
|
||||
template<const int T_SDIM>
|
||||
void PADiffusionSetup2D(const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
const Vector &c,
|
||||
Vector &d);
|
||||
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -9,8 +9,8 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../pfespace.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "pfespace.hpp"
|
||||
#include <algorithm>
|
||||
|
||||
namespace mfem
|
||||
@@ -9,9 +9,9 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -34,7 +34,7 @@ static void EAConvectionAssemble1D(const int NE,
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -86,7 +86,7 @@ static void EAConvectionAssemble2D(const int NE,
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, 2, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -163,7 +163,7 @@ static void EAConvectionAssemble3D(const int NE,
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 3, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
mfem::forall_3D(NE, D1D, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -9,9 +9,12 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../ceed/integrators/convection/convection.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "ceed/integrators/convection/convection.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -9,15 +9,18 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../ceed/integrators/convection/convection.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "qfunction.hpp"
|
||||
#include "ceed/integrators/convection/convection.hpp"
|
||||
#include "quadinterpolator.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// PA Convection Integrator
|
||||
|
||||
// PA Convection Assemble 2D kernel
|
||||
static void PAConvectionSetup2D(const int NQ,
|
||||
const int NE,
|
||||
@@ -38,7 +41,7 @@ static void PAConvectionSetup2D(const int NQ,
|
||||
Reshape(vel.Read(), DIM,NQ,NE);
|
||||
auto y = Reshape(op.Write(), NQ,DIM,NE);
|
||||
|
||||
mfem::forall(NE*NQ, [=] MFEM_HOST_DEVICE (int q_global)
|
||||
MFEM_FORALL(q_global, NE*NQ,
|
||||
{
|
||||
const int e = q_global / NQ;
|
||||
const int q = q_global % NQ;
|
||||
@@ -75,7 +78,7 @@ static void PAConvectionSetup3D(const int NQ,
|
||||
Reshape(vel.Read(), 3,1,1) :
|
||||
Reshape(vel.Read(), 3,NQ,NE);
|
||||
auto y = Reshape(op.Write(), NQ,3,NE);
|
||||
mfem::forall(NE*NQ, [=] MFEM_HOST_DEVICE (int q_global)
|
||||
MFEM_FORALL(q_global, NE*NQ,
|
||||
{
|
||||
const int e = q_global / NQ;
|
||||
const int q = q_global % NQ;
|
||||
@@ -132,61 +135,6 @@ static void PAConvectionSetup(const int dim,
|
||||
}
|
||||
}
|
||||
|
||||
void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
const MemoryType mt = (pa_mt == MemoryType::DEFAULT) ?
|
||||
Device::GetDeviceMemoryType() : pa_mt;
|
||||
// Assumes tensor-product elements
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetFE(0);
|
||||
ElementTransformation &Trans = *fes.GetElementTransformation(0);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, Trans);
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
|
||||
fes.IsVariableOrder();
|
||||
if (mixed)
|
||||
{
|
||||
ceedOp = new ceed::MixedPAConvectionIntegrator(*this, fes, Q, alpha);
|
||||
}
|
||||
else
|
||||
{
|
||||
ceedOp = new ceed::PAConvectionIntegrator(fes, *ir, Q, alpha);
|
||||
}
|
||||
return;
|
||||
}
|
||||
const int dims = el.GetDim();
|
||||
const int symmDims = dims;
|
||||
nq = ir->GetNPoints();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(symmDims * nq * ne, mt);
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector vel(*Q, qs, CoefficientStorage::COMPRESSED);
|
||||
|
||||
PAConvectionSetup(dim, nq, ne, ir->GetWeights(), geom->J,
|
||||
vel, alpha, pa_data);
|
||||
}
|
||||
|
||||
void ConvectionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("AssembleDiagonalPA not yet implemented for"
|
||||
" ConvectionIntegrator.");
|
||||
}
|
||||
}
|
||||
|
||||
// PA Convection Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0> static
|
||||
void PAConvectionApply2D(const int ne,
|
||||
@@ -211,7 +159,7 @@ void PAConvectionApply2D(const int ne,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, 2, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -331,7 +279,7 @@ void SmemPAConvectionApply2D(const int ne,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, 2, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -458,7 +406,7 @@ void PAConvectionApply3D(const int ne,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, 3, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -639,7 +587,7 @@ void SmemPAConvectionApply3D(const int ne,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, 3, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -843,7 +791,7 @@ void PAConvectionApplyT2D(const int ne,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, 2, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -959,7 +907,7 @@ void SmemPAConvectionApplyT2D(const int ne,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, 2, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -1081,7 +1029,7 @@ void PAConvectionApplyT3D(const int ne,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, 3, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -1257,7 +1205,7 @@ void SmemPAConvectionApplyT3D(const int ne,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, 3, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -1427,6 +1375,48 @@ void SmemPAConvectionApplyT3D(const int ne,
|
||||
});
|
||||
}
|
||||
|
||||
void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
const MemoryType mt = (pa_mt == MemoryType::DEFAULT) ?
|
||||
Device::GetDeviceMemoryType() : pa_mt;
|
||||
// Assumes tensor-product elements
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetFE(0);
|
||||
ElementTransformation &Trans = *fes.GetElementTransformation(0);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, Trans);
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
|
||||
fes.IsVariableOrder();
|
||||
if (mixed)
|
||||
{
|
||||
ceedOp = new ceed::MixedPAConvectionIntegrator(*this, fes, Q, alpha);
|
||||
}
|
||||
else
|
||||
{
|
||||
ceedOp = new ceed::PAConvectionIntegrator(fes, *ir, Q, alpha);
|
||||
}
|
||||
return;
|
||||
}
|
||||
const int dims = el.GetDim();
|
||||
const int symmDims = dims;
|
||||
nq = ir->GetNPoints();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(symmDims * nq * ne, mt);
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector vel(*Q, qs, CoefficientStorage::COMPRESSED);
|
||||
|
||||
PAConvectionSetup(dim, nq, ne, ir->GetWeights(), geom->J,
|
||||
vel, alpha, pa_data);
|
||||
}
|
||||
|
||||
static void PAConvectionApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
@@ -1531,6 +1521,7 @@ static void PAConvectionApplyT(const int dim,
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
|
||||
// PA Convection Apply kernel
|
||||
void ConvectionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
@@ -1545,6 +1536,7 @@ void ConvectionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
// PA Convection Apply transpose kernel
|
||||
void ConvectionIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
@@ -1560,4 +1552,17 @@ void ConvectionIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
void ConvectionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("AssembleDiagonalPA not yet implemented for"
|
||||
" ConvectionIntegrator.");
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -9,9 +9,9 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -26,7 +26,7 @@ static void EADGTraceAssemble1DInt(const int NF,
|
||||
auto D = Reshape(padata.Read(), 2, 2, NF);
|
||||
auto A_int = Reshape(eadata_int.ReadWrite(), 2, NF);
|
||||
auto A_ext = Reshape(eadata_ext.ReadWrite(), 2, NF);
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, NF,
|
||||
{
|
||||
double val_int0, val_int1, val_ext01, val_ext10;
|
||||
val_int0 = D(0, 0, f);
|
||||
@@ -58,7 +58,7 @@ static void EADGTraceAssemble1DBdr(const int NF,
|
||||
{
|
||||
auto D = Reshape(padata.Read(), 2, 2, NF);
|
||||
auto A_bdr = Reshape(eadata_bdr.ReadWrite(), NF);
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, NF,
|
||||
{
|
||||
if (add)
|
||||
{
|
||||
@@ -89,7 +89,7 @@ static void EADGTraceAssemble2DInt(const int NF,
|
||||
auto D = Reshape(padata.Read(), Q1D, 2, 2, NF);
|
||||
auto A_int = Reshape(eadata_int.ReadWrite(), D1D, D1D, 2, NF);
|
||||
auto A_ext = Reshape(eadata_ext.ReadWrite(), D1D, D1D, 2, NF);
|
||||
mfem::forall_2D(NF, D1D, D1D, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL_3D(f, NF, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -143,7 +143,7 @@ static void EADGTraceAssemble2DBdr(const int NF,
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, 2, 2, NF);
|
||||
auto A_bdr = Reshape(eadata_bdr.ReadWrite(), D1D, D1D, NF);
|
||||
mfem::forall_2D(NF, D1D, D1D, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL_3D(f, NF, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -187,7 +187,7 @@ static void EADGTraceAssemble3DInt(const int NF,
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, 2, 2, NF);
|
||||
auto A_int = Reshape(eadata_int.ReadWrite(), D1D, D1D, D1D, D1D, 2, NF);
|
||||
auto A_ext = Reshape(eadata_ext.ReadWrite(), D1D, D1D, D1D, D1D, 2, NF);
|
||||
mfem::forall_2D(NF, D1D, D1D, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL_3D(f, NF, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -283,7 +283,7 @@ static void EADGTraceAssemble3DBdr(const int NF,
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, 2, 2, NF);
|
||||
auto A_bdr = Reshape(eadata_bdr.ReadWrite(), D1D, D1D, D1D, D1D, NF);
|
||||
mfem::forall_2D(NF, D1D, D1D, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL_3D(f, NF, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -9,15 +9,16 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../restriction.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "qfunction.hpp"
|
||||
#include "restriction.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// PA DG Trace Integrator
|
||||
static void PADGTraceSetup2D(const int Q1D,
|
||||
const int NF,
|
||||
@@ -43,7 +44,7 @@ static void PADGTraceSetup2D(const int Q1D,
|
||||
auto W = w.Read();
|
||||
auto qd = Reshape(op.Write(), Q1D, 2, 2, NF);
|
||||
|
||||
mfem::forall(Q1D*NF, [=] MFEM_HOST_DEVICE (int tid)
|
||||
MFEM_FORALL(tid, Q1D*NF,
|
||||
{
|
||||
const int f = tid / Q1D;
|
||||
const int q = tid % Q1D;
|
||||
@@ -86,7 +87,7 @@ static void PADGTraceSetup3D(const int Q1D,
|
||||
auto W = w.Read();
|
||||
auto qd = Reshape(op.Write(), Q1D, Q1D, 2, 2, NF);
|
||||
|
||||
mfem::forall(Q1D*Q1D*NF, [=] MFEM_HOST_DEVICE (int tid)
|
||||
MFEM_FORALL(tid, Q1D*Q1D*NF,
|
||||
{
|
||||
int f = tid / (Q1D * Q1D);
|
||||
int q2 = (tid / Q1D) % Q1D;
|
||||
@@ -98,7 +99,7 @@ static void PADGTraceSetup3D(const int Q1D,
|
||||
const double v1 = const_v ? V(1,0,0,0) : V(1,q1,q2,f);
|
||||
const double v2 = const_v ? V(2,0,0,0) : V(2,q1,q2,f);
|
||||
const double dot = n(q1,q2,0,f) * v0 + n(q1,q2,1,f) * v1 +
|
||||
n(q1,q2,2,f) * v2;
|
||||
/* */ n(q1,q2,2,f) * v2;
|
||||
const double abs = dot > 0.0 ? dot : -dot;
|
||||
const double w = W[q1+q2*Q1D]*r*d(q1,q2,f);
|
||||
qd(q1,q2,0,0,f) = w*( alpha/2 * dot + beta * abs );
|
||||
@@ -266,7 +267,7 @@ void PADGTraceApply2D(const int NF,
|
||||
auto x = Reshape(x_.Read(), D1D, VDIM, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, VDIM, 2, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, NF,
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -357,7 +358,7 @@ void PADGTraceApply3D(const int NF,
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, 2, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, NF,
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -502,7 +503,7 @@ void SmemPADGTraceApply3D(const int NF,
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NF);
|
||||
|
||||
mfem::forall_2D_batch(NF, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL_2D(f, NF, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -667,7 +668,7 @@ void PADGTraceApplyTranspose2D(const int NF,
|
||||
auto x = Reshape(x_.Read(), D1D, VDIM, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, VDIM, 2, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, NF,
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -763,7 +764,7 @@ void PADGTraceApplyTranspose3D(const int NF,
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, 2, NF);
|
||||
|
||||
mfem::forall(NF, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, NF,
|
||||
{
|
||||
const int VDIM = 1;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -919,7 +920,7 @@ void SmemPADGTraceApplyTranspose3D(const int NF,
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, 2, NF);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, 2, NF);
|
||||
|
||||
mfem::forall_2D_batch(NF, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL_2D(f, NF, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -9,9 +9,9 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -33,7 +33,7 @@ static void EADiffusionAssemble1D(const int NE,
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -85,7 +85,7 @@ static void EADiffusionAssemble2D(const int NE,
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, 3, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -162,7 +162,7 @@ static void EADiffusionAssemble3D(const int NE,
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, 6, NE);
|
||||
auto A = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
mfem::forall_3D(NE, D1D, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -9,9 +9,12 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "ceed/integrators/diffusion/diffusion.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -9,42 +9,189 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
|
||||
#define MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "qfunction.hpp"
|
||||
#include "ceed/integrators/diffusion/diffusion.hpp"
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace internal
|
||||
// PA Diffusion Integrator
|
||||
|
||||
// OCCA 2D Assemble kernel
|
||||
#ifdef MFEM_USE_OCCA
|
||||
static void OccaPADiffusionSetup2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &W,
|
||||
const Vector &J,
|
||||
const Vector &C,
|
||||
Vector &op)
|
||||
{
|
||||
occa::properties props;
|
||||
props["defines/D1D"] = D1D;
|
||||
props["defines/Q1D"] = Q1D;
|
||||
const occa::memory o_W = OccaMemoryRead(W.GetMemory(), W.Size());
|
||||
const occa::memory o_J = OccaMemoryRead(J.GetMemory(), J.Size());
|
||||
const occa::memory o_C = OccaMemoryRead(C.GetMemory(), C.Size());
|
||||
occa::memory o_op = OccaMemoryWrite(op.GetMemory(), op.Size());
|
||||
const bool const_c = C.Size() == 1;
|
||||
const occa_id_t id = std::make_pair(D1D,Q1D);
|
||||
static occa_kernel_t OccaDiffSetup2D_ker;
|
||||
if (OccaDiffSetup2D_ker.find(id) == OccaDiffSetup2D_ker.end())
|
||||
{
|
||||
const occa::kernel DiffusionSetup2D =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"DiffusionSetup2D", props);
|
||||
OccaDiffSetup2D_ker.emplace(id, DiffusionSetup2D);
|
||||
}
|
||||
OccaDiffSetup2D_ker.at(id)(NE, o_W, o_J, o_C, o_op, const_c);
|
||||
}
|
||||
|
||||
void PADiffusionSetup(const int dim,
|
||||
const int sdim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &W,
|
||||
const Vector &J,
|
||||
const Vector &C,
|
||||
Vector &D);
|
||||
static void OccaPADiffusionSetup3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &W,
|
||||
const Vector &J,
|
||||
const Vector &C,
|
||||
Vector &op)
|
||||
{
|
||||
occa::properties props;
|
||||
props["defines/D1D"] = D1D;
|
||||
props["defines/Q1D"] = Q1D;
|
||||
const occa::memory o_W = OccaMemoryRead(W.GetMemory(), W.Size());
|
||||
const occa::memory o_J = OccaMemoryRead(J.GetMemory(), J.Size());
|
||||
const occa::memory o_C = OccaMemoryRead(C.GetMemory(), C.Size());
|
||||
occa::memory o_op = OccaMemoryWrite(op.GetMemory(), op.Size());
|
||||
const bool const_c = C.Size() == 1;
|
||||
const occa_id_t id = std::make_pair(D1D,Q1D);
|
||||
static occa_kernel_t OccaDiffSetup3D_ker;
|
||||
if (OccaDiffSetup3D_ker.find(id) == OccaDiffSetup3D_ker.end())
|
||||
{
|
||||
const occa::kernel DiffusionSetup3D =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"DiffusionSetup3D", props);
|
||||
OccaDiffSetup3D_ker.emplace(id, DiffusionSetup3D);
|
||||
}
|
||||
OccaDiffSetup3D_ker.at(id)(NE, o_W, o_J, o_C, o_op, const_c);
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
|
||||
// PA Diffusion Assemble 2D kernel
|
||||
template<int T_SDIM>
|
||||
void PADiffusionSetup2D(const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
const Vector &c,
|
||||
Vector &d);
|
||||
template<>
|
||||
void PADiffusionSetup2D<2>(const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
const Vector &c,
|
||||
Vector &d)
|
||||
{
|
||||
const bool symmetric = (coeffDim != 4);
|
||||
const bool const_c = c.Size() == 1;
|
||||
MFEM_VERIFY(coeffDim < 3 ||
|
||||
!const_c, "Constant matrix coefficient not supported");
|
||||
const auto W = Reshape(w.Read(), Q1D,Q1D);
|
||||
const auto J = Reshape(j.Read(), Q1D,Q1D,2,2,NE);
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1,1,1) :
|
||||
Reshape(c.Read(), coeffDim,Q1D,Q1D,NE);
|
||||
auto D = Reshape(d.Write(), Q1D,Q1D, symmetric ? 3 : 4, NE);
|
||||
MFEM_FORALL_2D(e, NE, Q1D,Q1D,1,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
const double J11 = J(qx,qy,0,0,e);
|
||||
const double J21 = J(qx,qy,1,0,e);
|
||||
const double J12 = J(qx,qy,0,1,e);
|
||||
const double J22 = J(qx,qy,1,1,e);
|
||||
const double w_detJ = W(qx,qy) / ((J11*J22)-(J21*J12));
|
||||
if (coeffDim == 3 || coeffDim == 4) // Matrix coefficient
|
||||
{
|
||||
// First compute entries of R = MJ^{-T}, without det J factor.
|
||||
const double M11 = C(0,qx,qy,e);
|
||||
const double M12 = C(1,qx,qy,e);
|
||||
const double M21 = symmetric ? M12 : C(2,qx,qy,e);
|
||||
const double M22 = symmetric ? C(2,qx,qy,e) : C(3,qx,qy,e);
|
||||
const double R11 = M11*J22 - M12*J12;
|
||||
const double R21 = M21*J22 - M22*J12;
|
||||
const double R12 = -M11*J21 + M12*J11;
|
||||
const double R22 = -M21*J21 + M22*J11;
|
||||
|
||||
// Now set y to J^{-1}R.
|
||||
D(qx,qy,0,e) = w_detJ * ( J22*R11 - J12*R21); // 1,1
|
||||
D(qx,qy,1,e) = w_detJ * (-J21*R11 + J11*R21); // 2,1
|
||||
D(qx,qy,2,e) = w_detJ * (symmetric ? (-J21*R12 + J11*R22) :
|
||||
(J22*R12 - J12*R22)); // 2,2 or 1,2
|
||||
if (!symmetric)
|
||||
{
|
||||
D(qx,qy,3,e) = w_detJ * (-J21*R12 + J11*R22); // 2,2
|
||||
}
|
||||
}
|
||||
else // Vector or scalar coefficient
|
||||
{
|
||||
const double C1 = const_c ? C(0,0,0,0) : C(0,qx,qy,e);
|
||||
const double C2 = const_c ? C(0,0,0,0) :
|
||||
(coeffDim == 2 ? C(1,qx,qy,e) : C(0,qx,qy,e));
|
||||
|
||||
D(qx,qy,0,e) = w_detJ * (C2*J12*J12 + C1*J22*J22); // 1,1
|
||||
D(qx,qy,1,e) = -w_detJ * (C2*J12*J11 + C1*J22*J21); // 1,2
|
||||
D(qx,qy,2,e) = w_detJ * (C2*J11*J11 + C1*J21*J21); // 2,2
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA Diffusion Assemble 2D kernel with 3D node coords
|
||||
template<>
|
||||
void PADiffusionSetup2D<3>(const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
const Vector &c,
|
||||
Vector &d)
|
||||
{
|
||||
MFEM_VERIFY(coeffDim == 1, "Matrix and vector coefficients not supported");
|
||||
constexpr int DIM = 2;
|
||||
constexpr int SDIM = 3;
|
||||
const bool const_c = c.Size() == 1;
|
||||
const auto W = Reshape(w.Read(), Q1D,Q1D);
|
||||
const auto J = Reshape(j.Read(), Q1D,Q1D,SDIM,DIM,NE);
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1,1) :
|
||||
Reshape(c.Read(), Q1D,Q1D,NE);
|
||||
auto D = Reshape(d.Write(), Q1D,Q1D, 3, NE);
|
||||
MFEM_FORALL_2D(e, NE, Q1D,Q1D,1,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
const double wq = W(qx,qy);
|
||||
const double J11 = J(qx,qy,0,0,e);
|
||||
const double J21 = J(qx,qy,1,0,e);
|
||||
const double J31 = J(qx,qy,2,0,e);
|
||||
const double J12 = J(qx,qy,0,1,e);
|
||||
const double J22 = J(qx,qy,1,1,e);
|
||||
const double J32 = J(qx,qy,2,1,e);
|
||||
const double E = J11*J11 + J21*J21 + J31*J31;
|
||||
const double G = J12*J12 + J22*J22 + J32*J32;
|
||||
const double F = J11*J12 + J21*J22 + J31*J32;
|
||||
const double iw = 1.0 / sqrt(E*G - F*F);
|
||||
const double coeff = const_c ? C(0,0,0) : C(qx,qy,e);
|
||||
const double alpha = wq * coeff * iw;
|
||||
D(qx,qy,0,e) = alpha * G; // 1,1
|
||||
D(qx,qy,1,e) = -alpha * F; // 1,2
|
||||
D(qx,qy,2,e) = alpha * E; // 2,2
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA Diffusion Assemble 3D kernel
|
||||
void PADiffusionSetup3D(const int Q1D,
|
||||
@@ -53,41 +200,217 @@ void PADiffusionSetup3D(const int Q1D,
|
||||
const Array<double> &w,
|
||||
const Vector &j,
|
||||
const Vector &c,
|
||||
Vector &d);
|
||||
Vector &d)
|
||||
{
|
||||
const bool symmetric = (coeffDim != 9);
|
||||
const bool const_c = c.Size() == 1;
|
||||
MFEM_VERIFY(coeffDim < 6 ||
|
||||
!const_c, "Constant matrix coefficient not supported");
|
||||
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
|
||||
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1,1,1,1) :
|
||||
Reshape(c.Read(), coeffDim,Q1D,Q1D,Q1D,NE);
|
||||
auto D = Reshape(d.Write(), Q1D,Q1D,Q1D, symmetric ? 6 : 9, NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
const double J11 = J(qx,qy,qz,0,0,e);
|
||||
const double J21 = J(qx,qy,qz,1,0,e);
|
||||
const double J31 = J(qx,qy,qz,2,0,e);
|
||||
const double J12 = J(qx,qy,qz,0,1,e);
|
||||
const double J22 = J(qx,qy,qz,1,1,e);
|
||||
const double J32 = J(qx,qy,qz,2,1,e);
|
||||
const double J13 = J(qx,qy,qz,0,2,e);
|
||||
const double J23 = J(qx,qy,qz,1,2,e);
|
||||
const double J33 = J(qx,qy,qz,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
const double w_detJ = W(qx,qy,qz) / detJ;
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J32 * J13) - (J12 * J33);
|
||||
const double A13 = (J12 * J23) - (J22 * J13);
|
||||
const double A21 = (J31 * J23) - (J21 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J21 * J13) - (J11 * J23);
|
||||
const double A31 = (J21 * J32) - (J31 * J22);
|
||||
const double A32 = (J31 * J12) - (J11 * J32);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
if (coeffDim == 6 || coeffDim == 9) // Matrix coefficient version
|
||||
{
|
||||
// Compute entries of R = MJ^{-T} = M adj(J)^T, without det J.
|
||||
const double M11 = C(0, qx,qy,qz, e);
|
||||
const double M12 = C(1, qx,qy,qz, e);
|
||||
const double M13 = C(2, qx,qy,qz, e);
|
||||
const double M21 = (!symmetric) ? C(3, qx,qy,qz, e) : M12;
|
||||
const double M22 = (!symmetric) ? C(4, qx,qy,qz, e) : C(3, qx,qy,qz, e);
|
||||
const double M23 = (!symmetric) ? C(5, qx,qy,qz, e) : C(4, qx,qy,qz, e);
|
||||
const double M31 = (!symmetric) ? C(6, qx,qy,qz, e) : M13;
|
||||
const double M32 = (!symmetric) ? C(7, qx,qy,qz, e) : M23;
|
||||
const double M33 = (!symmetric) ? C(8, qx,qy,qz, e) : C(5, qx,qy,qz, e);
|
||||
|
||||
const double R11 = M11*A11 + M12*A12 + M13*A13;
|
||||
const double R12 = M11*A21 + M12*A22 + M13*A23;
|
||||
const double R13 = M11*A31 + M12*A32 + M13*A33;
|
||||
const double R21 = M21*A11 + M22*A12 + M23*A13;
|
||||
const double R22 = M21*A21 + M22*A22 + M23*A23;
|
||||
const double R23 = M21*A31 + M22*A32 + M23*A33;
|
||||
const double R31 = M31*A11 + M32*A12 + M33*A13;
|
||||
const double R32 = M31*A21 + M32*A22 + M33*A23;
|
||||
const double R33 = M31*A31 + M32*A32 + M33*A33;
|
||||
|
||||
// Now set D to J^{-1} R = adj(J) R
|
||||
D(qx,qy,qz,0,e) = w_detJ * (A11*R11 + A12*R21 + A13*R31); // 1,1
|
||||
const double D12 = w_detJ * (A11*R12 + A12*R22 + A13*R32);
|
||||
D(qx,qy,qz,1,e) = D12; // 1,2
|
||||
D(qx,qy,qz,2,e) = w_detJ * (A11*R13 + A12*R23 + A13*R33); // 1,3
|
||||
|
||||
const double D22 = w_detJ * (A21*R12 + A22*R22 + A23*R32);
|
||||
const double D23 = w_detJ * (A21*R13 + A22*R23 + A23*R33);
|
||||
|
||||
const double D33 = w_detJ * (A31*R13 + A32*R23 + A33*R33);
|
||||
|
||||
D(qx,qy,qz,4,e) = symmetric ? D23 : D22; // 2,3 or 2,2
|
||||
D(qx,qy,qz,5,e) = symmetric ? D33 : D23; // 3,3 or 2,3
|
||||
|
||||
if (symmetric)
|
||||
{
|
||||
D(qx,qy,qz,3,e) = D22; // 2,2
|
||||
}
|
||||
else
|
||||
{
|
||||
D(qx,qy,qz,3,e) = w_detJ * (A21*R11 + A22*R21 + A23*R31); // 2,1
|
||||
D(qx,qy,qz,6,e) = w_detJ * (A31*R11 + A32*R21 + A33*R31); // 3,1
|
||||
D(qx,qy,qz,7,e) = w_detJ * (A31*R12 + A32*R22 + A33*R32); // 3,2
|
||||
D(qx,qy,qz,8,e) = D33; // 3,3
|
||||
}
|
||||
}
|
||||
else // Vector or scalar coefficient version
|
||||
{
|
||||
const double C1 = const_c ? C(0,0,0,0,0) : C(0,qx,qy,qz,e);
|
||||
const double C2 = const_c ? C(0,0,0,0,0) :
|
||||
(coeffDim == 3 ? C(1,qx,qy,qz,e) : C(0,qx,qy,qz,e));
|
||||
const double C3 = const_c ? C(0,0,0,0,0) :
|
||||
(coeffDim == 3 ? C(2,qx,qy,qz,e) : C(0,qx,qy,qz,e));
|
||||
|
||||
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
|
||||
D(qx,qy,qz,0,e) = w_detJ * (C1*A11*A11 + C2*A12*A12 + C3*A13*A13); // 1,1
|
||||
D(qx,qy,qz,1,e) = w_detJ * (C1*A11*A21 + C2*A12*A22 + C3*A13*A23); // 2,1
|
||||
D(qx,qy,qz,2,e) = w_detJ * (C1*A11*A31 + C2*A12*A32 + C3*A13*A33); // 3,1
|
||||
D(qx,qy,qz,3,e) = w_detJ * (C1*A21*A21 + C2*A22*A22 + C3*A23*A23); // 2,2
|
||||
D(qx,qy,qz,4,e) = w_detJ * (C1*A21*A31 + C2*A22*A32 + C3*A23*A33); // 3,2
|
||||
D(qx,qy,qz,5,e) = w_detJ * (C1*A31*A31 + C2*A32*A32 + C3*A33*A33); // 3,3
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PADiffusionSetup(const int dim,
|
||||
const int sdim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int coeffDim,
|
||||
const int NE,
|
||||
const Array<double> &W,
|
||||
const Vector &J,
|
||||
const Vector &C,
|
||||
Vector &D)
|
||||
{
|
||||
if (dim == 1) { MFEM_ABORT("dim==1 not supported in PADiffusionSetup"); }
|
||||
if (dim == 2)
|
||||
{
|
||||
#ifdef MFEM_USE_OCCA
|
||||
// OCCA 2D Assemble kernel
|
||||
void OccaPADiffusionSetup2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &W,
|
||||
const Vector &J,
|
||||
const Vector &C,
|
||||
Vector &op);
|
||||
|
||||
// OCCA 3D Assemble kernel
|
||||
void OccaPADiffusionSetup3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &W,
|
||||
const Vector &J,
|
||||
const Vector &C,
|
||||
Vector &op);
|
||||
if (DeviceCanUseOcca())
|
||||
{
|
||||
OccaPADiffusionSetup2D(D1D, Q1D, NE, W, J, C, D);
|
||||
return;
|
||||
}
|
||||
#else
|
||||
MFEM_CONTRACT_VAR(D1D);
|
||||
#endif // MFEM_USE_OCCA
|
||||
if (sdim == 2) { PADiffusionSetup2D<2>(Q1D, coeffDim, NE, W, J, C, D); }
|
||||
if (sdim == 3) { PADiffusionSetup2D<3>(Q1D, coeffDim, NE, W, J, C, D); }
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
#ifdef MFEM_USE_OCCA
|
||||
if (DeviceCanUseOcca())
|
||||
{
|
||||
OccaPADiffusionSetup3D(D1D, Q1D, NE, W, J, C, D);
|
||||
return;
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
PADiffusionSetup3D(Q1D, coeffDim, NE, W, J, C, D);
|
||||
}
|
||||
}
|
||||
|
||||
void PADiffusionAssembleDiagonal(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symm,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Vector &D,
|
||||
Vector &Y);
|
||||
void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
const MemoryType mt = (pa_mt == MemoryType::DEFAULT) ?
|
||||
Device::GetDeviceMemoryType() : pa_mt;
|
||||
// Assuming the same element type
|
||||
fespace = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
if (mesh->GetNE() == 0) { return; }
|
||||
const FiniteElement &el = *fes.GetFE(0);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
MFEM_VERIFY(!VQ && !MQ,
|
||||
"Only scalar coefficient supported for DiffusionIntegrator"
|
||||
" with libCEED");
|
||||
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
|
||||
fes.IsVariableOrder();
|
||||
if (mixed)
|
||||
{
|
||||
ceedOp = new ceed::MixedPADiffusionIntegrator(*this, fes, Q);
|
||||
}
|
||||
else
|
||||
{
|
||||
ceedOp = new ceed::PADiffusionIntegrator(fes, *ir, Q);
|
||||
}
|
||||
return;
|
||||
}
|
||||
const int dims = el.GetDim();
|
||||
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
|
||||
const int nq = ir->GetNPoints();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
|
||||
const int sdim = mesh->SpaceDimension();
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector coeff(qs, CoefficientStorage::COMPRESSED);
|
||||
|
||||
if (MQ) { coeff.ProjectTranspose(*MQ); }
|
||||
else if (VQ) { coeff.Project(*VQ); }
|
||||
else if (Q) { coeff.Project(*Q); }
|
||||
else { coeff.SetConstant(1.0); }
|
||||
|
||||
const int coeff_dim = coeff.GetVDim();
|
||||
symmetric = (coeff_dim != dims*dims);
|
||||
const int pa_size = symmetric ? symmDims : dims*dims;
|
||||
|
||||
pa_data.SetSize(pa_size * nq * ne, mt);
|
||||
PADiffusionSetup(dim, sdim, dofs1D, quad1D, coeff_dim, ne, ir->GetWeights(),
|
||||
geom->J, coeff, pa_data);
|
||||
}
|
||||
|
||||
// PA Diffusion Diagonal 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void PADiffusionDiagonal2D(const int NE,
|
||||
static void PADiffusionDiagonal2D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
@@ -106,7 +429,7 @@ inline void PADiffusionDiagonal2D(const int NE,
|
||||
// store necessary entries
|
||||
auto D = Reshape(d.Read(), Q1D*Q1D, symmetric ? 3 : 4, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -153,7 +476,7 @@ inline void PADiffusionDiagonal2D(const int NE,
|
||||
|
||||
// Shared memory PA Diffusion Diagonal 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
|
||||
inline void SmemPADiffusionDiagonal2D(const int NE,
|
||||
static void SmemPADiffusionDiagonal2D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
@@ -173,7 +496,7 @@ inline void SmemPADiffusionDiagonal2D(const int NE,
|
||||
auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d_.Read(), Q1D*Q1D, symmetric ? 3 : 4, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -246,9 +569,8 @@ inline void SmemPADiffusionDiagonal2D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
// PA Diffusion Diagonal 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void PADiffusionDiagonal3D(const int NE,
|
||||
static void PADiffusionDiagonal3D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
@@ -268,7 +590,7 @@ inline void PADiffusionDiagonal3D(const int NE,
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, symmetric ? 6 : 9, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -292,8 +614,8 @@ inline void PADiffusionDiagonal3D(const int NE,
|
||||
{
|
||||
const int q = qx + (qy + qz * Q1D) * Q1D;
|
||||
const int ksym = j >= i ?
|
||||
3 - (3-i)*(2-i)/2 + j:
|
||||
3 - (3-j)*(2-j)/2 + i;
|
||||
3 - (3-i)*(2-i)/2 + j:
|
||||
3 - (3-j)*(2-j)/2 + i;
|
||||
const int k = symmetric ? ksym : (i*DIM) + j;
|
||||
const double O = Q(q,k,e);
|
||||
const double Bz = B(qz,dz);
|
||||
@@ -349,7 +671,7 @@ inline void PADiffusionDiagonal3D(const int NE,
|
||||
|
||||
// Shared memory PA Diffusion Diagonal 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void SmemPADiffusionDiagonal3D(const int NE,
|
||||
static void SmemPADiffusionDiagonal3D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
@@ -369,7 +691,7 @@ inline void SmemPADiffusionDiagonal3D(const int NE,
|
||||
auto g = Reshape(g_.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d_.Read(), Q1D*Q1D*Q1D, symmetric ? 6 : 9, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -466,48 +788,169 @@ inline void SmemPADiffusionDiagonal3D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
void PADiffusionApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symm,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Array<double> &Bt,
|
||||
const Array<double> &Gt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y);
|
||||
static void PADiffusionAssembleDiagonal(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symm,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Vector &D,
|
||||
Vector &Y)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x22: return SmemPADiffusionDiagonal2D<2,2,8>(NE,symm,B,G,D,Y);
|
||||
case 0x33: return SmemPADiffusionDiagonal2D<3,3,8>(NE,symm,B,G,D,Y);
|
||||
case 0x44: return SmemPADiffusionDiagonal2D<4,4,4>(NE,symm,B,G,D,Y);
|
||||
case 0x55: return SmemPADiffusionDiagonal2D<5,5,4>(NE,symm,B,G,D,Y);
|
||||
case 0x66: return SmemPADiffusionDiagonal2D<6,6,2>(NE,symm,B,G,D,Y);
|
||||
case 0x77: return SmemPADiffusionDiagonal2D<7,7,2>(NE,symm,B,G,D,Y);
|
||||
case 0x88: return SmemPADiffusionDiagonal2D<8,8,1>(NE,symm,B,G,D,Y);
|
||||
case 0x99: return SmemPADiffusionDiagonal2D<9,9,1>(NE,symm,B,G,D,Y);
|
||||
default: return PADiffusionDiagonal2D(NE,symm,B,G,D,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x22: return SmemPADiffusionDiagonal3D<2,2>(NE,symm,B,G,D,Y);
|
||||
case 0x23: return SmemPADiffusionDiagonal3D<2,3>(NE,symm,B,G,D,Y);
|
||||
case 0x34: return SmemPADiffusionDiagonal3D<3,4>(NE,symm,B,G,D,Y);
|
||||
case 0x45: return SmemPADiffusionDiagonal3D<4,5>(NE,symm,B,G,D,Y);
|
||||
case 0x46: return SmemPADiffusionDiagonal3D<4,6>(NE,symm,B,G,D,Y);
|
||||
case 0x56: return SmemPADiffusionDiagonal3D<5,6>(NE,symm,B,G,D,Y);
|
||||
case 0x67: return SmemPADiffusionDiagonal3D<6,7>(NE,symm,B,G,D,Y);
|
||||
case 0x78: return SmemPADiffusionDiagonal3D<7,8>(NE,symm,B,G,D,Y);
|
||||
case 0x89: return SmemPADiffusionDiagonal3D<8,9>(NE,symm,B,G,D,Y);
|
||||
case 0x9A: return SmemPADiffusionDiagonal3D<9,10>(NE,symm,B,G,D,Y);
|
||||
default: return PADiffusionDiagonal3D(NE,symm,B,G,D,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
|
||||
void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (pa_data.Size()==0) { AssemblePA(*fespace); }
|
||||
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne, symmetric,
|
||||
maps->B, maps->G, pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#ifdef MFEM_USE_OCCA
|
||||
// OCCA PA Diffusion Apply 2D kernel
|
||||
void OccaPADiffusionApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Array<double> &Bt,
|
||||
const Array<double> &Gt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y);
|
||||
static void OccaPADiffusionApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Array<double> &Bt,
|
||||
const Array<double> &Gt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y)
|
||||
{
|
||||
occa::properties props;
|
||||
props["defines/D1D"] = D1D;
|
||||
props["defines/Q1D"] = Q1D;
|
||||
const occa::memory o_B = OccaMemoryRead(B.GetMemory(), B.Size());
|
||||
const occa::memory o_G = OccaMemoryRead(G.GetMemory(), G.Size());
|
||||
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
|
||||
const occa::memory o_Gt = OccaMemoryRead(Gt.GetMemory(), Gt.Size());
|
||||
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
|
||||
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
|
||||
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
|
||||
const occa_id_t id = std::make_pair(D1D,Q1D);
|
||||
if (!Device::Allows(Backend::OCCA_CUDA))
|
||||
{
|
||||
static occa_kernel_t OccaDiffApply2D_cpu;
|
||||
if (OccaDiffApply2D_cpu.find(id) == OccaDiffApply2D_cpu.end())
|
||||
{
|
||||
const occa::kernel DiffusionApply2D_CPU =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"DiffusionApply2D_CPU", props);
|
||||
OccaDiffApply2D_cpu.emplace(id, DiffusionApply2D_CPU);
|
||||
}
|
||||
OccaDiffApply2D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
|
||||
}
|
||||
else
|
||||
{
|
||||
static occa_kernel_t OccaDiffApply2D_gpu;
|
||||
if (OccaDiffApply2D_gpu.find(id) == OccaDiffApply2D_gpu.end())
|
||||
{
|
||||
const occa::kernel DiffusionApply2D_GPU =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"DiffusionApply2D_GPU", props);
|
||||
OccaDiffApply2D_gpu.emplace(id, DiffusionApply2D_GPU);
|
||||
}
|
||||
OccaDiffApply2D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
|
||||
}
|
||||
}
|
||||
|
||||
// OCCA PA Diffusion Apply 3D kernel
|
||||
void OccaPADiffusionApply3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Array<double> &Bt,
|
||||
const Array<double> &Gt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y);
|
||||
static void OccaPADiffusionApply3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Array<double> &Bt,
|
||||
const Array<double> &Gt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y)
|
||||
{
|
||||
occa::properties props;
|
||||
props["defines/D1D"] = D1D;
|
||||
props["defines/Q1D"] = Q1D;
|
||||
const occa::memory o_B = OccaMemoryRead(B.GetMemory(), B.Size());
|
||||
const occa::memory o_G = OccaMemoryRead(G.GetMemory(), G.Size());
|
||||
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
|
||||
const occa::memory o_Gt = OccaMemoryRead(Gt.GetMemory(), Gt.Size());
|
||||
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
|
||||
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
|
||||
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
|
||||
const occa_id_t id = std::make_pair(D1D,Q1D);
|
||||
if (!Device::Allows(Backend::OCCA_CUDA))
|
||||
{
|
||||
static occa_kernel_t OccaDiffApply3D_cpu;
|
||||
if (OccaDiffApply3D_cpu.find(id) == OccaDiffApply3D_cpu.end())
|
||||
{
|
||||
const occa::kernel DiffusionApply3D_CPU =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"DiffusionApply3D_CPU", props);
|
||||
OccaDiffApply3D_cpu.emplace(id, DiffusionApply3D_CPU);
|
||||
}
|
||||
OccaDiffApply3D_cpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
|
||||
}
|
||||
else
|
||||
{
|
||||
static occa_kernel_t OccaDiffApply3D_gpu;
|
||||
if (OccaDiffApply3D_gpu.find(id) == OccaDiffApply3D_gpu.end())
|
||||
{
|
||||
const occa::kernel DiffusionApply3D_GPU =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"DiffusionApply3D_GPU", props);
|
||||
OccaDiffApply3D_gpu.emplace(id, DiffusionApply3D_GPU);
|
||||
}
|
||||
OccaDiffApply3D_gpu.at(id)(NE, o_B, o_G, o_Bt, o_Gt, o_D, o_X, o_Y);
|
||||
}
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
|
||||
// PA Diffusion Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void PADiffusionApply2D(const int NE,
|
||||
static void PADiffusionApply2D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
@@ -530,7 +973,7 @@ inline void PADiffusionApply2D(const int NE,
|
||||
auto D = Reshape(d_.Read(), Q1D*Q1D, symmetric ? 3 : 4, NE);
|
||||
auto X = Reshape(x_.Read(), D1D, D1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -629,7 +1072,7 @@ inline void PADiffusionApply2D(const int NE,
|
||||
|
||||
// Shared memory PA Diffusion Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
|
||||
inline void SmemPADiffusionApply2D(const int NE,
|
||||
static void SmemPADiffusionApply2D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
@@ -651,7 +1094,7 @@ inline void SmemPADiffusionApply2D(const int NE,
|
||||
auto D = Reshape(d_.Read(), Q1D*Q1D, symmetric ? 3 : 4, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE(int e)
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
@@ -787,7 +1230,7 @@ inline void SmemPADiffusionApply2D(const int NE,
|
||||
|
||||
// PA Diffusion Apply 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void PADiffusionApply3D(const int NE,
|
||||
static void PADiffusionApply3D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
@@ -809,7 +1252,7 @@ inline void PADiffusionApply3D(const int NE,
|
||||
auto D = Reshape(d_.Read(), Q1D*Q1D*Q1D, symmetric ? 6 : 9, NE);
|
||||
auto X = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -978,9 +1421,8 @@ inline void PADiffusionApply3D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
// Shared memory PA Diffusion Apply 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void SmemPADiffusionApply3D(const int NE,
|
||||
static void SmemPADiffusionApply3D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &g_,
|
||||
@@ -1001,7 +1443,7 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -1201,8 +1643,99 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
static void PADiffusionApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const bool symm,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Array<double> &Bt,
|
||||
const Array<double> &Gt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y)
|
||||
{
|
||||
#ifdef MFEM_USE_OCCA
|
||||
if (DeviceCanUseOcca())
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
OccaPADiffusionApply2D(D1D,Q1D,NE,B,G,Bt,Gt,D,X,Y);
|
||||
return;
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
OccaPADiffusionApply3D(D1D,Q1D,NE,B,G,Bt,Gt,D,X,Y);
|
||||
return;
|
||||
}
|
||||
MFEM_ABORT("OCCA PADiffusionApply unknown kernel!");
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
const int id = (D1D << 4) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x22: return SmemPADiffusionApply2D<2,2,16>(NE,symm,B,G,D,X,Y);
|
||||
case 0x33: return SmemPADiffusionApply2D<3,3,16>(NE,symm,B,G,D,X,Y);
|
||||
case 0x44: return SmemPADiffusionApply2D<4,4,8>(NE,symm,B,G,D,X,Y);
|
||||
case 0x55: return SmemPADiffusionApply2D<5,5,8>(NE,symm,B,G,D,X,Y);
|
||||
case 0x66: return SmemPADiffusionApply2D<6,6,4>(NE,symm,B,G,D,X,Y);
|
||||
case 0x77: return SmemPADiffusionApply2D<7,7,4>(NE,symm,B,G,D,X,Y);
|
||||
case 0x88: return SmemPADiffusionApply2D<8,8,2>(NE,symm,B,G,D,X,Y);
|
||||
case 0x99: return SmemPADiffusionApply2D<9,9,2>(NE,symm,B,G,D,X,Y);
|
||||
default: return PADiffusionApply2D(NE,symm,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
|
||||
if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x22: return SmemPADiffusionApply3D<2,2>(NE,symm,B,G,D,X,Y);
|
||||
case 0x23: return SmemPADiffusionApply3D<2,3>(NE,symm,B,G,D,X,Y);
|
||||
case 0x34: return SmemPADiffusionApply3D<3,4>(NE,symm,B,G,D,X,Y);
|
||||
case 0x45: return SmemPADiffusionApply3D<4,5>(NE,symm,B,G,D,X,Y);
|
||||
case 0x46: return SmemPADiffusionApply3D<4,6>(NE,symm,B,G,D,X,Y);
|
||||
case 0x56: return SmemPADiffusionApply3D<5,6>(NE,symm,B,G,D,X,Y);
|
||||
case 0x58: return SmemPADiffusionApply3D<5,8>(NE,symm,B,G,D,X,Y);
|
||||
case 0x67: return SmemPADiffusionApply3D<6,7>(NE,symm,B,G,D,X,Y);
|
||||
case 0x78: return SmemPADiffusionApply3D<7,8>(NE,symm,B,G,D,X,Y);
|
||||
case 0x89: return SmemPADiffusionApply3D<8,9>(NE,symm,B,G,D,X,Y);
|
||||
default: return PADiffusionApply3D(NE,symm,B,G,Bt,Gt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("Unknown kernel: 0x"<<std::hex << id << std::dec);
|
||||
}
|
||||
|
||||
// PA Diffusion Apply kernel
|
||||
void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->AddMult(x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
PADiffusionApply(dim, dofs1D, quad1D, ne, symmetric,
|
||||
maps->B, maps->G, maps->Bt, maps->Gt,
|
||||
pa_data, x, y);
|
||||
}
|
||||
}
|
||||
|
||||
void DiffusionIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (symmetric)
|
||||
{
|
||||
AddMultPA(x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("DiffusionIntegrator::AddMultTransposePA only implemented in "
|
||||
"the symmetric case.")
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
@@ -9,13 +9,17 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// PA Divergence Integrator
|
||||
|
||||
// PA Divergence Assemble 2D kernel
|
||||
static void PADivergenceSetup2D(const int Q1D,
|
||||
const int NE,
|
||||
@@ -29,7 +33,7 @@ static void PADivergenceSetup2D(const int Q1D,
|
||||
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
|
||||
auto y = Reshape(op.Write(), NQ, 2, 2, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -58,7 +62,7 @@ static void PADivergenceSetup3D(const int Q1D,
|
||||
auto W = w.Read();
|
||||
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
|
||||
auto y = Reshape(op.Write(), NQ, 3, 3, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -179,7 +183,7 @@ static void PADivergenceApply2D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D*Q1D, 2,2, NE);
|
||||
auto x = Reshape(x_.Read(), TR_D1D, TR_D1D, 2, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
|
||||
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
|
||||
@@ -317,7 +321,7 @@ static void PADivergenceApplyTranspose2D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D*Q1D, 2,2, NE);
|
||||
auto x = Reshape(x_.Read(), TE_D1D, TE_D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), TR_D1D, TR_D1D, 2, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
|
||||
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
|
||||
@@ -433,7 +437,7 @@ static void PADivergenceApply3D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D*Q1D*Q1D, 3,3, NE);
|
||||
auto x = Reshape(x_.Read(), TR_D1D, TR_D1D, TR_D1D, 3, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
|
||||
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
|
||||
@@ -616,7 +620,7 @@ static void PADivergenceApplyTranspose3D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D*Q1D*Q1D, 3,3, NE);
|
||||
auto x = Reshape(x_.Read(), TE_D1D, TE_D1D, TE_D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), TR_D1D, TR_D1D, TR_D1D, 3, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
|
||||
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
|
||||
@@ -797,7 +801,7 @@ static void SmemPADivergenceApply3D(const int NE,
|
||||
auto x = Reshape(x_.Read(), TR_D1D, TR_D1D, TR_D1D, 3, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, NE);
|
||||
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
constexpr int VDIM = 3;
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
@@ -1036,25 +1040,11 @@ static void PADivergenceApply(const int dim,
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
if (transpose)
|
||||
{
|
||||
return PADivergenceApplyTranspose2D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
|
||||
}
|
||||
else
|
||||
{
|
||||
return PADivergenceApply2D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
|
||||
}
|
||||
return PADivergenceApply2D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
if (transpose)
|
||||
{
|
||||
return PADivergenceApplyTranspose3D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
|
||||
}
|
||||
else
|
||||
{
|
||||
return PADivergenceApply3D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
|
||||
}
|
||||
return PADivergenceApply3D(NE,B,G,Bt,op,x,y,TR_D1D,TE_D1D,Q1D);
|
||||
}
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
@@ -9,14 +9,18 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "qfunction.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// PA Gradient Integrator
|
||||
|
||||
/* Description of the *SetupND functions
|
||||
Inputs are as follows
|
||||
\b Q1D number of quadrature points in one dimension.
|
||||
@@ -58,8 +62,8 @@ namespace mfem
|
||||
The shared memory (Smem) versions of the kernels differ from the regular
|
||||
versions in the following properties.
|
||||
|
||||
\b mfem::forall is using only one level of parallelism.
|
||||
\b mfem::forall_ND uses an additional level of parallelism
|
||||
\b MFEM_FORALL is using only one level of parallelism.
|
||||
\b MFEM_FORALL_ND uses an additional level of parallelism
|
||||
\b MFEM_FOREACH_THREAD
|
||||
|
||||
These macros allow automatic mapping of manually defined blocks to
|
||||
@@ -83,7 +87,7 @@ static void PAGradientSetup2D(const int Q1D,
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1) :
|
||||
Reshape(c.Read(), NQ, NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -118,7 +122,7 @@ static void PAGradientSetup3D(const int Q1D,
|
||||
const auto C = const_c ? Reshape(c.Read(), 1,1) :
|
||||
Reshape(c.Read(), NQ,NE);
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -238,7 +242,7 @@ static void PAGradientApply2D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D*Q1D, 2,2, NE);
|
||||
auto x = Reshape(x_.Read(), TR_D1D, TR_D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, 2, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
|
||||
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
|
||||
@@ -368,7 +372,7 @@ static void PAGradientApply3D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D*Q1D*Q1D, 3,3, NE);
|
||||
auto x = Reshape(x_.Read(), TR_D1D, TR_D1D, TR_D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, 3, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int TR_D1D = T_TR_D1D ? T_TR_D1D : tr_d1d;
|
||||
const int TE_D1D = T_TE_D1D ? T_TE_D1D : te_d1d;
|
||||
@@ -568,8 +572,7 @@ static void SmemPAGradientApply3D(const int NE,
|
||||
auto x = Reshape(x_.Read(), TR_D1D, TR_D1D, TR_D1D, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), TE_D1D, TE_D1D, TE_D1D, 3, NE);
|
||||
|
||||
mfem::forall_3D(NE, (Q1D>8)?8:Q1D, (Q1D>8)?8:Q1D, (Q1D>8)?8:Q1D,
|
||||
[=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, (Q1D>8)?8:Q1D, (Q1D>8)?8:Q1D, (Q1D>8)?8:Q1D,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1DR = T_TR_D1D ? T_TR_D1D : tr_d1d;
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -9,9 +9,9 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -32,7 +32,7 @@ static void EAMassAssemble1D(const int NE,
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -82,7 +82,7 @@ static void EAMassAssemble2D(const int NE,
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, NE);
|
||||
mfem::forall_2D(NE, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, 1,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -154,51 +154,20 @@ static void EAMassAssemble3D(const int NE,
|
||||
auto B = Reshape(basis.Read(), Q1D, D1D);
|
||||
auto D = Reshape(padata.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto M = Reshape(eadata.ReadWrite(), D1D, D1D, D1D, D1D, D1D, D1D, NE);
|
||||
mfem::forall_3D(NE, D1D, D1D, D1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_3D(e, NE, D1D, D1D, D1D,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int DQ = T_D1D * T_Q1D;
|
||||
|
||||
// For quadratic and lower it's better to use registers but for higher-order you start to
|
||||
// spill and it's better to use shared memory
|
||||
constexpr bool USE_REG = DQ != 0 && DQ <= 12;
|
||||
constexpr int MD1r = USE_REG ? MD1 : 1;
|
||||
constexpr int MQ1r = USE_REG ? MQ1 : 1;
|
||||
constexpr int MD1s = USE_REG ? 1 : MD1;
|
||||
constexpr int MQ1s = USE_REG ? 1 : MQ1;
|
||||
|
||||
MFEM_SHARED double s_B[MQ1s][MD1s];
|
||||
double r_B[MQ1r][MD1r];
|
||||
double (*l_B)[MD1] = nullptr;
|
||||
if (USE_REG)
|
||||
double r_B[MQ1][MD1];
|
||||
for (int d = 0; d < D1D; d++)
|
||||
{
|
||||
for (int d = 0; d < D1D; d++)
|
||||
for (int q = 0; q < Q1D; q++)
|
||||
{
|
||||
for (int q = 0; q < Q1D; q++)
|
||||
{
|
||||
r_B[q][d] = B(q,d);
|
||||
}
|
||||
r_B[q][d] = B(q,d);
|
||||
}
|
||||
l_B = (double (*)[MD1])r_B;
|
||||
}
|
||||
else
|
||||
{
|
||||
if (MFEM_THREAD_ID(z) == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,x,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,y,Q1D)
|
||||
{
|
||||
s_B[q][d] = B(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
l_B = (double (*)[MD1])s_B;
|
||||
}
|
||||
|
||||
MFEM_SHARED double s_D[MQ1][MQ1][MQ1];
|
||||
MFEM_FOREACH_THREAD(k1,x,Q1D)
|
||||
{
|
||||
@@ -230,9 +199,9 @@ static void EAMassAssemble3D(const int NE,
|
||||
{
|
||||
for (int k3 = 0; k3 < Q1D; ++k3)
|
||||
{
|
||||
val += l_B[k1][i1] * l_B[k1][j1]
|
||||
* l_B[k2][i2] * l_B[k2][j2]
|
||||
* l_B[k3][i3] * l_B[k3][j3]
|
||||
val += r_B[k1][i1] * r_B[k1][j1]
|
||||
* r_B[k2][i2] * r_B[k2][j2]
|
||||
* r_B[k3][i3] * r_B[k3][j3]
|
||||
* s_D[k1][k2][k3];
|
||||
}
|
||||
}
|
||||
@@ -9,9 +9,12 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../ceed/integrators/mass/mass.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "ceed/integrators/mass/mass.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -0,0 +1,736 @@
|
||||
// Copyright (c) 2010-2023, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "qfunction.hpp"
|
||||
#include "ceed/integrators/mass/mass.hpp"
|
||||
#include "bilininteg_mass_pa.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// PA Mass Integrator
|
||||
|
||||
// PA Mass Assemble kernel
|
||||
|
||||
void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
const MemoryType mt = (pa_mt == MemoryType::DEFAULT) ?
|
||||
Device::GetDeviceMemoryType() : pa_mt;
|
||||
|
||||
// Assuming the same element type
|
||||
fespace = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
if (mesh->GetNE() == 0) { return; }
|
||||
const FiniteElement &el = *fes.GetFE(0);
|
||||
ElementTransformation *T0 = mesh->GetElementTransformation(0);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T0);
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
|
||||
fes.IsVariableOrder();
|
||||
if (mixed)
|
||||
{
|
||||
ceedOp = new ceed::MixedPAMassIntegrator(*this, fes, Q);
|
||||
}
|
||||
else
|
||||
{
|
||||
ceedOp = new ceed::PAMassIntegrator(fes, *ir, Q);
|
||||
}
|
||||
return;
|
||||
}
|
||||
int map_type = el.GetMapType();
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetMesh()->GetNE();
|
||||
nq = ir->GetNPoints();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::DETERMINANTS, mt);
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
pa_data.SetSize(ne*nq, mt);
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
|
||||
|
||||
if (dim==1) { MFEM_ABORT("Not supported yet... stay tuned!"); }
|
||||
if (dim==2)
|
||||
{
|
||||
const int NE = ne;
|
||||
const int Q1D = quad1D;
|
||||
const bool const_c = coeff.Size() == 1;
|
||||
const bool by_val = map_type == FiniteElement::VALUE;
|
||||
const auto W = Reshape(ir->GetWeights().Read(), Q1D,Q1D);
|
||||
const auto J = Reshape(geom->detJ.Read(), Q1D,Q1D,NE);
|
||||
const auto C = const_c ? Reshape(coeff.Read(), 1,1,1) :
|
||||
Reshape(coeff.Read(), Q1D,Q1D,NE);
|
||||
auto v = Reshape(pa_data.Write(), Q1D,Q1D, NE);
|
||||
MFEM_FORALL_2D(e, NE, Q1D,Q1D,1,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
const double detJ = J(qx,qy,e);
|
||||
const double coeff = const_c ? C(0,0,0) : C(qx,qy,e);
|
||||
v(qx,qy,e) = W(qx,qy) * coeff * (by_val ? detJ : 1.0/detJ);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
if (dim==3)
|
||||
{
|
||||
const int NE = ne;
|
||||
const int Q1D = quad1D;
|
||||
const bool const_c = coeff.Size() == 1;
|
||||
const bool by_val = map_type == FiniteElement::VALUE;
|
||||
const auto W = Reshape(ir->GetWeights().Read(), Q1D,Q1D,Q1D);
|
||||
const auto J = Reshape(geom->detJ.Read(), Q1D,Q1D,Q1D,NE);
|
||||
const auto C = const_c ? Reshape(coeff.Read(), 1,1,1,1) :
|
||||
Reshape(coeff.Read(), Q1D,Q1D,Q1D,NE);
|
||||
auto v = Reshape(pa_data.Write(), Q1D,Q1D,Q1D,NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz,z,Q1D)
|
||||
{
|
||||
const double detJ = J(qx,qy,qz,e);
|
||||
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
|
||||
v(qx,qy,qz,e) = W(qx,qy,qz) * coeff * (by_val ? detJ : 1.0/detJ);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAMassAssembleDiagonal2D(const int NE,
|
||||
const Array<double> &b,
|
||||
const Vector &d,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d.Read(), Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
double QD[MQ1][MD1];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
QD[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
QD[qx][dy] += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
Y(dx,dy,e) += B(qx, dx) * B(qx, dx) * QD[qx][dy];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
|
||||
static void SmemPAMassAssembleDiagonal2D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Vector &d_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d_.Read(), Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_SHARED double B[MQ1][MD1];
|
||||
MFEM_SHARED double QDZ[NBZ][MQ1][MD1];
|
||||
double (*QD)[MD1] = (double (*)[MD1])(QDZ + tidz);
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B[q][d] = b(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QD[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
QD[qx][dy] += B[qy][dy] * B[qy][dy] * D(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
// might need absolute values on next line
|
||||
Y(dx,dy,e) += B[qx][dx] * B[qx][dx] * QD[qx][dy];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAMassAssembleDiagonal3D(const int NE,
|
||||
const Array<double> &b,
|
||||
const Vector &d,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
double QQD[MQ1][MQ1][MD1];
|
||||
double QDD[MQ1][MD1][MD1];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
QQD[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
QQD[qx][qy][dz] += B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
QDD[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
QDD[qx][dy][dz] += B(qy, dy) * B(qy, dy) * QQD[qx][qy][dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double t = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
t += B(qx, dx) * B(qx, dx) * QDD[qx][dy][dz];
|
||||
}
|
||||
Y(dx, dy, dz, e) += t;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void SmemPAMassAssembleDiagonal3D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Vector &d_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d_.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
MFEM_SHARED double B[MQ1][MD1];
|
||||
MFEM_SHARED double QQD[MQ1][MQ1][MD1];
|
||||
MFEM_SHARED double QDD[MQ1][MD1][MD1];
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B[q][d] = b(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
QQD[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
QQD[qx][qy][dz] += B[qz][dz] * B[qz][dz] * D(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QDD[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
QDD[qx][dy][dz] += B[qy][dy] * B[qy][dy] * QQD[qx][qy][dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double t = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
t += B[qx][dx] * B[qx][dx] * QDD[qx][dy][dz];
|
||||
}
|
||||
Y(dx, dy, dz, e) += t;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PAMassAssembleDiagonal(const int dim, const int D1D,
|
||||
const int Q1D, const int NE,
|
||||
const Array<double> &B,
|
||||
const Vector &D,
|
||||
Vector &Y)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x22: return SmemPAMassAssembleDiagonal2D<2,2,16>(NE,B,D,Y);
|
||||
case 0x33: return SmemPAMassAssembleDiagonal2D<3,3,16>(NE,B,D,Y);
|
||||
case 0x44: return SmemPAMassAssembleDiagonal2D<4,4,8>(NE,B,D,Y);
|
||||
case 0x55: return SmemPAMassAssembleDiagonal2D<5,5,8>(NE,B,D,Y);
|
||||
case 0x66: return SmemPAMassAssembleDiagonal2D<6,6,4>(NE,B,D,Y);
|
||||
case 0x77: return SmemPAMassAssembleDiagonal2D<7,7,4>(NE,B,D,Y);
|
||||
case 0x88: return SmemPAMassAssembleDiagonal2D<8,8,2>(NE,B,D,Y);
|
||||
case 0x99: return SmemPAMassAssembleDiagonal2D<9,9,2>(NE,B,D,Y);
|
||||
default: return PAMassAssembleDiagonal2D(NE,B,D,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
switch ((D1D << 4 ) | Q1D)
|
||||
{
|
||||
case 0x23: return SmemPAMassAssembleDiagonal3D<2,3>(NE,B,D,Y);
|
||||
case 0x24: return SmemPAMassAssembleDiagonal3D<2,4>(NE,B,D,Y);
|
||||
case 0x26: return SmemPAMassAssembleDiagonal3D<2,6>(NE,B,D,Y);
|
||||
case 0x34: return SmemPAMassAssembleDiagonal3D<3,4>(NE,B,D,Y);
|
||||
case 0x35: return SmemPAMassAssembleDiagonal3D<3,5>(NE,B,D,Y);
|
||||
case 0x45: return SmemPAMassAssembleDiagonal3D<4,5>(NE,B,D,Y);
|
||||
case 0x48: return SmemPAMassAssembleDiagonal3D<4,8>(NE,B,D,Y);
|
||||
case 0x56: return SmemPAMassAssembleDiagonal3D<5,6>(NE,B,D,Y);
|
||||
case 0x67: return SmemPAMassAssembleDiagonal3D<6,7>(NE,B,D,Y);
|
||||
case 0x78: return SmemPAMassAssembleDiagonal3D<7,8>(NE,B,D,Y);
|
||||
case 0x89: return SmemPAMassAssembleDiagonal3D<8,9>(NE,B,D,Y);
|
||||
default: return PAMassAssembleDiagonal3D(NE,B,D,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
|
||||
void MassIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#ifdef MFEM_USE_OCCA
|
||||
// OCCA PA Mass Apply 2D kernel
|
||||
static void OccaPAMassApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &Bt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y)
|
||||
{
|
||||
occa::properties props;
|
||||
props["defines/D1D"] = D1D;
|
||||
props["defines/Q1D"] = Q1D;
|
||||
const occa::memory o_B = OccaMemoryRead(B.GetMemory(), B.Size());
|
||||
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
|
||||
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
|
||||
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
|
||||
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
|
||||
const occa_id_t id = std::make_pair(D1D,Q1D);
|
||||
if (!Device::Allows(Backend::OCCA_CUDA))
|
||||
{
|
||||
static occa_kernel_t OccaMassApply2D_cpu;
|
||||
if (OccaMassApply2D_cpu.find(id) == OccaMassApply2D_cpu.end())
|
||||
{
|
||||
const occa::kernel MassApply2D_CPU =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"MassApply2D_CPU", props);
|
||||
OccaMassApply2D_cpu.emplace(id, MassApply2D_CPU);
|
||||
}
|
||||
OccaMassApply2D_cpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
|
||||
}
|
||||
else
|
||||
{
|
||||
static occa_kernel_t OccaMassApply2D_gpu;
|
||||
if (OccaMassApply2D_gpu.find(id) == OccaMassApply2D_gpu.end())
|
||||
{
|
||||
const occa::kernel MassApply2D_GPU =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"MassApply2D_GPU", props);
|
||||
OccaMassApply2D_gpu.emplace(id, MassApply2D_GPU);
|
||||
}
|
||||
OccaMassApply2D_gpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
|
||||
}
|
||||
}
|
||||
|
||||
// OCCA PA Mass Apply 3D kernel
|
||||
static void OccaPAMassApply3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &Bt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y)
|
||||
{
|
||||
occa::properties props;
|
||||
props["defines/D1D"] = D1D;
|
||||
props["defines/Q1D"] = Q1D;
|
||||
const occa::memory o_B = OccaMemoryRead(B.GetMemory(), B.Size());
|
||||
const occa::memory o_Bt = OccaMemoryRead(Bt.GetMemory(), Bt.Size());
|
||||
const occa::memory o_D = OccaMemoryRead(D.GetMemory(), D.Size());
|
||||
const occa::memory o_X = OccaMemoryRead(X.GetMemory(), X.Size());
|
||||
occa::memory o_Y = OccaMemoryReadWrite(Y.GetMemory(), Y.Size());
|
||||
const occa_id_t id = std::make_pair(D1D,Q1D);
|
||||
if (!Device::Allows(Backend::OCCA_CUDA))
|
||||
{
|
||||
static occa_kernel_t OccaMassApply3D_cpu;
|
||||
if (OccaMassApply3D_cpu.find(id) == OccaMassApply3D_cpu.end())
|
||||
{
|
||||
const occa::kernel MassApply3D_CPU =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"MassApply3D_CPU", props);
|
||||
OccaMassApply3D_cpu.emplace(id, MassApply3D_CPU);
|
||||
}
|
||||
OccaMassApply3D_cpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
|
||||
}
|
||||
else
|
||||
{
|
||||
static occa_kernel_t OccaMassApply3D_gpu;
|
||||
if (OccaMassApply3D_gpu.find(id) == OccaMassApply3D_gpu.end())
|
||||
{
|
||||
const occa::kernel MassApply3D_GPU =
|
||||
mfem::OccaDev().buildKernel("occa://mfem/fem/occa.okl",
|
||||
"MassApply3D_GPU", props);
|
||||
OccaMassApply3D_gpu.emplace(id, MassApply3D_GPU);
|
||||
}
|
||||
OccaMassApply3D_gpu.at(id)(NE, o_B, o_Bt, o_D, o_X, o_Y);
|
||||
}
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAMassApply2D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bt_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_VERIFY(T_D1D ? T_D1D : d1d <= MAX_D1D, "");
|
||||
MFEM_VERIFY(T_Q1D ? T_Q1D : q1d <= MAX_Q1D, "");
|
||||
|
||||
const auto B = b_.Read();
|
||||
const auto Bt = bt_.Read();
|
||||
const auto D = d_.Read();
|
||||
const auto X = x_.Read();
|
||||
auto Y = y_.ReadWrite();
|
||||
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
internal::PAMassApply2D_Element(e, NE, B, Bt, D, X, Y, d1d, q1d);
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
|
||||
static void SmemPAMassApply2D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bt_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
const auto b = b_.Read();
|
||||
const auto D = d_.Read();
|
||||
const auto x = x_.Read();
|
||||
auto Y = y_.ReadWrite();
|
||||
MFEM_FORALL_2D(e, NE, Q1D, Q1D, NBZ,
|
||||
{
|
||||
internal::SmemPAMassApply2D_Element<T_D1D,T_Q1D,T_NBZ>(e, NE, b, D, x, Y, d1d, q1d);
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAMassApply3D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bt_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_VERIFY(T_D1D ? T_D1D : d1d <= MAX_D1D, "");
|
||||
MFEM_VERIFY(T_Q1D ? T_Q1D : q1d <= MAX_Q1D, "");
|
||||
|
||||
const auto B = b_.Read();
|
||||
const auto Bt = bt_.Read();
|
||||
const auto D = d_.Read();
|
||||
const auto X = x_.Read();
|
||||
auto Y = y_.ReadWrite();
|
||||
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
internal::PAMassApply3D_Element(e, NE, B, Bt, D, X, Y, d1d, q1d);
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void SmemPAMassApply3D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bt_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int M1D = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= M1D, "");
|
||||
MFEM_VERIFY(Q1D <= M1Q, "");
|
||||
auto b = b_.Read();
|
||||
auto d = d_.Read();
|
||||
auto x = x_.Read();
|
||||
auto y = y_.ReadWrite();
|
||||
MFEM_FORALL_3D(e, NE, Q1D, Q1D, 1,
|
||||
{
|
||||
internal::SmemPAMassApply3D_Element<T_D1D,T_Q1D>(e, NE, b, d, x, y, d1d, q1d);
|
||||
});
|
||||
}
|
||||
|
||||
static void PAMassApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &Bt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y)
|
||||
{
|
||||
#ifdef MFEM_USE_OCCA
|
||||
if (DeviceCanUseOcca())
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return OccaPAMassApply2D(D1D,Q1D,NE,B,Bt,D,X,Y);
|
||||
}
|
||||
if (dim == 3)
|
||||
{
|
||||
return OccaPAMassApply3D(D1D,Q1D,NE,B,Bt,D,X,Y);
|
||||
}
|
||||
MFEM_ABORT("OCCA PA Mass Apply unknown kernel!");
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
const int id = (D1D << 4) | Q1D;
|
||||
|
||||
if (dim == 2)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x22: return SmemPAMassApply2D<2,2,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x24: return SmemPAMassApply2D<2,4,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x33: return SmemPAMassApply2D<3,3,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x34: return SmemPAMassApply2D<3,4,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x35: return SmemPAMassApply2D<3,5,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x36: return SmemPAMassApply2D<3,6,16>(NE,B,Bt,D,X,Y);
|
||||
case 0x44: return SmemPAMassApply2D<4,4,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x46: return SmemPAMassApply2D<4,6,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x48: return SmemPAMassApply2D<4,8,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x55: return SmemPAMassApply2D<5,5,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x57: return SmemPAMassApply2D<5,7,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x58: return SmemPAMassApply2D<5,8,2>(NE,B,Bt,D,X,Y);
|
||||
case 0x66: return SmemPAMassApply2D<6,6,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x77: return SmemPAMassApply2D<7,7,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x88: return SmemPAMassApply2D<8,8,2>(NE,B,Bt,D,X,Y);
|
||||
case 0x99: return SmemPAMassApply2D<9,9,2>(NE,B,Bt,D,X,Y);
|
||||
default: return PAMassApply2D(NE,B,Bt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
switch (id)
|
||||
{
|
||||
case 0x22: return SmemPAMassApply3D<2,2>(NE,B,Bt,D,X,Y);
|
||||
case 0x23: return SmemPAMassApply3D<2,3>(NE,B,Bt,D,X,Y);
|
||||
case 0x24: return SmemPAMassApply3D<2,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x26: return SmemPAMassApply3D<2,6>(NE,B,Bt,D,X,Y);
|
||||
case 0x34: return SmemPAMassApply3D<3,4>(NE,B,Bt,D,X,Y);
|
||||
case 0x35: return SmemPAMassApply3D<3,5>(NE,B,Bt,D,X,Y);
|
||||
case 0x36: return SmemPAMassApply3D<3,6>(NE,B,Bt,D,X,Y);
|
||||
case 0x37: return SmemPAMassApply3D<3,7>(NE,B,Bt,D,X,Y);
|
||||
case 0x45: return SmemPAMassApply3D<4,5>(NE,B,Bt,D,X,Y);
|
||||
case 0x46: return SmemPAMassApply3D<4,6>(NE,B,Bt,D,X,Y);
|
||||
case 0x48: return SmemPAMassApply3D<4,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x56: return SmemPAMassApply3D<5,6>(NE,B,Bt,D,X,Y);
|
||||
case 0x58: return SmemPAMassApply3D<5,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x67: return SmemPAMassApply3D<6,7>(NE,B,Bt,D,X,Y);
|
||||
case 0x78: return SmemPAMassApply3D<7,8>(NE,B,Bt,D,X,Y);
|
||||
case 0x89: return SmemPAMassApply3D<8,9>(NE,B,Bt,D,X,Y);
|
||||
case 0x9A: return SmemPAMassApply3D<9,10>(NE,B,Bt,D,X,Y);
|
||||
default: return PAMassApply3D(NE,B,Bt,D,X,Y,D1D,Q1D);
|
||||
}
|
||||
}
|
||||
mfem::out << "Unknown kernel 0x" << std::hex << id << std::endl;
|
||||
MFEM_ABORT("Unknown kernel.");
|
||||
}
|
||||
|
||||
void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->AddMult(x, y);
|
||||
}
|
||||
else
|
||||
{
|
||||
PAMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
|
||||
}
|
||||
}
|
||||
|
||||
void MassIntegrator::AddMultTransposePA(const Vector &x, Vector &y) const
|
||||
{
|
||||
// Mass integrator is symmetric
|
||||
AddMultPA(x, y);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -9,15 +9,12 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_BILININTEG_MASS_KERNELS_HPP
|
||||
#define MFEM_BILININTEG_MASS_KERNELS_HPP
|
||||
#ifndef MFEM_BILININTEG_MASS_PA_HPP
|
||||
#define MFEM_BILININTEG_MASS_PA_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../../linalg/dtensor.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../config/config.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "../linalg/dtensor.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -25,315 +22,6 @@ namespace mfem
|
||||
namespace internal
|
||||
{
|
||||
|
||||
void PAMassAssembleDiagonal(const int dim, const int D1D,
|
||||
const int Q1D, const int NE,
|
||||
const Array<double> &B,
|
||||
const Vector &D,
|
||||
Vector &Y);
|
||||
|
||||
// PA Mass Diagonal 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void PAMassAssembleDiagonal2D(const int NE,
|
||||
const Array<double> &b,
|
||||
const Vector &d,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d.Read(), Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
double QD[MQ1][MD1];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
QD[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
QD[qx][dy] += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
Y(dx,dy,e) += B(qx, dx) * B(qx, dx) * QD[qx][dy];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Shared memory PA Mass Diagonal 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
|
||||
inline void SmemPAMassAssembleDiagonal2D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Vector &d_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d_.Read(), Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, NE);
|
||||
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_SHARED double B[MQ1][MD1];
|
||||
MFEM_SHARED double QDZ[NBZ][MQ1][MD1];
|
||||
double (*QD)[MD1] = (double (*)[MD1])(QDZ + tidz);
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B[q][d] = b(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QD[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
QD[qx][dy] += B[qy][dy] * B[qy][dy] * D(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
// might need absolute values on next line
|
||||
Y(dx,dy,e) += B[qx][dx] * B[qx][dx] * QD[qx][dy];
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// PA Mass Diagonal 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void PAMassAssembleDiagonal3D(const int NE,
|
||||
const Array<double> &b,
|
||||
const Vector &d,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
double QQD[MQ1][MQ1][MD1];
|
||||
double QDD[MQ1][MD1][MD1];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
QQD[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
QQD[qx][qy][dz] += B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
QDD[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
QDD[qx][dy][dz] += B(qy, dy) * B(qy, dy) * QQD[qx][qy][dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double t = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
t += B(qx, dx) * B(qx, dx) * QDD[qx][dy][dz];
|
||||
}
|
||||
Y(dx, dy, dz, e) += t;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Shared memory PA Mass Diagonal 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void SmemPAMassAssembleDiagonal3D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Vector &d_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
auto b = Reshape(b_.Read(), Q1D, D1D);
|
||||
auto D = Reshape(d_.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto Y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
|
||||
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int tidz = MFEM_THREAD_ID(z);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
MFEM_SHARED double B[MQ1][MD1];
|
||||
MFEM_SHARED double QQD[MQ1][MQ1][MD1];
|
||||
MFEM_SHARED double QDD[MQ1][MD1][MD1];
|
||||
if (tidz == 0)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(d,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(q,x,Q1D)
|
||||
{
|
||||
B[q][d] = b(q,d);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy,y,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
QQD[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
QQD[qx][qy][dz] += B[qz][dz] * B[qz][dz] * D(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(qx,x,Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
QDD[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
QDD[qx][dy][dz] += B[qy][dy] * B[qy][dy] * QQD[qx][qy][dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(dz,z,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy,y,D1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx,x,D1D)
|
||||
{
|
||||
double t = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
t += B[qx][dx] * B[qx][dx] * QDD[qx][dy][dz];
|
||||
}
|
||||
Y(dx, dy, dz, e) += t;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void PAMassApply(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &Bt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y);
|
||||
|
||||
#ifdef MFEM_USE_OCCA
|
||||
// OCCA PA Mass Apply 2D kernel
|
||||
void OccaPAMassApply2D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &Bt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y);
|
||||
|
||||
// OCCA PA Mass Apply 3D kernel
|
||||
void OccaPAMassApply3D(const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &Bt,
|
||||
const Vector &D,
|
||||
const Vector &X,
|
||||
Vector &Y);
|
||||
#endif // MFEM_USE_OCCA
|
||||
|
||||
template <bool ACCUMULATE = true>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void PAMassApply2D_Element(const int e,
|
||||
@@ -937,116 +625,6 @@ void SmemPAMassApply3D_Element(const int e,
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
// PA Mass Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void PAMassApply2D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bt_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_VERIFY(T_D1D ? T_D1D : d1d <= MAX_D1D, "");
|
||||
MFEM_VERIFY(T_Q1D ? T_Q1D : q1d <= MAX_Q1D, "");
|
||||
|
||||
const auto B = b_.Read();
|
||||
const auto Bt = bt_.Read();
|
||||
const auto D = d_.Read();
|
||||
const auto X = x_.Read();
|
||||
auto Y = y_.ReadWrite();
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
internal::PAMassApply2D_Element(e, NE, B, Bt, D, X, Y, d1d, q1d);
|
||||
});
|
||||
}
|
||||
|
||||
// Shared memory PA Mass Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_NBZ = 0>
|
||||
inline void SmemPAMassApply2D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bt_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int NBZ = T_NBZ ? T_NBZ : 1;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
const auto b = b_.Read();
|
||||
const auto D = d_.Read();
|
||||
const auto x = x_.Read();
|
||||
auto Y = y_.ReadWrite();
|
||||
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
internal::SmemPAMassApply2D_Element<T_D1D,T_Q1D,T_NBZ>(e, NE, b, D, x, Y, d1d,
|
||||
q1d);
|
||||
});
|
||||
}
|
||||
|
||||
// PA Mass Apply 3D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void PAMassApply3D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bt_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_VERIFY(T_D1D ? T_D1D : d1d <= MAX_D1D, "");
|
||||
MFEM_VERIFY(T_Q1D ? T_Q1D : q1d <= MAX_Q1D, "");
|
||||
|
||||
const auto B = b_.Read();
|
||||
const auto Bt = bt_.Read();
|
||||
const auto D = d_.Read();
|
||||
const auto X = x_.Read();
|
||||
auto Y = y_.ReadWrite();
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
internal::PAMassApply3D_Element(e, NE, B, Bt, D, X, Y, d1d, q1d);
|
||||
});
|
||||
}
|
||||
|
||||
// Shared memory PA Mass Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
inline void SmemPAMassApply3D(const int NE,
|
||||
const Array<double> &b_,
|
||||
const Array<double> &bt_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
MFEM_CONTRACT_VAR(bt_);
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int M1Q = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int M1D = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= M1D, "");
|
||||
MFEM_VERIFY(Q1D <= M1Q, "");
|
||||
auto b = b_.Read();
|
||||
auto d = d_.Read();
|
||||
auto x = x_.Read();
|
||||
auto y = y_.ReadWrite();
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
internal::SmemPAMassApply3D_Element<T_D1D,T_Q1D>(e, NE, b, d, x, y, d1d, q1d);
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
|
||||
} // namespace mfem
|
||||
@@ -9,8 +9,8 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -27,7 +27,7 @@ void TransposeIntegrator::AssembleEA(const FiniteElementSpace &fes,
|
||||
const int dofs = fes.GetFE(0)->GetDof();
|
||||
auto A = Reshape(ea_data_tmp.Read(), dofs, dofs, ne);
|
||||
auto AT = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
|
||||
mfem::forall(ne, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int i = 0; i < dofs; i++)
|
||||
{
|
||||
@@ -46,7 +46,7 @@ void TransposeIntegrator::AssembleEA(const FiniteElementSpace &fes,
|
||||
if (ne == 0) { return; }
|
||||
const int dofs = fes.GetFE(0)->GetDof();
|
||||
auto A = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
|
||||
mfem::forall(ne, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int i = 0; i < dofs; i++)
|
||||
{
|
||||
@@ -80,7 +80,7 @@ void TransposeIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace& fes,
|
||||
auto A_ext = Reshape(ea_data_ext_tmp.Read(), faceDofs, faceDofs, 2, nf);
|
||||
auto AT_int = Reshape(ea_data_int.ReadWrite(), faceDofs, faceDofs, 2, nf);
|
||||
auto AT_ext = Reshape(ea_data_ext.ReadWrite(), faceDofs, faceDofs, 2, nf);
|
||||
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, nf,
|
||||
{
|
||||
for (int i = 0; i < faceDofs; i++)
|
||||
{
|
||||
@@ -105,7 +105,7 @@ void TransposeIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace& fes,
|
||||
fes.GetMesh()->GetFaceGeometry(0))->GetDof();
|
||||
auto A_int = Reshape(ea_data_int.ReadWrite(), faceDofs, faceDofs, 2, nf);
|
||||
auto A_ext = Reshape(ea_data_ext.ReadWrite(), faceDofs, faceDofs, 2, nf);
|
||||
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, nf,
|
||||
{
|
||||
for (int i = 0; i < faceDofs; i++)
|
||||
{
|
||||
@@ -149,7 +149,7 @@ void TransposeIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace& fes,
|
||||
fes.GetMesh()->GetFaceGeometry(0))->GetDof();
|
||||
auto A_bdr = Reshape(ea_data_bdr_tmp.Read(), faceDofs, faceDofs, nf);
|
||||
auto AT_bdr = Reshape(ea_data_bdr.ReadWrite(), faceDofs, faceDofs, nf);
|
||||
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, nf,
|
||||
{
|
||||
for (int i = 0; i < faceDofs; i++)
|
||||
{
|
||||
@@ -167,7 +167,7 @@ void TransposeIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace& fes,
|
||||
const int faceDofs = fes.GetTraceElement(0,
|
||||
fes.GetMesh()->GetFaceGeometry(0))->GetDof();
|
||||
auto A_bdr = Reshape(ea_data_bdr.ReadWrite(), faceDofs, faceDofs, nf);
|
||||
mfem::forall(nf, [=] MFEM_HOST_DEVICE (int f)
|
||||
MFEM_FORALL(f, nf,
|
||||
{
|
||||
for (int i = 0; i < faceDofs; i++)
|
||||
{
|
||||
@@ -9,15 +9,19 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../qfunction.hpp"
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "qfunction.hpp"
|
||||
#include "ceed/integrators/diffusion/diffusion.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// PA Vector Diffusion Integrator
|
||||
|
||||
// PA Diffusion Assemble 2D kernel
|
||||
static void PAVectorDiffusionSetup2D(const int Q1D,
|
||||
const int NE,
|
||||
@@ -37,7 +41,7 @@ static void PAVectorDiffusionSetup2D(const int Q1D,
|
||||
Reshape(c.Read(), NQ, NE);
|
||||
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -73,7 +77,7 @@ static void PAVectorDiffusionSetup3D(const int Q1D,
|
||||
Reshape(c.Read(), NQ,NE);
|
||||
|
||||
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -87,8 +91,8 @@ static void PAVectorDiffusionSetup3D(const int Q1D,
|
||||
const double J23 = J(q,1,2,e);
|
||||
const double J33 = J(q,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
J21 * (J12 * J33 - J32 * J13) +
|
||||
J31 * (J12 * J23 - J22 * J13);
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
|
||||
const double C1 = const_c ? C(0,0) : C(q,e);
|
||||
|
||||
@@ -193,7 +197,7 @@ void VectorDiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
const auto C = const_c ? Reshape(coeff.Read(), 1,1) :
|
||||
Reshape(coeff.Read(), NQ,ne);
|
||||
|
||||
mfem::forall(ne, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, ne,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -222,209 +226,6 @@ void VectorDiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
}
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAVectorDiffusionDiagonal2D(const int NE,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
const Vector &d,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
// note the different shape for D, this is a (symmetric) matrix so we only
|
||||
// store necessary entries
|
||||
auto D = Reshape(d.Read(), Q1D*Q1D, 3, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, 2, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
// gradphi \cdot Q \gradphi has four terms
|
||||
double QD0[MQ1][MD1];
|
||||
double QD1[MQ1][MD1];
|
||||
double QD2[MQ1][MD1];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
QD0[qx][dy] = 0.0;
|
||||
QD1[qx][dy] = 0.0;
|
||||
QD2[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const int q = qx + qy * Q1D;
|
||||
const double D0 = D(q,0,e);
|
||||
const double D1 = D(q,1,e);
|
||||
const double D2 = D(q,2,e);
|
||||
QD0[qx][dy] += B(qy, dy) * B(qy, dy) * D0;
|
||||
QD1[qx][dy] += B(qy, dy) * G(qy, dy) * D1;
|
||||
QD2[qx][dy] += G(qy, dy) * G(qy, dy) * D2;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double temp = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
temp += G(qx, dx) * G(qx, dx) * QD0[qx][dy];
|
||||
temp += G(qx, dx) * B(qx, dx) * QD1[qx][dy];
|
||||
temp += B(qx, dx) * G(qx, dx) * QD1[qx][dy];
|
||||
temp += B(qx, dx) * B(qx, dx) * QD2[qx][dy];
|
||||
}
|
||||
Y(dx,dy,0,e) += temp;
|
||||
Y(dx,dy,1,e) += temp;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAVectorDiffusionDiagonal3D(const int NE,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
const Vector &d,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 6, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, 3, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
double QQD[MQ1][MQ1][MD1];
|
||||
double QDD[MQ1][MD1][MD1];
|
||||
for (int i = 0; i < DIM; ++i)
|
||||
{
|
||||
for (int j = 0; j < DIM; ++j)
|
||||
{
|
||||
// first tensor contraction, along z direction
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
QQD[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const int q = qx + (qy + qz * Q1D) * Q1D;
|
||||
const int k = j >= i ?
|
||||
3 - (3-i)*(2-i)/2 + j:
|
||||
3 - (3-j)*(2-j)/2 + i;
|
||||
const double O = Q(q,k,e);
|
||||
const double Bz = B(qz,dz);
|
||||
const double Gz = G(qz,dz);
|
||||
const double L = i==2 ? Gz : Bz;
|
||||
const double R = j==2 ? Gz : Bz;
|
||||
QQD[qx][qy][dz] += L * O * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// second tensor contraction, along y direction
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
QDD[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double By = B(qy,dy);
|
||||
const double Gy = G(qy,dy);
|
||||
const double L = i==1 ? Gy : By;
|
||||
const double R = j==1 ? Gy : By;
|
||||
QDD[qx][dy][dz] += L * QQD[qx][qy][dz] * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// third tensor contraction, along x direction
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double temp = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double Bx = B(qx,dx);
|
||||
const double Gx = G(qx,dx);
|
||||
const double L = i==0 ? Gx : Bx;
|
||||
const double R = j==0 ? Gx : Bx;
|
||||
temp += L * QDD[qx][dy][dz] * R;
|
||||
}
|
||||
Y(dx, dy, dz, 0, e) += temp;
|
||||
Y(dx, dy, dz, 1, e) += temp;
|
||||
Y(dx, dy, dz, 2, e) += temp;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PAVectorDiffusionAssembleDiagonal(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Vector &op,
|
||||
Vector &y)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return PAVectorDiffusionDiagonal2D(NE, B, G, op, y, D1D, Q1D);
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
return PAVectorDiffusionDiagonal3D(NE, B, G, op, y, D1D, Q1D);
|
||||
}
|
||||
MFEM_ABORT("Dimension not implemented.");
|
||||
}
|
||||
|
||||
void VectorDiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
PAVectorDiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
|
||||
maps->B, maps->G,
|
||||
pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
// PA Diffusion Apply 2D kernel
|
||||
template<int T_D1D = 0, int T_Q1D = 0, int T_VDIM = 0> static
|
||||
void PAVectorDiffusionApply2D(const int NE,
|
||||
@@ -451,7 +252,7 @@ void PAVectorDiffusionApply2D(const int NE,
|
||||
auto D = Reshape(d_.Read(), Q1D*Q1D, 3, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -572,7 +373,7 @@ void PAVectorDiffusionApply3D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D*Q1D*Q1D, 6, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -780,4 +581,212 @@ void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAVectorDiffusionDiagonal2D(const int NE,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
const Vector &d,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
// note the different shape for D, this is a (symmetric) matrix so we only
|
||||
// store necessary entries
|
||||
auto D = Reshape(d.Read(), Q1D*Q1D, 3, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, 2, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
// gradphi \cdot Q \gradphi has four terms
|
||||
double QD0[MQ1][MD1];
|
||||
double QD1[MQ1][MD1];
|
||||
double QD2[MQ1][MD1];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
QD0[qx][dy] = 0.0;
|
||||
QD1[qx][dy] = 0.0;
|
||||
QD2[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const int q = qx + qy * Q1D;
|
||||
const double D0 = D(q,0,e);
|
||||
const double D1 = D(q,1,e);
|
||||
const double D2 = D(q,2,e);
|
||||
QD0[qx][dy] += B(qy, dy) * B(qy, dy) * D0;
|
||||
QD1[qx][dy] += B(qy, dy) * G(qy, dy) * D1;
|
||||
QD2[qx][dy] += G(qy, dy) * G(qy, dy) * D2;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double temp = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
temp += G(qx, dx) * G(qx, dx) * QD0[qx][dy];
|
||||
temp += G(qx, dx) * B(qx, dx) * QD1[qx][dy];
|
||||
temp += B(qx, dx) * G(qx, dx) * QD1[qx][dy];
|
||||
temp += B(qx, dx) * B(qx, dx) * QD2[qx][dy];
|
||||
}
|
||||
Y(dx,dy,0,e) += temp;
|
||||
Y(dx,dy,1,e) += temp;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_D1D = 0, int T_Q1D = 0>
|
||||
static void PAVectorDiffusionDiagonal3D(const int NE,
|
||||
const Array<double> &b,
|
||||
const Array<double> &g,
|
||||
const Vector &d,
|
||||
Vector &y,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
constexpr int DIM = 3;
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
MFEM_VERIFY(D1D <= MD1, "");
|
||||
MFEM_VERIFY(Q1D <= MQ1, "");
|
||||
auto B = Reshape(b.Read(), Q1D, D1D);
|
||||
auto G = Reshape(g.Read(), Q1D, D1D);
|
||||
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 6, NE);
|
||||
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, 3, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int MD1 = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int MQ1 = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
double QQD[MQ1][MQ1][MD1];
|
||||
double QDD[MQ1][MD1][MD1];
|
||||
for (int i = 0; i < DIM; ++i)
|
||||
{
|
||||
for (int j = 0; j < DIM; ++j)
|
||||
{
|
||||
// first tensor contraction, along z direction
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
QQD[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
const int q = qx + (qy + qz * Q1D) * Q1D;
|
||||
const int k = j >= i ?
|
||||
3 - (3-i)*(2-i)/2 + j:
|
||||
3 - (3-j)*(2-j)/2 + i;
|
||||
const double O = Q(q,k,e);
|
||||
const double Bz = B(qz,dz);
|
||||
const double Gz = G(qz,dz);
|
||||
const double L = i==2 ? Gz : Bz;
|
||||
const double R = j==2 ? Gz : Bz;
|
||||
QQD[qx][qy][dz] += L * O * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// second tensor contraction, along y direction
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
QDD[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
const double By = B(qy,dy);
|
||||
const double Gy = G(qy,dy);
|
||||
const double L = i==1 ? Gy : By;
|
||||
const double R = j==1 ? Gy : By;
|
||||
QDD[qx][dy][dz] += L * QQD[qx][qy][dz] * R;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// third tensor contraction, along x direction
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double temp = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
const double Bx = B(qx,dx);
|
||||
const double Gx = G(qx,dx);
|
||||
const double L = i==0 ? Gx : Bx;
|
||||
const double R = j==0 ? Gx : Bx;
|
||||
temp += L * QDD[qx][dy][dz] * R;
|
||||
}
|
||||
Y(dx, dy, dz, 0, e) += temp;
|
||||
Y(dx, dy, dz, 1, e) += temp;
|
||||
Y(dx, dy, dz, 2, e) += temp;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PAVectorDiffusionAssembleDiagonal(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &G,
|
||||
const Vector &op,
|
||||
Vector &y)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return PAVectorDiffusionDiagonal2D(NE, B, G, op, y, D1D, Q1D);
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
return PAVectorDiffusionDiagonal3D(NE, B, G, op, y, D1D, Q1D);
|
||||
}
|
||||
MFEM_ABORT("Dimension not implemented.");
|
||||
}
|
||||
|
||||
void VectorDiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
PAVectorDiffusionAssembleDiagonal(dim,
|
||||
dofs1D,
|
||||
quad1D,
|
||||
ne,
|
||||
maps->B,
|
||||
maps->G,
|
||||
pa_data,
|
||||
diag);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -9,9 +9,12 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "ceed/integrators/diffusion/diffusion.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -9,14 +9,19 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../../general/forall.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../ceed/integrators/mass/mass.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "ceed/integrators/mass/mass.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// PA Mass Integrator
|
||||
|
||||
// PA Mass Assemble kernel
|
||||
void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
{
|
||||
// Assuming the same element type
|
||||
@@ -69,7 +74,7 @@ void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
auto w = ir->GetWeights().Read();
|
||||
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
|
||||
auto v = Reshape(pa_data.Write(), NQ, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -90,7 +95,7 @@ void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
auto W = ir->GetWeights().Read();
|
||||
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
|
||||
auto v = Reshape(pa_data.Write(), NQ,NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
for (int q = 0; q < NQ; ++q)
|
||||
{
|
||||
@@ -98,177 +103,16 @@ void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
const double J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
|
||||
const double J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
|
||||
const double detJ = J11 * (J22 * J33 - J32 * J23) -
|
||||
J21 * (J12 * J33 - J32 * J13) +
|
||||
J31 * (J12 * J23 - J22 * J13);
|
||||
/* */ J21 * (J12 * J33 - J32 * J13) +
|
||||
/* */ J31 * (J12 * J23 - J22 * J13);
|
||||
v(q,e) = W[q] * constant * detJ;
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassAssembleDiagonal2D(const int NE,
|
||||
const Array<double> &B_,
|
||||
const Array<double> &Bt_,
|
||||
const Vector &op_,
|
||||
Vector &diag_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 2;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
|
||||
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
double temp[max_Q1D][max_D1D];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
temp[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
temp[qx][dy] += B(qy, dy) * B(qy, dy) * op(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double temp1 = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
temp1 += B(qx, dx) * B(qx, dx) * temp[qx][dy];
|
||||
}
|
||||
y(dx, dy, 0, e) = temp1;
|
||||
y(dx, dy, 1, e) = temp1;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassAssembleDiagonal3D(const int NE,
|
||||
const Array<double> &B_,
|
||||
const Array<double> &Bt_,
|
||||
const Vector &op_,
|
||||
Vector &diag_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 3;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
double temp[max_Q1D][max_Q1D][max_D1D];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
temp[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
temp[qx][qy][dz] += B(qz, dz) * B(qz, dz) * op(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
double temp2[max_Q1D][max_D1D][max_D1D];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
temp2[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
temp2[qx][dy][dz] += B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double temp3 = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
temp3 += B(qx, dx) * B(qx, dx)
|
||||
* temp2[qx][dy][dz];
|
||||
}
|
||||
y(dx, dy, dz, 0, e) = temp3;
|
||||
y(dx, dy, dz, 1, e) = temp3;
|
||||
y(dx, dy, dz, 2, e) = temp3;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PAVectorMassAssembleDiagonal(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &Bt,
|
||||
const Vector &op,
|
||||
Vector &y)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return PAVectorMassAssembleDiagonal2D(NE, B, Bt, op, y, D1D, Q1D);
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
return PAVectorMassAssembleDiagonal3D(NE, B, Bt, op, y, D1D, Q1D);
|
||||
}
|
||||
MFEM_ABORT("Dimension not implemented.");
|
||||
}
|
||||
|
||||
void VectorMassIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
PAVectorMassAssembleDiagonal(dim, dofs1D, quad1D, ne,
|
||||
maps->B, maps->Bt,
|
||||
pa_data, diag);
|
||||
}
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
template<const int T_D1D = 0,
|
||||
const int T_Q1D = 0>
|
||||
static void PAVectorMassApply2D(const int NE,
|
||||
const Array<double> &B_,
|
||||
const Array<double> &Bt_,
|
||||
@@ -288,7 +132,7 @@ static void PAVectorMassApply2D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -364,7 +208,8 @@ static void PAVectorMassApply2D(const int NE,
|
||||
});
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
template<const int T_D1D = 0,
|
||||
const int T_Q1D = 0>
|
||||
static void PAVectorMassApply3D(const int NE,
|
||||
const Array<double> &B_,
|
||||
const Array<double> &Bt_,
|
||||
@@ -384,7 +229,7 @@ static void PAVectorMassApply3D(const int NE,
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
|
||||
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
@@ -542,4 +387,171 @@ void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassAssembleDiagonal2D(const int NE,
|
||||
const Array<double> &B_,
|
||||
const Array<double> &Bt_,
|
||||
const Vector &op_,
|
||||
Vector &diag_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 2;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
|
||||
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, VDIM, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
double temp[max_Q1D][max_D1D];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
temp[qx][dy] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
temp[qx][dy] += B(qy, dy) * B(qy, dy) * op(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double temp1 = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
temp1 += B(qx, dx) * B(qx, dx) * temp[qx][dy];
|
||||
}
|
||||
y(dx, dy, 0, e) = temp1;
|
||||
y(dx, dy, 1, e) = temp1;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<const int T_D1D = 0, const int T_Q1D = 0>
|
||||
static void PAVectorMassAssembleDiagonal3D(const int NE,
|
||||
const Array<double> &B_,
|
||||
const Array<double> &Bt_,
|
||||
const Vector &op_,
|
||||
Vector &diag_,
|
||||
const int d1d = 0,
|
||||
const int q1d = 0)
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
constexpr int VDIM = 3;
|
||||
MFEM_VERIFY(D1D <= MAX_D1D, "");
|
||||
MFEM_VERIFY(Q1D <= MAX_Q1D, "");
|
||||
auto B = Reshape(B_.Read(), Q1D, D1D);
|
||||
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
|
||||
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
|
||||
MFEM_FORALL(e, NE,
|
||||
{
|
||||
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
// the following variables are evaluated at compile time
|
||||
constexpr int max_D1D = T_D1D ? T_D1D : MAX_D1D;
|
||||
constexpr int max_Q1D = T_Q1D ? T_Q1D : MAX_Q1D;
|
||||
|
||||
double temp[max_Q1D][max_Q1D][max_D1D];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
temp[qx][qy][dz] = 0.0;
|
||||
for (int qz = 0; qz < Q1D; ++qz)
|
||||
{
|
||||
temp[qx][qy][dz] += B(qz, dz) * B(qz, dz) * op(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
double temp2[max_Q1D][max_D1D][max_D1D];
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
temp2[qx][dy][dz] = 0.0;
|
||||
for (int qy = 0; qy < Q1D; ++qy)
|
||||
{
|
||||
temp2[qx][dy][dz] += B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dz = 0; dz < D1D; ++dz)
|
||||
{
|
||||
for (int dy = 0; dy < D1D; ++dy)
|
||||
{
|
||||
for (int dx = 0; dx < D1D; ++dx)
|
||||
{
|
||||
double temp3 = 0.0;
|
||||
for (int qx = 0; qx < Q1D; ++qx)
|
||||
{
|
||||
temp3 += B(qx, dx) * B(qx, dx)
|
||||
* temp2[qx][dy][dz];
|
||||
}
|
||||
y(dx, dy, dz, 0, e) = temp3;
|
||||
y(dx, dy, dz, 1, e) = temp3;
|
||||
y(dx, dy, dz, 2, e) = temp3;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
static void PAVectorMassAssembleDiagonal(const int dim,
|
||||
const int D1D,
|
||||
const int Q1D,
|
||||
const int NE,
|
||||
const Array<double> &B,
|
||||
const Array<double> &Bt,
|
||||
const Vector &op,
|
||||
Vector &y)
|
||||
{
|
||||
if (dim == 2)
|
||||
{
|
||||
return PAVectorMassAssembleDiagonal2D(NE, B, Bt, op, y, D1D, Q1D);
|
||||
}
|
||||
else if (dim == 3)
|
||||
{
|
||||
return PAVectorMassAssembleDiagonal3D(NE, B, Bt, op, y, D1D, Q1D);
|
||||
}
|
||||
MFEM_ABORT("Dimension not implemented.");
|
||||
}
|
||||
|
||||
void VectorMassIntegrator::AssembleDiagonalPA(Vector &diag)
|
||||
{
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
ceedOp->GetDiagonal(diag);
|
||||
}
|
||||
else
|
||||
{
|
||||
PAVectorMassAssembleDiagonal(dim,
|
||||
dofs1D,
|
||||
quad1D,
|
||||
ne,
|
||||
maps->B,
|
||||
maps->Bt,
|
||||
pa_data,
|
||||
diag);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
@@ -9,13 +9,19 @@
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../bilininteg.hpp"
|
||||
#include "../gridfunc.hpp"
|
||||
#include "../ceed/integrators/mass/mass.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
#include "ceed/integrators/mass/mass.hpp"
|
||||
|
||||
using namespace std;
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// MF Mass Integrator
|
||||
|
||||
// MF Mass Assemble kernel
|
||||
void VectorMassIntegrator::AssembleMF(const FiniteElementSpace &fes)
|
||||
{
|
||||
// Assuming the same element type
|
||||
File diff suppressed because it is too large
Load Diff
@@ -288,7 +288,7 @@ void InitCoefficientWithIndices(mfem::Coefficient *Q, mfem::Mesh &mesh,
|
||||
auto in = Reshape(qFun.Read(), nq, ne);
|
||||
auto d_indices = Read(m_indices, nelem);
|
||||
auto out = Reshape(ceedCoeff->coeff.Write(), nq, nelem);
|
||||
mfem::forall(nelem * nq, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i, nelem * nq,
|
||||
{
|
||||
const int q = i%nq;
|
||||
const int sub_e = i/nq;
|
||||
@@ -378,7 +378,7 @@ void InitCoefficientWithIndices(mfem::VectorCoefficient *VQ, mfem::Mesh &mesh,
|
||||
auto in = Reshape(qFun.Read(), dim, nq, ne);
|
||||
auto d_indices = Read(m_indices, nelem);
|
||||
auto out = Reshape(ceedCoeff->coeff.Write(), dim, nq, nelem);
|
||||
mfem::forall(nelem * nq, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i, nelem * nq,
|
||||
{
|
||||
const int q = i%nq;
|
||||
const int sub_e = i/nq;
|
||||
|
||||
+12
-27
@@ -13,13 +13,13 @@
|
||||
#define MFEM_LIBCEED_UTIL
|
||||
|
||||
#include "../../../config/config.hpp"
|
||||
#include <functional>
|
||||
#include <string>
|
||||
#include <tuple>
|
||||
#include <unordered_map>
|
||||
#include <string>
|
||||
|
||||
#include "ceed.hpp"
|
||||
#ifdef MFEM_USE_CEED
|
||||
#include <ceed/hash.h>
|
||||
#include <ceed/backend.h> // for CeedOperatorField
|
||||
#endif
|
||||
|
||||
@@ -105,21 +105,6 @@ const IntegrationRule & GetRule(
|
||||
/// Return the path to the libCEED q-function headers.
|
||||
const std::string &GetCeedPath();
|
||||
|
||||
/// Wrapper for std::hash.
|
||||
template <typename T>
|
||||
inline std::size_t CeedHash(const T key)
|
||||
{
|
||||
return std::hash<T> {}(key);
|
||||
}
|
||||
|
||||
/// Effective way to combine hashes (from libCEED).
|
||||
inline std::size_t CeedHashCombine(std::size_t seed, std::size_t hash)
|
||||
{
|
||||
// See https://doi.org/10.1002/asi.10170, or
|
||||
// https://dl.acm.org/citation.cfm?id=759509.
|
||||
return seed ^ (hash + (seed << 6) + (seed >> 2));
|
||||
}
|
||||
|
||||
// Hash table for CeedBasis
|
||||
using BasisKey = std::tuple<const mfem::FiniteElementSpace*,
|
||||
const mfem::IntegrationRule*,
|
||||
@@ -130,12 +115,12 @@ struct BasisHash
|
||||
{
|
||||
return CeedHashCombine(
|
||||
CeedHashCombine(
|
||||
CeedHash(std::get<0>(k)),
|
||||
CeedHash(std::get<1>(k))),
|
||||
CeedHashInt(reinterpret_cast<CeedHash64_t>(std::get<0>(k))),
|
||||
CeedHashInt(reinterpret_cast<CeedHash64_t>(std::get<1>(k)))),
|
||||
CeedHashCombine(
|
||||
CeedHashCombine(CeedHash(std::get<2>(k)),
|
||||
CeedHash(std::get<3>(k))),
|
||||
CeedHash(std::get<4>(k))));
|
||||
CeedHashCombine(CeedHashInt(std::get<2>(k)),
|
||||
CeedHashInt(std::get<3>(k))),
|
||||
CeedHashInt(std::get<4>(k))));
|
||||
}
|
||||
};
|
||||
using BasisMap = std::unordered_map<const BasisKey, CeedBasis, BasisHash>;
|
||||
@@ -152,11 +137,11 @@ struct RestrHash
|
||||
return CeedHashCombine(
|
||||
CeedHashCombine(
|
||||
CeedHashCombine(
|
||||
CeedHash(std::get<0>(k)),
|
||||
CeedHash(std::get<1>(k))),
|
||||
CeedHashCombine(CeedHash(std::get<2>(k)),
|
||||
CeedHash(std::get<3>(k)))),
|
||||
CeedHash(std::get<4>(k)));
|
||||
CeedHashInt(reinterpret_cast<CeedHash64_t>(std::get<0>(k))),
|
||||
CeedHashInt(std::get<1>(k))),
|
||||
CeedHashCombine(CeedHashInt(std::get<2>(k)),
|
||||
CeedHashInt(std::get<3>(k)))),
|
||||
CeedHashInt(std::get<4>(k)));
|
||||
}
|
||||
};
|
||||
using RestrMap =
|
||||
|
||||
@@ -519,7 +519,7 @@ int CeedVectorPointwiseMult(CeedVector a, const CeedVector b)
|
||||
ierr = CeedVectorGetArray(a, mem, &a_data); CeedChk(ierr);
|
||||
ierr = CeedVectorGetArrayRead(b, mem, &b_data); CeedChk(ierr);
|
||||
MFEM_VERIFY(int(length) == length, "length overflow");
|
||||
mfem::forall(length, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i, length,
|
||||
{a_data[i] *= b_data[i];});
|
||||
|
||||
ierr = CeedVectorRestoreArray(a, &a_data); CeedChk(ierr);
|
||||
@@ -593,7 +593,7 @@ void AlgebraicInterpolation::MultTranspose(const mfem::Vector& x,
|
||||
&multiplicitydata); PCeedChk(ierr);
|
||||
ierr = CeedVectorGetArrayWrite(fine_work, mem, &workdata); PCeedChk(ierr);
|
||||
MFEM_VERIFY((int)length == length, "length overflow");
|
||||
mfem::forall(length, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i, length,
|
||||
{workdata[i] = in_ptr[i] * multiplicitydata[i];});
|
||||
ierr = CeedVectorRestoreArrayRead(fine_multiplicity_r,
|
||||
&multiplicitydata);
|
||||
|
||||
+5
-55
@@ -144,54 +144,11 @@ double FunctionCoefficient::Eval(ElementTransformation & T,
|
||||
}
|
||||
}
|
||||
|
||||
double CartesianCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
T.Transform(ip, transip);
|
||||
return transip[comp];
|
||||
}
|
||||
|
||||
double CylindricalRadialCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
T.Transform(ip, transip);
|
||||
return sqrt(transip[0] * transip[0] + transip[1] * transip[1]);
|
||||
}
|
||||
|
||||
double CylindricalAzimuthalCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
T.Transform(ip, transip);
|
||||
return atan2(transip[1], transip[0]);
|
||||
}
|
||||
|
||||
double SphericalRadialCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
T.Transform(ip, transip);
|
||||
return sqrt(transip * transip);
|
||||
}
|
||||
|
||||
double SphericalAzimuthalCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
T.Transform(ip, transip);
|
||||
return atan2(transip[1], transip[0]);
|
||||
}
|
||||
|
||||
double SphericalPolarCoefficient::Eval(ElementTransformation & T,
|
||||
const IntegrationPoint & ip)
|
||||
{
|
||||
T.Transform(ip, transip);
|
||||
return atan2(sqrt(transip[0] * transip[0] + transip[1] * transip[1]),
|
||||
transip[2]);
|
||||
}
|
||||
|
||||
double GridFunctionCoefficient::Eval (ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
Mesh *gf_mesh = GridF->FESpace()->GetMesh();
|
||||
if (T.mesh->GetNE() == gf_mesh->GetNE())
|
||||
if (T.mesh == gf_mesh)
|
||||
{
|
||||
return GridF->GetValue(T, ip, Component);
|
||||
}
|
||||
@@ -356,13 +313,6 @@ void PWVectorCoefficient::Eval(Vector &V, ElementTransformation &T,
|
||||
V = 0.0;
|
||||
}
|
||||
|
||||
void PositionVectorCoefficient::Eval(Vector &V, ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
V.SetSize(vdim);
|
||||
T.Transform(ip, V);
|
||||
}
|
||||
|
||||
void VectorFunctionCoefficient::Eval(Vector &V, ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
@@ -446,7 +396,7 @@ void VectorGridFunctionCoefficient::Eval(Vector &V, ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
Mesh *gf_mesh = GridFunc->FESpace()->GetMesh();
|
||||
if (T.mesh->GetNE() == gf_mesh->GetNE())
|
||||
if (T.mesh == gf_mesh)
|
||||
{
|
||||
GridFunc->GetVectorValue(T, ip, V);
|
||||
}
|
||||
@@ -494,7 +444,7 @@ void GradientGridFunctionCoefficient::Eval(Vector &V, ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
Mesh *gf_mesh = GridFunc->FESpace()->GetMesh();
|
||||
if (T.mesh->GetNE() == gf_mesh->GetNE())
|
||||
if (T.mesh == gf_mesh)
|
||||
{
|
||||
GridFunc->GetGradient(T, V);
|
||||
}
|
||||
@@ -535,7 +485,7 @@ void CurlGridFunctionCoefficient::Eval(Vector &V, ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
Mesh *gf_mesh = GridFunc->FESpace()->GetMesh();
|
||||
if (T.mesh->GetNE() == gf_mesh->GetNE())
|
||||
if (T.mesh == gf_mesh)
|
||||
{
|
||||
GridFunc->GetCurl(T, V);
|
||||
}
|
||||
@@ -557,7 +507,7 @@ double DivergenceGridFunctionCoefficient::Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip)
|
||||
{
|
||||
Mesh *gf_mesh = GridFunc->FESpace()->GetMesh();
|
||||
if (T.mesh->GetNE() == gf_mesh->GetNE())
|
||||
if (T.mesh == gf_mesh)
|
||||
{
|
||||
return GridFunc->GetDivergence(T);
|
||||
}
|
||||
|
||||
@@ -258,124 +258,6 @@ public:
|
||||
const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
/// A common base class for returning individual components of the domain's
|
||||
/// Cartesian coordinates.
|
||||
class CartesianCoefficient : public Coefficient
|
||||
{
|
||||
protected:
|
||||
int comp;
|
||||
mutable Vector transip;
|
||||
|
||||
/// @a comp_ index of the desired component (0 -> x, 1 -> y, 2 -> z)
|
||||
CartesianCoefficient(int comp_) : comp(comp_), transip(3) {}
|
||||
|
||||
public:
|
||||
/// Evaluate the coefficient at @a ip.
|
||||
virtual double Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
/// Scalar coefficient which returns the x-component of the evaluation point
|
||||
class CartesianXCoefficient : public CartesianCoefficient
|
||||
{
|
||||
public:
|
||||
CartesianXCoefficient() : CartesianCoefficient(0) {}
|
||||
};
|
||||
|
||||
/// Scalar coefficient which returns the y-component of the evaluation point
|
||||
class CartesianYCoefficient : public CartesianCoefficient
|
||||
{
|
||||
public:
|
||||
CartesianYCoefficient() : CartesianCoefficient(1) {}
|
||||
};
|
||||
|
||||
/// Scalar coefficient which returns the z-component of the evaluation point
|
||||
class CartesianZCoefficient : public CartesianCoefficient
|
||||
{
|
||||
public:
|
||||
CartesianZCoefficient() : CartesianCoefficient(2) {}
|
||||
};
|
||||
|
||||
/// Scalar coefficient which returns the radial distance from the axis of
|
||||
/// the evaluation point in the cylindrical coordinate system
|
||||
class CylindricalRadialCoefficient : public Coefficient
|
||||
{
|
||||
private:
|
||||
mutable Vector transip;
|
||||
|
||||
public:
|
||||
CylindricalRadialCoefficient() : transip(3) {}
|
||||
|
||||
/// Evaluate the coefficient at @a ip.
|
||||
virtual double Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
/// Scalar coefficient which returns the angular position or azimuth (often
|
||||
/// denoted by theta) of the evaluation point in the cylindrical coordinate
|
||||
/// system
|
||||
class CylindricalAzimuthalCoefficient : public Coefficient
|
||||
{
|
||||
private:
|
||||
mutable Vector transip;
|
||||
|
||||
public:
|
||||
CylindricalAzimuthalCoefficient() : transip(3) {}
|
||||
|
||||
/// Evaluate the coefficient at @a ip.
|
||||
virtual double Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
/// Scalar coefficient which returns the height or altitude of
|
||||
/// the evaluation point in the cylindrical coordinate system
|
||||
typedef CartesianZCoefficient CylindricalZCoefficient;
|
||||
|
||||
/// Scalar coefficient which returns the radial distance from the origin of
|
||||
/// the evaluation point in the spherical coordinate system
|
||||
class SphericalRadialCoefficient : public Coefficient
|
||||
{
|
||||
private:
|
||||
mutable Vector transip;
|
||||
|
||||
public:
|
||||
SphericalRadialCoefficient() : transip(3) {}
|
||||
|
||||
/// Evaluate the coefficient at @a ip.
|
||||
virtual double Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
/// Scalar coefficient which returns the azimuthal angle (often denoted by phi)
|
||||
/// of the evaluation point in the spherical coordinate system
|
||||
class SphericalAzimuthalCoefficient : public Coefficient
|
||||
{
|
||||
private:
|
||||
mutable Vector transip;
|
||||
|
||||
public:
|
||||
SphericalAzimuthalCoefficient() : transip(3) {}
|
||||
|
||||
/// Evaluate the coefficient at @a ip.
|
||||
virtual double Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
/// Scalar coefficient which returns the polar angle (often denoted by theta)
|
||||
/// of the evaluation point in the spherical coordinate system
|
||||
class SphericalPolarCoefficient : public Coefficient
|
||||
{
|
||||
private:
|
||||
mutable Vector transip;
|
||||
|
||||
public:
|
||||
SphericalPolarCoefficient() : transip(3) {}
|
||||
|
||||
/// Evaluate the coefficient at @a ip.
|
||||
virtual double Eval(ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
};
|
||||
|
||||
class GridFunction;
|
||||
|
||||
/// Coefficient defined by a GridFunction. This coefficient is mesh dependent.
|
||||
@@ -718,22 +600,6 @@ public:
|
||||
using VectorCoefficient::Eval;
|
||||
};
|
||||
|
||||
/// A vector coefficient which returns the physical location of the
|
||||
/// evaluation point in the Cartesian coordinate system.
|
||||
class PositionVectorCoefficient : public VectorCoefficient
|
||||
{
|
||||
public:
|
||||
|
||||
PositionVectorCoefficient(int dim) : VectorCoefficient(dim) {}
|
||||
|
||||
using VectorCoefficient::Eval;
|
||||
/// Evaluate the vector coefficient at @a ip.
|
||||
virtual void Eval(Vector &V, ElementTransformation &T,
|
||||
const IntegrationPoint &ip);
|
||||
|
||||
virtual ~PositionVectorCoefficient() { }
|
||||
};
|
||||
|
||||
/// A general vector function coefficient
|
||||
class VectorFunctionCoefficient : public VectorCoefficient
|
||||
{
|
||||
|
||||
+2
-2
@@ -497,7 +497,7 @@ SesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
auto d_X_r = X_r.Read();
|
||||
auto d_X_i = X_i.Read();
|
||||
auto d_idx = ess_tdof_list.Read();
|
||||
mfem::forall(n, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i, n,
|
||||
{
|
||||
const int j = d_idx[i];
|
||||
d_B_r[j] = d_X_r[j];
|
||||
@@ -1230,7 +1230,7 @@ ParSesquilinearForm::FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
auto d_X_r = X_r.Read();
|
||||
auto d_X_i = X_i.Read();
|
||||
auto d_idx = ess_tdof_list.Read();
|
||||
mfem::forall(n, [=] MFEM_HOST_DEVICE (int i)
|
||||
MFEM_FORALL(i, n,
|
||||
{
|
||||
const int j = d_idx[i];
|
||||
d_B_r[j] = d_X_r[j];
|
||||
|
||||
+2
-2
@@ -107,7 +107,7 @@ void DGMassInverse::Update()
|
||||
{
|
||||
M->Assemble();
|
||||
M->AssembleDiagonal(diag_inv);
|
||||
diag_inv.Reciprocal();
|
||||
internal::MakeReciprocal(diag_inv.Size(), diag_inv.ReadWrite());
|
||||
}
|
||||
|
||||
DGMassInverse::~DGMassInverse()
|
||||
@@ -168,7 +168,7 @@ void DGMassInverse::DGMassCGIteration(const Vector &b_, Vector &u_) const
|
||||
|
||||
constexpr int NB = Q1D ? Q1D : 1; // block size
|
||||
|
||||
mfem::forall_2D(NE, NB, NB, [=] MFEM_HOST_DEVICE (int e)
|
||||
MFEM_FORALL_2D(e, NE, NB, NB, 1,
|
||||
{
|
||||
constexpr int NB = Q1D ? Q1D : 1; // redefine here for some compilers
|
||||
|
||||
|
||||
+2
-4
@@ -87,8 +87,6 @@ public:
|
||||
///
|
||||
/// If @ref iterative_mode is @a true, @a u is used as an initial guess.
|
||||
void Mult(const Vector &b, Vector &u) const;
|
||||
/// Same as Mult() since the mass matrix is symmetric.
|
||||
void MultTranspose(const Vector &b, Vector &u) const { Mult(b, u); }
|
||||
/// Not implemented. Aborts.
|
||||
void SetOperator(const Operator &op);
|
||||
/// Set the relative tolerance.
|
||||
@@ -103,8 +101,8 @@ public:
|
||||
~DGMassInverse();
|
||||
|
||||
/// @brief Solve the system M b = u. <b>Not part of the public interface.</b>
|
||||
/// @note This member function must be public because it defines an
|
||||
/// extended lambda used in an mfem::forall kernel (nvcc limitation)
|
||||
/// @note This member function must be public because it contains an
|
||||
/// MFEM_FORALL kernel (nvcc limitation)
|
||||
template<int DIM, int D1D = 0, int Q1D = 0>
|
||||
void DGMassCGIteration(const Vector &b_, Vector &u_) const;
|
||||
};
|
||||
|
||||
@@ -12,9 +12,9 @@
|
||||
#ifndef MFEM_DGMASSINV_KERNELS_HPP
|
||||
#define MFEM_DGMASSINV_KERNELS_HPP
|
||||
|
||||
#include "bilininteg_mass_pa.hpp"
|
||||
#include "../linalg/kernels.hpp"
|
||||
#include "kernels.hpp"
|
||||
#include "integ/bilininteg_mass_kernels.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -22,6 +22,11 @@ namespace mfem
|
||||
namespace internal
|
||||
{
|
||||
|
||||
void MakeReciprocal(int n, double *x)
|
||||
{
|
||||
MFEM_FORALL(i, n, x[i] = 1.0/x[i]; );
|
||||
}
|
||||
|
||||
template <int DIM, int D1D, int Q1D>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void DGMassApply(const int e,
|
||||
|
||||
+300
-66
@@ -14,6 +14,54 @@
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void DofTransformation::TransformPrimal(Vector &v) const
|
||||
{
|
||||
TransformPrimal(v.GetData());
|
||||
}
|
||||
|
||||
void DofTransformation::TransformPrimalCols(DenseMatrix &V) const
|
||||
{
|
||||
for (int c=0; c<V.Width(); c++)
|
||||
{
|
||||
TransformPrimal(V.GetColumn(c));
|
||||
}
|
||||
}
|
||||
|
||||
void DofTransformation::TransformDual(Vector &v) const
|
||||
{
|
||||
TransformDual(v.GetData());
|
||||
}
|
||||
|
||||
void DofTransformation::TransformDual(DenseMatrix &V) const
|
||||
{
|
||||
TransformDualCols(V);
|
||||
TransformDualRows(V);
|
||||
}
|
||||
|
||||
void DofTransformation::TransformDualRows(DenseMatrix &V) const
|
||||
{
|
||||
Vector row;
|
||||
for (int r=0; r<V.Height(); r++)
|
||||
{
|
||||
V.GetRow(r, row);
|
||||
TransformDual(row);
|
||||
V.SetRow(r, row);
|
||||
}
|
||||
}
|
||||
|
||||
void DofTransformation::TransformDualCols(DenseMatrix &V) const
|
||||
{
|
||||
for (int c=0; c<V.Width(); c++)
|
||||
{
|
||||
TransformDual(V.GetColumn(c));
|
||||
}
|
||||
}
|
||||
|
||||
void DofTransformation::InvTransformPrimal(Vector &v) const
|
||||
{
|
||||
InvTransformPrimal(v.GetData());
|
||||
}
|
||||
|
||||
void TransformPrimal(const DofTransformation *ran_dof_trans,
|
||||
const DofTransformation *dom_dof_trans,
|
||||
DenseMatrix &elmat)
|
||||
@@ -37,6 +85,11 @@ void TransformPrimal(const DofTransformation *ran_dof_trans,
|
||||
}
|
||||
}
|
||||
|
||||
void DofTransformation::InvTransformDual(Vector &v) const
|
||||
{
|
||||
InvTransformDual(v.GetData());
|
||||
}
|
||||
|
||||
void TransformDual(const DofTransformation *ran_dof_trans,
|
||||
const DofTransformation *dom_dof_trans,
|
||||
DenseMatrix &elmat)
|
||||
@@ -60,16 +113,15 @@ void TransformDual(const DofTransformation *ran_dof_trans,
|
||||
}
|
||||
}
|
||||
|
||||
void StatelessVDofTransformation::TransformPrimal(const Array<int> & face_ori,
|
||||
double *v) const
|
||||
void VDofTransformation::TransformPrimal(double *v) const
|
||||
{
|
||||
int size = sdoftrans_->Size();
|
||||
int size = doftrans_->Size();
|
||||
|
||||
if ((Ordering::Type)ordering_ == Ordering::byNODES || vdim_ == 1)
|
||||
{
|
||||
for (int i=0; i<vdim_; i++)
|
||||
{
|
||||
sdoftrans_->TransformPrimal(face_ori, &v[i*size]);
|
||||
doftrans_->TransformPrimal(&v[i*size]);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -81,7 +133,7 @@ void StatelessVDofTransformation::TransformPrimal(const Array<int> & face_ori,
|
||||
{
|
||||
vec(j) = v[j*vdim_+i];
|
||||
}
|
||||
sdoftrans_->TransformPrimal(face_ori, vec);
|
||||
doftrans_->TransformPrimal(vec);
|
||||
for (int j=0; j<size; j++)
|
||||
{
|
||||
v[j*vdim_+i] = vec(j);
|
||||
@@ -90,17 +142,15 @@ void StatelessVDofTransformation::TransformPrimal(const Array<int> & face_ori,
|
||||
}
|
||||
}
|
||||
|
||||
void StatelessVDofTransformation::InvTransformPrimal(
|
||||
const Array<int> & face_ori,
|
||||
double *v) const
|
||||
void VDofTransformation::InvTransformPrimal(double *v) const
|
||||
{
|
||||
int size = sdoftrans_->Height();
|
||||
int size = doftrans_->Height();
|
||||
|
||||
if ((Ordering::Type)ordering_ == Ordering::byNODES)
|
||||
{
|
||||
for (int i=0; i<vdim_; i++)
|
||||
{
|
||||
sdoftrans_->InvTransformPrimal(face_ori, &v[i*size]);
|
||||
doftrans_->InvTransformPrimal(&v[i*size]);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -112,7 +162,7 @@ void StatelessVDofTransformation::InvTransformPrimal(
|
||||
{
|
||||
vec(j) = v[j*vdim_+i];
|
||||
}
|
||||
sdoftrans_->InvTransformPrimal(face_ori, vec);
|
||||
doftrans_->InvTransformPrimal(vec);
|
||||
for (int j=0; j<size; j++)
|
||||
{
|
||||
v[j*vdim_+i] = vec(j);
|
||||
@@ -121,16 +171,15 @@ void StatelessVDofTransformation::InvTransformPrimal(
|
||||
}
|
||||
}
|
||||
|
||||
void StatelessVDofTransformation::TransformDual(const Array<int> & face_ori,
|
||||
double *v) const
|
||||
void VDofTransformation::TransformDual(double *v) const
|
||||
{
|
||||
int size = sdoftrans_->Size();
|
||||
int size = doftrans_->Size();
|
||||
|
||||
if ((Ordering::Type)ordering_ == Ordering::byNODES)
|
||||
{
|
||||
for (int i=0; i<vdim_; i++)
|
||||
{
|
||||
sdoftrans_->TransformDual(face_ori, &v[i*size]);
|
||||
doftrans_->TransformDual(&v[i*size]);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -142,7 +191,7 @@ void StatelessVDofTransformation::TransformDual(const Array<int> & face_ori,
|
||||
{
|
||||
vec(j) = v[j*vdim_+i];
|
||||
}
|
||||
sdoftrans_->TransformDual(face_ori, vec);
|
||||
doftrans_->TransformDual(vec);
|
||||
for (int j=0; j<size; j++)
|
||||
{
|
||||
v[j*vdim_+i] = vec(j);
|
||||
@@ -151,16 +200,15 @@ void StatelessVDofTransformation::TransformDual(const Array<int> & face_ori,
|
||||
}
|
||||
}
|
||||
|
||||
void StatelessVDofTransformation::InvTransformDual(const Array<int> & face_ori,
|
||||
double *v) const
|
||||
void VDofTransformation::InvTransformDual(double *v) const
|
||||
{
|
||||
int size = sdoftrans_->Size();
|
||||
int size = doftrans_->Size();
|
||||
|
||||
if ((Ordering::Type)ordering_ == Ordering::byNODES)
|
||||
{
|
||||
for (int i=0; i<vdim_; i++)
|
||||
{
|
||||
sdoftrans_->InvTransformDual(face_ori, &v[i*size]);
|
||||
doftrans_->InvTransformDual(&v[i*size]);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -172,7 +220,7 @@ void StatelessVDofTransformation::InvTransformDual(const Array<int> & face_ori,
|
||||
{
|
||||
vec(j) = v[j*vdim_+i];
|
||||
}
|
||||
sdoftrans_->InvTransformDual(face_ori, vec);
|
||||
doftrans_->InvTransformDual(vec);
|
||||
for (int j=0; j<size; j++)
|
||||
{
|
||||
v[j*vdim_+i] = vec(j);
|
||||
@@ -181,8 +229,7 @@ void StatelessVDofTransformation::InvTransformDual(const Array<int> & face_ori,
|
||||
}
|
||||
}
|
||||
|
||||
// ordering (i0j0, i1j0, i0j1, i1j1), each row is a column major matrix
|
||||
const double ND_StatelessDofTransformation::T_data[24] =
|
||||
const double ND_DofTransformation::T_data[24] =
|
||||
{
|
||||
1.0, 0.0, 0.0, 1.0,
|
||||
-1.0, -1.0, 0.0, 1.0,
|
||||
@@ -192,11 +239,10 @@ const double ND_StatelessDofTransformation::T_data[24] =
|
||||
0.0, 1.0, 1.0, 0.0
|
||||
};
|
||||
|
||||
const DenseTensor ND_StatelessDofTransformation
|
||||
::T(const_cast<double*>(ND_StatelessDofTransformation::T_data), 2, 2, 6);
|
||||
const DenseTensor ND_DofTransformation
|
||||
::T(const_cast<double*>(ND_DofTransformation::T_data), 2, 2, 6);
|
||||
|
||||
// ordering (i0j0, i1j0, i0j1, i1j1), each row is a column major matrix
|
||||
const double ND_StatelessDofTransformation::TInv_data[24] =
|
||||
const double ND_DofTransformation::TInv_data[24] =
|
||||
{
|
||||
1.0, 0.0, 0.0, 1.0,
|
||||
-1.0, -1.0, 0.0, 1.0,
|
||||
@@ -206,113 +252,301 @@ const double ND_StatelessDofTransformation::TInv_data[24] =
|
||||
0.0, 1.0, 1.0, 0.0
|
||||
};
|
||||
|
||||
const DenseTensor ND_StatelessDofTransformation
|
||||
const DenseTensor ND_DofTransformation
|
||||
::TInv(const_cast<double*>(TInv_data), 2, 2, 6);
|
||||
|
||||
ND_StatelessDofTransformation::ND_StatelessDofTransformation(int size, int p,
|
||||
int num_edges,
|
||||
int num_tri_faces)
|
||||
: StatelessDofTransformation(size)
|
||||
ND_DofTransformation::ND_DofTransformation(int size, int p)
|
||||
: DofTransformation(size)
|
||||
, order(p)
|
||||
, nedofs(p)
|
||||
, nfdofs(p*(p-1))
|
||||
, nedges(num_edges)
|
||||
, nfaces(num_tri_faces)
|
||||
{
|
||||
}
|
||||
|
||||
void ND_StatelessDofTransformation::TransformPrimal(const Array<int> & Fo,
|
||||
double *v) const
|
||||
ND_TriDofTransformation::ND_TriDofTransformation(int p)
|
||||
: ND_DofTransformation(p*(p + 2), p)
|
||||
{
|
||||
}
|
||||
|
||||
void ND_TriDofTransformation::TransformPrimal(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= nfaces,
|
||||
"Face orientation array is shorter than the number of faces in "
|
||||
"ND_StatelessDofTransformation");
|
||||
MFEM_VERIFY(Fo.Size() >= 1,
|
||||
"Face orientations are unset in ND_TriDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<nfaces; f++)
|
||||
for (int f=0; f<1; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[nedges*nedofs + f*nfdofs + 2*i];
|
||||
T(Fo[f]).Mult(v2, &v[nedges*nedofs + f*nfdofs + 2*i]);
|
||||
v2 = &v[3*nedofs + f*nfdofs + 2*i];
|
||||
T(Fo[f]).Mult(v2, &v[3*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ND_StatelessDofTransformation::InvTransformPrimal(const Array<int> & Fo,
|
||||
double *v) const
|
||||
void
|
||||
ND_TriDofTransformation::InvTransformPrimal(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= nfaces,
|
||||
"Face orientation array is shorter than the number of faces in "
|
||||
"ND_StatelessDofTransformation");
|
||||
MFEM_VERIFY(Fo.Size() >= 1,
|
||||
"Face orientations are unset in ND_TriDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<nfaces; f++)
|
||||
for (int f=0; f<1; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[nedges*nedofs + f*nfdofs + 2*i];
|
||||
TInv(Fo[f]).Mult(v2, &v[nedges*nedofs + f*nfdofs + 2*i]);
|
||||
v2 = &v[3*nedofs + f*nfdofs + 2*i];
|
||||
TInv(Fo[f]).Mult(v2, &v[3*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ND_StatelessDofTransformation::TransformDual(const Array<int> & Fo,
|
||||
double *v) const
|
||||
void
|
||||
ND_TriDofTransformation::TransformDual(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= nfaces,
|
||||
"Face orientation array is shorter than the number of faces in "
|
||||
"ND_StatelessDofTransformation");
|
||||
MFEM_VERIFY(Fo.Size() >= 1,
|
||||
"Face orientations are unset in ND_TriDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<nfaces; f++)
|
||||
for (int f=0; f<1; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[nedges*nedofs + f*nfdofs + 2*i];
|
||||
TInv(Fo[f]).MultTranspose(v2, &v[nedges*nedofs + f*nfdofs + 2*i]);
|
||||
v2 = &v[3*nedofs + f*nfdofs + 2*i];
|
||||
TInv(Fo[f]).MultTranspose(v2, &v[3*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ND_StatelessDofTransformation::InvTransformDual(const Array<int> & Fo,
|
||||
double *v) const
|
||||
void
|
||||
ND_TriDofTransformation::InvTransformDual(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= nfaces,
|
||||
"Face orientation array is shorter than the number of faces in "
|
||||
"ND_StatelessDofTransformation");
|
||||
MFEM_VERIFY(Fo.Size() >= 1,
|
||||
"Face orientations are unset in ND_TriDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<nfaces; f++)
|
||||
for (int f=0; f<1; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[nedges*nedofs + f*nfdofs + 2*i];
|
||||
T(Fo[f]).MultTranspose(v2, &v[nedges*nedofs + f*nfdofs + 2*i]);
|
||||
v2 = &v[3*nedofs + f*nfdofs + 2*i];
|
||||
T(Fo[f]).MultTranspose(v2, &v[3*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ND_TetDofTransformation::ND_TetDofTransformation(int p)
|
||||
: ND_DofTransformation(p*(p + 2)*(p + 3)/2, p)
|
||||
{
|
||||
}
|
||||
|
||||
void ND_TetDofTransformation::TransformPrimal(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= 4,
|
||||
"Face orientations are unset in ND_TetDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<4; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[6*nedofs + f*nfdofs + 2*i];
|
||||
T(Fo[f]).Mult(v2, &v[6*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
ND_TetDofTransformation::InvTransformPrimal(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= 4,
|
||||
"Face orientations are unset in ND_TetDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<4; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[6*nedofs + f*nfdofs + 2*i];
|
||||
TInv(Fo[f]).Mult(v2, &v[6*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
ND_TetDofTransformation::TransformDual(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= 4,
|
||||
"Face orientations are unset in ND_TetDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<4; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[6*nedofs + f*nfdofs + 2*i];
|
||||
TInv(Fo[f]).MultTranspose(v2, &v[6*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
ND_TetDofTransformation::InvTransformDual(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= 4,
|
||||
"Face orientations are unset in ND_TetDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform face DoFs
|
||||
for (int f=0; f<4; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[6*nedofs + f*nfdofs + 2*i];
|
||||
T(Fo[f]).MultTranspose(v2, &v[6*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ND_WedgeDofTransformation::ND_WedgeDofTransformation(int p)
|
||||
: ND_DofTransformation(3 * p * ((p + 1) * (p + 2))/2, p)
|
||||
{
|
||||
}
|
||||
|
||||
void ND_WedgeDofTransformation::TransformPrimal(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= 2,
|
||||
"Face orientations are unset in ND_WedgeDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform triangular face DoFs
|
||||
for (int f=0; f<2; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[9*nedofs + f*nfdofs + 2*i];
|
||||
T(Fo[f]).Mult(v2, &v[9*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
ND_WedgeDofTransformation::InvTransformPrimal(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= 2,
|
||||
"Face orientations are unset in ND_WedgeDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform triangular face DoFs
|
||||
for (int f=0; f<2; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[9*nedofs + f*nfdofs + 2*i];
|
||||
TInv(Fo[f]).Mult(v2, &v[9*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
ND_WedgeDofTransformation::TransformDual(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= 2,
|
||||
"Face orientations are unset in ND_WedgeDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform triangular face DoFs
|
||||
for (int f=0; f<2; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[9*nedofs + f*nfdofs + 2*i];
|
||||
TInv(Fo[f]).MultTranspose(v2, &v[9*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
ND_WedgeDofTransformation::InvTransformDual(double *v) const
|
||||
{
|
||||
// Return immediately when no face DoFs are present
|
||||
if (nfdofs < 2) { return; }
|
||||
|
||||
MFEM_VERIFY(Fo.Size() >= 2,
|
||||
"Face orientations are unset in ND_WedgeDofTransformation");
|
||||
|
||||
double data[2];
|
||||
Vector v2(data, 2);
|
||||
|
||||
// Transform triangular face DoFs
|
||||
for (int f=0; f<2; f++)
|
||||
{
|
||||
for (int i=0; i<nfdofs/2; i++)
|
||||
{
|
||||
v2 = &v[9*nedofs + f*nfdofs + 2*i];
|
||||
T(Fo[f]).MultTranspose(v2, &v[9*nedofs + f*nfdofs + 2*i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+90
-334
@@ -15,31 +15,19 @@
|
||||
#include "../config/config.hpp"
|
||||
#include "../linalg/linalg.hpp"
|
||||
#include "intrules.hpp"
|
||||
#include "fe.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/** The StatelessDofTransformation class is an abstract base class for a family
|
||||
of transformations that map local degrees of freedom (DoFs), contained
|
||||
within individual elements, to global degrees of freedom, stored within
|
||||
GridFunction objects.
|
||||
|
||||
In this context "stateless" means that the concrete classes derived from
|
||||
StatelessDofTransformation do not store information about the relative
|
||||
orientations of the faces with respect to their neighboring elements. In
|
||||
other words there is no information specific to a particular element (aside
|
||||
from the element type e.g. tetrahedron, wedge, or pyramid). The
|
||||
StatelessDofTransformation provides access to the transformation operators
|
||||
for specific relative face orientations. These are useful, for example, when
|
||||
relating DoFs associated with distinct overlapping meshes such as parent and
|
||||
sub-meshes.
|
||||
|
||||
These transformations are necessary to ensure that basis functions in
|
||||
neighboring (or overlapping) elements align correctly. Closely related but
|
||||
/** The DofTransformation class is an abstract base class for a family of
|
||||
transformations that map local degrees of freedom (DoFs), contained within
|
||||
individual elements, to global degrees of freedom, stored within
|
||||
GridFunction objects. These transformations are necessary to ensure that
|
||||
basis functions in neighboring elements align correctly. Closely related but
|
||||
complementary transformations are required for the entries stored in
|
||||
LinearForm and BilinearForm objects. The StatelessDofTransformation class
|
||||
is designed to apply the action of both of these types of DoF
|
||||
transformations.
|
||||
LinearForm and BilinearForm objects. The DofTransformation class is designed
|
||||
to apply the action of both of these types of DoF transformations.
|
||||
|
||||
Let the "primal transformation" be given by the operator T. This means that
|
||||
given a local element vector v the data that must be placed into a
|
||||
@@ -65,84 +53,24 @@ namespace mfem
|
||||
D_t = T * D * T^{-1}. This can be accomplished by using a primal
|
||||
transformation on the columns of D and a dual transformation on its rows.
|
||||
*/
|
||||
class StatelessDofTransformation
|
||||
class DofTransformation
|
||||
{
|
||||
protected:
|
||||
int size_;
|
||||
|
||||
StatelessDofTransformation(int size)
|
||||
Array<int> Fo;
|
||||
|
||||
DofTransformation(int size)
|
||||
: size_(size) {}
|
||||
|
||||
public:
|
||||
|
||||
inline int Size() const { return size_; }
|
||||
inline int Height() const { return size_; }
|
||||
inline int NumRows() const { return size_; }
|
||||
inline int Width() const { return size_; }
|
||||
inline int NumCols() const { return size_; }
|
||||
|
||||
/** Transform local DoFs to align with the global DoFs. For example, this
|
||||
transformation can be used to map the local vector computed by
|
||||
FiniteElement::Project() to the transformed vector stored within a
|
||||
GridFunction object. */
|
||||
virtual void TransformPrimal(const Array<int> & face_orientation,
|
||||
double *v) const = 0;
|
||||
inline void TransformPrimal(const Array<int> & face_orientation,
|
||||
Vector &v) const
|
||||
{ TransformPrimal(face_orientation, v.GetData()); }
|
||||
|
||||
/** Inverse transform local DoFs. Used to transform DoFs from a global vector
|
||||
back to their element-local form. For example, this must be used to
|
||||
transform the vector obtained using GridFunction::GetSubVector before it
|
||||
can be used to compute a local interpolation.
|
||||
*/
|
||||
virtual void InvTransformPrimal(const Array<int> & face_orientation,
|
||||
double *v) const = 0;
|
||||
inline void InvTransformPrimal(const Array<int> & face_orientation,
|
||||
Vector &v) const
|
||||
{ InvTransformPrimal(face_orientation, v.GetData()); }
|
||||
|
||||
/** Transform dual DoFs as computed by a LinearFormIntegrator before summing
|
||||
into a LinearForm object. */
|
||||
virtual void TransformDual(const Array<int> & face_orientation,
|
||||
double *v) const = 0;
|
||||
inline void TransformDual(const Array<int> & face_orientation,
|
||||
Vector &v) const
|
||||
{ TransformDual(face_orientation, v.GetData()); }
|
||||
|
||||
/** Inverse Transform dual DoFs */
|
||||
virtual void InvTransformDual(const Array<int> & face_orientation,
|
||||
double *v) const = 0;
|
||||
inline void InvTransformDual(const Array<int> & face_orientation,
|
||||
Vector &v) const
|
||||
{ InvTransformDual(face_orientation, v.GetData()); }
|
||||
};
|
||||
|
||||
/** The DofTransformation class is an extension of the
|
||||
StatelessDofTransformation which stores the face orientations used to
|
||||
select the necessary transformations which allows it to offer a collection
|
||||
of convenience methods.
|
||||
|
||||
DofTransformation objects are provided by the FiniteElementSpace which has
|
||||
access to the mesh and can therefore provide the face orientations. This is
|
||||
convenient when working with GridFunction, LinearForm, or BilinearForm
|
||||
obejcts or their parallel counterparts.
|
||||
|
||||
StatelessDofTransformation objects are provided by FiniteElement or
|
||||
FiniteElementCollection objects which do not have access to face
|
||||
orientation information. This can be useful in non-standard contexts such as
|
||||
transferring finite element degrees of freedom between different meshes.
|
||||
For examples of its use see the TransferMap used by the SubMesh class.
|
||||
*/
|
||||
class DofTransformation : virtual public StatelessDofTransformation
|
||||
{
|
||||
protected:
|
||||
Array<int> Fo;
|
||||
|
||||
DofTransformation(int size)
|
||||
: StatelessDofTransformation(size) {}
|
||||
|
||||
public:
|
||||
|
||||
/** @brief Configure the transformation using face orientations for the
|
||||
current element. */
|
||||
/// The face_orientation array can be obtained from Mesh::GetElementFaces.
|
||||
@@ -151,82 +79,42 @@ public:
|
||||
|
||||
inline const Array<int> & GetFaceOrientations() const { return Fo; }
|
||||
|
||||
using StatelessDofTransformation::TransformPrimal;
|
||||
using StatelessDofTransformation::InvTransformPrimal;
|
||||
using StatelessDofTransformation::TransformDual;
|
||||
using StatelessDofTransformation::InvTransformDual;
|
||||
|
||||
/** Transform local DoFs to align with the global DoFs. For example, this
|
||||
transformation can be used to map the local vector computed by
|
||||
FiniteElement::Project() to the transformed vector stored within a
|
||||
GridFunction object. */
|
||||
inline void TransformPrimal(double *v) const
|
||||
{ TransformPrimal(Fo, v); }
|
||||
inline void TransformPrimal(Vector &v) const
|
||||
{ TransformPrimal(v.GetData()); }
|
||||
virtual void TransformPrimal(double *v) const = 0;
|
||||
virtual void TransformPrimal(Vector &v) const;
|
||||
|
||||
/// Transform groups of DoFs stored as dense matrices
|
||||
inline void TransformPrimalCols(DenseMatrix &V) const
|
||||
{
|
||||
for (int c=0; c<V.Width(); c++)
|
||||
{
|
||||
TransformPrimal(V.GetColumn(c));
|
||||
}
|
||||
}
|
||||
virtual void TransformPrimalCols(DenseMatrix &V) const;
|
||||
|
||||
/** Inverse transform local DoFs. Used to transform DoFs from a global vector
|
||||
back to their element-local form. For example, this must be used to
|
||||
transform the vector obtained using GridFunction::GetSubVector before it
|
||||
can be used to compute a local interpolation.
|
||||
*/
|
||||
inline void InvTransformPrimal(double *v) const
|
||||
{ InvTransformPrimal(Fo, v); }
|
||||
inline void InvTransformPrimal(Vector &v) const
|
||||
{ InvTransformPrimal(v.GetData()); }
|
||||
virtual void InvTransformPrimal(double *v) const = 0;
|
||||
virtual void InvTransformPrimal(Vector &v) const;
|
||||
|
||||
/** Transform dual DoFs as computed by a LinearFormIntegrator before summing
|
||||
into a LinearForm object. */
|
||||
inline void TransformDual(double *v) const
|
||||
{ TransformDual(Fo, v); }
|
||||
inline void TransformDual(Vector &v) const
|
||||
{ TransformDual(v.GetData()); }
|
||||
virtual void TransformDual(double *v) const = 0;
|
||||
virtual void TransformDual(Vector &v) const;
|
||||
|
||||
/** Inverse Transform dual DoFs */
|
||||
inline void InvTransformDual(double *v) const
|
||||
{ InvTransformDual(Fo, v); }
|
||||
inline void InvTransformDual(Vector &v) const
|
||||
{ InvTransformDual(v.GetData()); }
|
||||
virtual void InvTransformDual(double *v) const = 0;
|
||||
virtual void InvTransformDual(Vector &v) const;
|
||||
|
||||
/** Transform a matrix of dual DoFs entries as computed by a
|
||||
BilinearFormIntegrator before summing into a BilinearForm object. */
|
||||
inline void TransformDual(DenseMatrix &V) const
|
||||
{
|
||||
TransformDualCols(V);
|
||||
TransformDualRows(V);
|
||||
}
|
||||
virtual void TransformDual(DenseMatrix &V) const;
|
||||
|
||||
/// Transform rows of a dense matrix containing dual DoFs
|
||||
inline void TransformDualRows(DenseMatrix &V) const
|
||||
{
|
||||
Vector row;
|
||||
for (int r=0; r<V.Height(); r++)
|
||||
{
|
||||
V.GetRow(r, row);
|
||||
TransformDual(row);
|
||||
V.SetRow(r, row);
|
||||
}
|
||||
}
|
||||
/// Transform groups of dual DoFs stored as dense matrices
|
||||
virtual void TransformDualRows(DenseMatrix &V) const;
|
||||
virtual void TransformDualCols(DenseMatrix &V) const;
|
||||
|
||||
/// Transform columns of a dense matrix containing dual DoFs
|
||||
inline void TransformDualCols(DenseMatrix &V) const
|
||||
{
|
||||
for (int c=0; c<V.Width(); c++)
|
||||
{
|
||||
TransformDual(V.GetColumn(c));
|
||||
}
|
||||
}
|
||||
|
||||
virtual ~DofTransformation() = default;
|
||||
virtual ~DofTransformation() {}
|
||||
};
|
||||
|
||||
/** Transform a matrix of DoFs entries from different finite element spaces as
|
||||
@@ -245,143 +133,66 @@ void TransformDual(const DofTransformation *ran_dof_trans,
|
||||
const DofTransformation *dom_dof_trans,
|
||||
DenseMatrix &elmat);
|
||||
|
||||
/** The StatelessVDofTransformation class implements a nested transformation
|
||||
where an arbitrary StatelessDofTransformation is replicated with a
|
||||
vdim >= 1.
|
||||
*/
|
||||
class StatelessVDofTransformation : virtual public StatelessDofTransformation
|
||||
{
|
||||
protected:
|
||||
int vdim_;
|
||||
int ordering_;
|
||||
StatelessDofTransformation * sdoftrans_;
|
||||
|
||||
public:
|
||||
/** @brief Default constructor which requires that SetDofTransformation be
|
||||
called before use. */
|
||||
StatelessVDofTransformation(int vdim = 1, int ordering = 0)
|
||||
: StatelessDofTransformation(0)
|
||||
, vdim_(vdim)
|
||||
, ordering_(ordering)
|
||||
, sdoftrans_(NULL)
|
||||
{}
|
||||
|
||||
/// Constructor with a known StatelessDofTransformation
|
||||
StatelessVDofTransformation(StatelessDofTransformation & doftrans,
|
||||
int vdim = 1,
|
||||
int ordering = 0)
|
||||
: StatelessDofTransformation(vdim * doftrans.Size())
|
||||
, vdim_(vdim)
|
||||
, ordering_(ordering)
|
||||
, sdoftrans_(&doftrans)
|
||||
{}
|
||||
|
||||
/// Set or change the vdim parameter
|
||||
inline void SetVDim(int vdim)
|
||||
{
|
||||
vdim_ = vdim;
|
||||
if (sdoftrans_)
|
||||
{
|
||||
size_ = vdim_ * sdoftrans_->Size();
|
||||
}
|
||||
}
|
||||
|
||||
/// Return the current vdim value
|
||||
inline int GetVDim() const { return vdim_; }
|
||||
|
||||
/// Set or change the nested StatelessDofTransformation object
|
||||
inline void SetDofTransformation(StatelessDofTransformation & doftrans)
|
||||
{
|
||||
size_ = vdim_ * doftrans.Size();
|
||||
sdoftrans_ = &doftrans;
|
||||
}
|
||||
|
||||
/// Return the nested StatelessDofTransformation object
|
||||
inline StatelessDofTransformation * GetDofTransformation() const
|
||||
{ return sdoftrans_; }
|
||||
|
||||
using StatelessDofTransformation::TransformPrimal;
|
||||
using StatelessDofTransformation::InvTransformPrimal;
|
||||
using StatelessDofTransformation::TransformDual;
|
||||
using StatelessDofTransformation::InvTransformDual;
|
||||
|
||||
/** Specializations of these base class methods which account for the vdim
|
||||
and ordering of the full set of DoFs.
|
||||
*/
|
||||
void TransformPrimal(const Array<int> & face_ori, double *v) const;
|
||||
void InvTransformPrimal(const Array<int> & face_ori, double *v) const;
|
||||
void TransformDual(const Array<int> & face_ori, double *v) const;
|
||||
void InvTransformDual(const Array<int> & face_ori, double *v) const;
|
||||
};
|
||||
|
||||
/** The VDofTransformation class implements a nested transformation where an
|
||||
arbitrary DofTransformation is replicated with a vdim >= 1.
|
||||
*/
|
||||
class VDofTransformation : public StatelessVDofTransformation,
|
||||
public DofTransformation
|
||||
class VDofTransformation : public DofTransformation
|
||||
{
|
||||
protected:
|
||||
private:
|
||||
int vdim_;
|
||||
int ordering_;
|
||||
DofTransformation * doftrans_;
|
||||
|
||||
public:
|
||||
/** @brief Default constructor which requires that SetDofTransformation be
|
||||
called before use. */
|
||||
VDofTransformation(int vdim = 1, int ordering = 0)
|
||||
: StatelessDofTransformation(0)
|
||||
, StatelessVDofTransformation(vdim, ordering)
|
||||
, DofTransformation(0)
|
||||
, doftrans_(NULL)
|
||||
{}
|
||||
: DofTransformation(0),
|
||||
vdim_(vdim), ordering_(ordering),
|
||||
doftrans_(NULL) {}
|
||||
|
||||
/// Constructor with a known DofTransformation
|
||||
/// @note The face orientations in @a doftrans will be copied into the
|
||||
/// new VDofTransformation object.
|
||||
VDofTransformation(DofTransformation & doftrans, int vdim = 1,
|
||||
int ordering = 0)
|
||||
: StatelessDofTransformation(vdim * doftrans.Size())
|
||||
, StatelessVDofTransformation(doftrans, vdim, ordering)
|
||||
, DofTransformation(vdim * doftrans.Size())
|
||||
, doftrans_(&doftrans)
|
||||
: DofTransformation(vdim * doftrans.Size()),
|
||||
vdim_(vdim), ordering_(ordering),
|
||||
doftrans_(&doftrans) {}
|
||||
|
||||
/// Set or change the vdim parameter
|
||||
inline void SetVDim(int vdim)
|
||||
{
|
||||
DofTransformation::SetFaceOrientations(doftrans.GetFaceOrientations());
|
||||
vdim_ = vdim;
|
||||
if (doftrans_)
|
||||
{
|
||||
size_ = vdim_ * doftrans_->Size();
|
||||
}
|
||||
}
|
||||
|
||||
using StatelessVDofTransformation::SetDofTransformation;
|
||||
/// Return the current vdim value
|
||||
inline int GetVDim() const { return vdim_; }
|
||||
|
||||
/// Set or change the nested DofTransformation object
|
||||
/// @note The face orientations in @a doftrans will be copied into the
|
||||
/// VDofTransformation object.
|
||||
void SetDofTransformation(DofTransformation & doftrans)
|
||||
inline void SetDofTransformation(DofTransformation & doftrans)
|
||||
{
|
||||
size_ = vdim_ * doftrans.Size();
|
||||
doftrans_ = &doftrans;
|
||||
StatelessVDofTransformation::SetDofTransformation(doftrans);
|
||||
DofTransformation::SetFaceOrientations(doftrans.GetFaceOrientations());
|
||||
}
|
||||
|
||||
/// Return the nested DofTransformation object
|
||||
inline DofTransformation * GetDofTransformation() const { return doftrans_; }
|
||||
|
||||
/// Set new face orientations in both the VDofTransformation and the
|
||||
/// DofTransformation contained within (if there is one).
|
||||
inline void SetFaceOrientations(const Array<int> & face_orientation)
|
||||
{
|
||||
DofTransformation::SetFaceOrientations(face_orientation);
|
||||
if (doftrans_) { doftrans_->SetFaceOrientations(face_orientation); }
|
||||
}
|
||||
inline void SetFaceOrientation(const Array<int> & face_orientation)
|
||||
{ Fo = face_orientation; doftrans_->SetFaceOrientations(face_orientation); }
|
||||
|
||||
using DofTransformation::TransformPrimal;
|
||||
using DofTransformation::InvTransformPrimal;
|
||||
using DofTransformation::TransformDual;
|
||||
using DofTransformation::InvTransformDual;
|
||||
|
||||
inline void TransformPrimal(double *v) const
|
||||
{ TransformPrimal(Fo, v); }
|
||||
inline void InvTransformPrimal(double *v) const
|
||||
{ InvTransformPrimal(Fo, v); }
|
||||
inline void TransformDual(double *v) const
|
||||
{ TransformDual(Fo, v); }
|
||||
inline void InvTransformDual(double *v) const
|
||||
{ InvTransformDual(Fo, v); }
|
||||
void TransformPrimal(double *v) const;
|
||||
void InvTransformPrimal(double *v) const;
|
||||
void TransformDual(double *v) const;
|
||||
void InvTransformDual(double *v) const;
|
||||
};
|
||||
|
||||
/** Abstract base class for high-order Nedelec spaces on elements with
|
||||
@@ -396,22 +207,17 @@ public:
|
||||
be accessed as DenseMatrices using the GetFaceTransform() and
|
||||
GetFaceInverseTransform() methods.
|
||||
*/
|
||||
class ND_StatelessDofTransformation : virtual public StatelessDofTransformation
|
||||
class ND_DofTransformation : public DofTransformation
|
||||
{
|
||||
private:
|
||||
protected:
|
||||
static const double T_data[24];
|
||||
static const double TInv_data[24];
|
||||
static const DenseTensor T, TInv;
|
||||
int order;
|
||||
int nedofs; // number of DoFs per edge
|
||||
int nfdofs; // number of DoFs per face
|
||||
|
||||
protected:
|
||||
const int order; // basis function order
|
||||
const int nedofs; // number of DoFs per edge
|
||||
const int nfdofs; // number of DoFs per face
|
||||
const int nedges; // number of edges per element
|
||||
const int nfaces; // number of triangular faces per element
|
||||
|
||||
ND_StatelessDofTransformation(int size, int order,
|
||||
int num_edges, int num_tri_faces);
|
||||
ND_DofTransformation(int size, int order);
|
||||
|
||||
public:
|
||||
// Return the 2x2 transformation operator for the given face orientation
|
||||
@@ -420,117 +226,67 @@ public:
|
||||
// Return the 2x2 inverse transformation operator
|
||||
static const DenseMatrix & GetFaceInverseTransform(int ori)
|
||||
{ return TInv(ori); }
|
||||
|
||||
void TransformPrimal(const Array<int> & face_orientation,
|
||||
double *v) const;
|
||||
|
||||
void InvTransformPrimal(const Array<int> & face_orientation,
|
||||
double *v) const;
|
||||
|
||||
void TransformDual(const Array<int> & face_orientation,
|
||||
double *v) const;
|
||||
|
||||
void InvTransformDual(const Array<int> & face_orientation,
|
||||
double *v) const;
|
||||
};
|
||||
|
||||
/// Stateless DoF transformation implementation for the Nedelec basis on
|
||||
/// triangles
|
||||
class ND_TriStatelessDofTransformation : public ND_StatelessDofTransformation
|
||||
{
|
||||
public:
|
||||
ND_TriStatelessDofTransformation(int order)
|
||||
: StatelessDofTransformation(order*(order + 2))
|
||||
, ND_StatelessDofTransformation(order*(order + 2), order, 3, 1)
|
||||
{}
|
||||
};
|
||||
|
||||
/// DoF transformation implementation for the Nedelec basis on triangles
|
||||
class ND_TriDofTransformation : public DofTransformation,
|
||||
public ND_TriStatelessDofTransformation
|
||||
class ND_TriDofTransformation : public ND_DofTransformation
|
||||
{
|
||||
public:
|
||||
ND_TriDofTransformation(int order)
|
||||
: StatelessDofTransformation(order*(order + 2))
|
||||
, DofTransformation(order*(order + 2))
|
||||
, ND_TriStatelessDofTransformation(order)
|
||||
{}
|
||||
ND_TriDofTransformation(int order);
|
||||
|
||||
using DofTransformation::TransformPrimal;
|
||||
using DofTransformation::InvTransformPrimal;
|
||||
using DofTransformation::TransformDual;
|
||||
using DofTransformation::InvTransformDual;
|
||||
|
||||
using ND_TriStatelessDofTransformation::TransformPrimal;
|
||||
using ND_TriStatelessDofTransformation::InvTransformPrimal;
|
||||
using ND_TriStatelessDofTransformation::TransformDual;
|
||||
using ND_TriStatelessDofTransformation::InvTransformDual;
|
||||
void TransformPrimal(double *v) const;
|
||||
|
||||
void InvTransformPrimal(double *v) const;
|
||||
|
||||
void TransformDual(double *v) const;
|
||||
|
||||
void InvTransformDual(double *v) const;
|
||||
using DofTransformation::InvTransformDual;
|
||||
};
|
||||
|
||||
/// DoF transformation implementation for the Nedelec basis on tetrahedra
|
||||
class ND_TetStatelessDofTransformation : public ND_StatelessDofTransformation
|
||||
class ND_TetDofTransformation : public ND_DofTransformation
|
||||
{
|
||||
public:
|
||||
ND_TetStatelessDofTransformation(int order)
|
||||
: StatelessDofTransformation(order*(order + 2)*(order + 3)/2)
|
||||
, ND_StatelessDofTransformation(order*(order + 2)*(order + 3)/2, order,
|
||||
6, 4)
|
||||
{}
|
||||
};
|
||||
|
||||
/// DoF transformation implementation for the Nedelec basis on tetrahedra
|
||||
class ND_TetDofTransformation : public DofTransformation,
|
||||
public ND_TetStatelessDofTransformation
|
||||
{
|
||||
public:
|
||||
ND_TetDofTransformation(int order)
|
||||
: StatelessDofTransformation(order*(order + 2)*(order + 3)/2)
|
||||
, DofTransformation(order*(order + 2)*(order + 3)/2)
|
||||
, ND_TetStatelessDofTransformation(order)
|
||||
{}
|
||||
ND_TetDofTransformation(int order);
|
||||
|
||||
using DofTransformation::TransformPrimal;
|
||||
using DofTransformation::InvTransformPrimal;
|
||||
using DofTransformation::TransformDual;
|
||||
using DofTransformation::InvTransformDual;
|
||||
|
||||
using ND_TetStatelessDofTransformation::TransformPrimal;
|
||||
using ND_TetStatelessDofTransformation::InvTransformPrimal;
|
||||
using ND_TetStatelessDofTransformation::TransformDual;
|
||||
using ND_TetStatelessDofTransformation::InvTransformDual;
|
||||
void TransformPrimal(double *v) const;
|
||||
|
||||
void InvTransformPrimal(double *v) const;
|
||||
|
||||
void TransformDual(double *v) const;
|
||||
|
||||
void InvTransformDual(double *v) const;
|
||||
};
|
||||
|
||||
/// DoF transformation implementation for the Nedelec basis on wedge elements
|
||||
class ND_WedgeStatelessDofTransformation : public ND_StatelessDofTransformation
|
||||
class ND_WedgeDofTransformation : public ND_DofTransformation
|
||||
{
|
||||
public:
|
||||
ND_WedgeStatelessDofTransformation(int order)
|
||||
: StatelessDofTransformation(3 * order * ((order + 1) * (order + 2))/2)
|
||||
, ND_StatelessDofTransformation(3 * order * ((order + 1) * (order + 2))/2,
|
||||
order, 9, 2)
|
||||
{}
|
||||
};
|
||||
|
||||
/// DoF transformation implementation for the Nedelec basis on wedge elements
|
||||
class ND_WedgeDofTransformation : public DofTransformation,
|
||||
public ND_WedgeStatelessDofTransformation
|
||||
{
|
||||
public:
|
||||
ND_WedgeDofTransformation(int order)
|
||||
: StatelessDofTransformation(3 * order * ((order + 1) * (order + 2))/2)
|
||||
, DofTransformation(3 * order * ((order + 1) * (order + 2))/2)
|
||||
, ND_WedgeStatelessDofTransformation(order)
|
||||
{}
|
||||
ND_WedgeDofTransformation(int order);
|
||||
|
||||
using DofTransformation::TransformPrimal;
|
||||
using DofTransformation::InvTransformPrimal;
|
||||
using DofTransformation::TransformDual;
|
||||
using DofTransformation::InvTransformDual;
|
||||
|
||||
using ND_WedgeStatelessDofTransformation::TransformPrimal;
|
||||
using ND_WedgeStatelessDofTransformation::InvTransformPrimal;
|
||||
using ND_WedgeStatelessDofTransformation::TransformDual;
|
||||
using ND_WedgeStatelessDofTransformation::InvTransformDual;
|
||||
void TransformPrimal(double *v) const;
|
||||
|
||||
void InvTransformPrimal(double *v) const;
|
||||
|
||||
void TransformDual(double *v) const;
|
||||
|
||||
void InvTransformDual(double *v) const;
|
||||
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -1,115 +0,0 @@
|
||||
// Copyright (c) 2010-2023, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
// Finite Element Base classes
|
||||
|
||||
#include "face_map_utils.hpp"
|
||||
#include <cmath> // std::pow
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace internal
|
||||
{
|
||||
|
||||
std::pair<int,int> GetFaceNormal3D(const int face_id)
|
||||
{
|
||||
switch (face_id)
|
||||
{
|
||||
case 0: return std::make_pair(2, 0); // z = 0
|
||||
case 1: return std::make_pair(1, 0); // y = 0
|
||||
case 2: return std::make_pair(0, 1); // x = 1
|
||||
case 3: return std::make_pair(1, 1); // y = 1
|
||||
case 4: return std::make_pair(0, 0); // x = 0
|
||||
case 5: return std::make_pair(2, 1); // z = 1
|
||||
default: MFEM_ABORT("Invalid face ID.")
|
||||
}
|
||||
return std::make_pair(-1, -1); // invalid
|
||||
}
|
||||
|
||||
void FillFaceMap(const int n_face_dofs_per_component,
|
||||
const std::vector<int> &offsets,
|
||||
const std::vector<int> &strides,
|
||||
const std::vector<int> &n_dofs_per_dim,
|
||||
Array<int> &face_map)
|
||||
{
|
||||
const int n_components = offsets.size();
|
||||
const int face_dim = strides.size() / n_components;
|
||||
for (int comp = 0; comp < n_components; ++comp)
|
||||
{
|
||||
const int offset = offsets[comp];
|
||||
for (int i = 0; i < n_face_dofs_per_component; ++i)
|
||||
{
|
||||
int idx = offset;
|
||||
int j = i;
|
||||
for (int d = 0; d < face_dim; ++d)
|
||||
{
|
||||
const int dof1d = n_dofs_per_dim[comp*(face_dim) + d];
|
||||
idx += strides[comp*(face_dim) + d]*(j % dof1d);
|
||||
j /= dof1d;
|
||||
}
|
||||
face_map[comp*n_face_dofs_per_component + i] = idx;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void GetTensorFaceMap(const int dim, const int order, const int face_id,
|
||||
Array<int> &face_map)
|
||||
{
|
||||
const int dof1d = order + 1;
|
||||
int n_face_dofs = int(std::pow(dof1d, dim - 1));
|
||||
std::vector<int> offsets, strides;
|
||||
switch (dim)
|
||||
{
|
||||
case 1:
|
||||
offsets = {(face_id == 0) ? 0 : dof1d - 1};
|
||||
break;
|
||||
case 2:
|
||||
strides = {(face_id == 0 || face_id == 2) ? 1 : dof1d};
|
||||
switch (face_id)
|
||||
{
|
||||
case 0: offsets = {0}; break; // y = 0
|
||||
case 1: offsets = {dof1d - 1}; break; // x = 1
|
||||
case 2: offsets = {(dof1d-1)*dof1d}; break; // y = 1
|
||||
case 3: offsets = {0}; break; // x = 0
|
||||
}
|
||||
break;
|
||||
case 3:
|
||||
{
|
||||
const auto f = GetFaceNormal3D(face_id);
|
||||
const int face_normal = f.first, level = f.second;
|
||||
if (face_normal == 0) // x-normal
|
||||
{
|
||||
offsets = {level ? dof1d-1 : 0};
|
||||
strides = {dof1d, dof1d*dof1d};
|
||||
}
|
||||
else if (face_normal == 1) // y-normal
|
||||
{
|
||||
offsets = {level ? (dof1d-1)*dof1d : 0};
|
||||
strides = {1, dof1d*dof1d};
|
||||
}
|
||||
else if (face_normal == 2) // z-normal
|
||||
{
|
||||
offsets = {level ? (dof1d-1)*dof1d*dof1d : 0};
|
||||
strides = {1, dof1d};
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// same number of DOFs in each dimension, repeat dof1d (dim - 1) times
|
||||
std::vector<int> n_dofs(dim - 1, dof1d);
|
||||
FillFaceMap(n_face_dofs, offsets, strides, n_dofs, face_map);
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
|
||||
} // namespace mfem
|
||||
@@ -1,58 +0,0 @@
|
||||
// Copyright (c) 2010-2023, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#ifndef MFEM_FACE_MAP_UTILS_HPP
|
||||
#define MFEM_FACE_MAP_UTILS_HPP
|
||||
|
||||
#include "../../general/array.hpp"
|
||||
#include <utility> // std::pair
|
||||
#include <vector>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace internal
|
||||
{
|
||||
|
||||
/// Each face of a hexahedron is given by a level set x_i = l, where x_i is one
|
||||
/// of x, y, or z (corresponding to i=0, i=1, i=2), and l is either 0 or 1.
|
||||
/// Returns i and level.
|
||||
std::pair<int,int> GetFaceNormal3D(const int face_id);
|
||||
|
||||
/// @brief Fills in the entries of the lexicographic face_map.
|
||||
///
|
||||
/// For use in FiniteElement::GetFaceMap.
|
||||
///
|
||||
/// n_face_dofs_per_component is the number of DOFs for each vector component
|
||||
/// on the face (there is only one vector component in all cases except for 3D
|
||||
/// Nedelec elements, where the face DOFs have two components to span the
|
||||
/// tangent space).
|
||||
///
|
||||
/// The DOFs for the i-th vector component begin at offsets[i] (i.e. the number
|
||||
/// of vector components is given by offsets.size()).
|
||||
///
|
||||
/// The DOFs for each vector component are arranged in a Cartesian grid defined
|
||||
/// by strides and n_dofs_per_dim.
|
||||
void FillFaceMap(const int n_face_dofs_per_component,
|
||||
const std::vector<int> &offsets,
|
||||
const std::vector<int> &strides,
|
||||
const std::vector<int> &n_dofs_per_dim,
|
||||
Array<int> &face_map);
|
||||
|
||||
/// Return the face map for nodal tensor elements (H1, L2, and Bernstein basis).
|
||||
void GetTensorFaceMap(const int dim, const int order, const int face_id,
|
||||
Array<int> &face_map);
|
||||
|
||||
} // namespace internal
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
+3
-21
@@ -12,7 +12,6 @@
|
||||
// Finite Element Base classes
|
||||
|
||||
#include "fe_base.hpp"
|
||||
#include "face_map_utils.hpp"
|
||||
#include "../coefficient.hpp"
|
||||
|
||||
namespace mfem
|
||||
@@ -401,7 +400,7 @@ const DofToQuad &FiniteElement::GetDofToQuad(const IntegrationRule &ir,
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (range_type == VECTOR)
|
||||
else
|
||||
{
|
||||
d2q->B.SetSize(nqpt*dim*dof);
|
||||
d2q->Bt.SetSize(dof*nqpt*dim);
|
||||
@@ -419,10 +418,6 @@ const DofToQuad &FiniteElement::GetDofToQuad(const IntegrationRule &ir,
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// Skip B and Bt for unknown range type
|
||||
}
|
||||
switch (deriv_type)
|
||||
{
|
||||
case GRAD:
|
||||
@@ -476,7 +471,7 @@ const DofToQuad &FiniteElement::GetDofToQuad(const IntegrationRule &ir,
|
||||
{
|
||||
for (int j = 0; j < dof; j++)
|
||||
{
|
||||
d2q->G[i+nqpt*(d+cdim*j)] = d2q->Gt[j+dof*(i+nqpt*d)] = curlshape(j, d);
|
||||
d2q->G[i+nqpt*(d+dim*j)] = d2q->Gt[j+dof*(i+nqpt*d)] = curlshape(j, d);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -484,19 +479,12 @@ const DofToQuad &FiniteElement::GetDofToQuad(const IntegrationRule &ir,
|
||||
}
|
||||
case NONE:
|
||||
default:
|
||||
// Skip G and Gt for unknown derivative type
|
||||
break;
|
||||
MFEM_ABORT("invalid finite element derivative type");
|
||||
}
|
||||
dof2quad_array.Append(d2q);
|
||||
return *d2q;
|
||||
}
|
||||
|
||||
void FiniteElement::GetFaceMap(const int face_id,
|
||||
Array<int> &face_map) const
|
||||
{
|
||||
MFEM_ABORT("method is not implemented for this element");
|
||||
}
|
||||
|
||||
FiniteElement::~FiniteElement()
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
@@ -2521,12 +2509,6 @@ void NodalTensorFiniteElement::SetMapType(const int map_type)
|
||||
}
|
||||
}
|
||||
|
||||
void NodalTensorFiniteElement::GetFaceMap(const int face_id,
|
||||
Array<int> &face_map) const
|
||||
{
|
||||
internal::GetTensorFaceMap(dim, order, face_id, face_map);
|
||||
}
|
||||
|
||||
VectorTensorFiniteElement::VectorTensorFiniteElement(const int dims,
|
||||
const int d,
|
||||
const int p,
|
||||
|
||||
+3
-27
@@ -14,7 +14,6 @@
|
||||
|
||||
#include "../intrules.hpp"
|
||||
#include "../geom.hpp"
|
||||
#include "../doftrans.hpp"
|
||||
|
||||
#include <map>
|
||||
|
||||
@@ -577,27 +576,6 @@ public:
|
||||
virtual const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode) const;
|
||||
|
||||
|
||||
/** @brief Return the mapping from lexicographic face DOFs to lexicographic
|
||||
element DOFs for the given local face @a face_id. */
|
||||
/** Given the @a ith DOF (lexicographically ordered) on the face referenced
|
||||
by @a face_id, face_map[i] gives the corresponding index of the DOF in
|
||||
the element (also lexicographically ordered).
|
||||
|
||||
@note For L2 spaces, this is only well-defined for "closed" bases such as
|
||||
the Gauss-Lobatto or Bernstein (positive) bases.
|
||||
|
||||
@warning GetFaceMap() is currently only implemented for tensor-product
|
||||
(quadrilateral and hexahedral) elements. Its functionality may change
|
||||
when simplex elements are supported in the future. */
|
||||
virtual void GetFaceMap(const int face_id, Array<int> &face_map) const;
|
||||
|
||||
/** @brief Return a DoF transformation object for this particular type of
|
||||
basis.
|
||||
*/
|
||||
virtual StatelessDofTransformation * GetDofTransformation() const
|
||||
{ return NULL; }
|
||||
|
||||
/// Deconstruct the FiniteElement
|
||||
virtual ~FiniteElement();
|
||||
|
||||
@@ -1270,8 +1248,6 @@ public:
|
||||
NodalFiniteElement::GetTransferMatrix(fe, Trans, I);
|
||||
}
|
||||
}
|
||||
|
||||
void GetFaceMap(const int face_id, Array<int> &face_map) const override;
|
||||
};
|
||||
|
||||
class VectorTensorFiniteElement : public VectorFiniteElement,
|
||||
@@ -1296,9 +1272,9 @@ public:
|
||||
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode) const override
|
||||
{
|
||||
return (mode == DofToQuad::FULL) ?
|
||||
FiniteElement::GetDofToQuad(ir, mode) :
|
||||
GetTensorDofToQuad(*this, ir, mode, basis1d, true, dof2quad_array);
|
||||
MFEM_VERIFY(mode != DofToQuad::FULL, "invalid mode requested");
|
||||
return GetTensorDofToQuad(*this, ir, mode, basis1d, true,
|
||||
dof2quad_array);
|
||||
}
|
||||
|
||||
const DofToQuad &GetDofToQuadOpen(const IntegrationRule &ir,
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user