Compare commits

..
181 changed files with 13532 additions and 30446 deletions
+1 -3
View File
@@ -15,9 +15,7 @@ install:
- msmpisdk.msi /passive
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
# Install METIS, use a mirror because the original source server is not always
# up. Original url:
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz
# Install METIS
- ps: Start-FileDownload 'https://mfem.github.io/tpls/metis-5.1.0.tar.gz'
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
- cd metis-5.1.0
+1 -8
View File
@@ -50,6 +50,7 @@ examples/ex1[04-9]
examples/ex1[0-9]p
examples/ex2[0-9]
examples/ex2[0-9]p
examples/ex25-gpu
examples/refined.mesh
examples/displaced.mesh
@@ -175,7 +176,6 @@ miniapps/meshing/mesh-optimizer
miniapps/meshing/pmesh-optimizer
miniapps/meshing/minimal-surface
miniapps/meshing/pminimal-surface
miniapps/meshing/polar-nc
miniapps/meshing/mobius-strip.mesh
miniapps/meshing/klein-bottle.mesh
@@ -188,7 +188,6 @@ miniapps/meshing/extruder.mesh
miniapps/meshing/trimmer.mesh
miniapps/meshing/optimized*
miniapps/meshing/perturbed*
miniapps/meshing/polar-nc.mesh
miniapps/performance/ex1
miniapps/performance/ex1p
@@ -235,7 +234,6 @@ miniapps/nurbs/mode_*
miniapps/nurbs/Example1*
miniapps/gslib/field-diff
miniapps/gslib/field-interp
miniapps/gslib/findpts
miniapps/gslib/pfindpts
@@ -262,10 +260,5 @@ tests/scripts/*.err
tests/scripts/*.out
tests/scripts/*.msg
# Other tests
tests/convergence/rates
tests/convergence/prates
tests/par-mesh-format/ex1p
# VPATH builds
build-*/*
+1 -17
View File
@@ -71,8 +71,6 @@ stages:
- build
- test
- deallocate
- lassen_build
- lassen_test
- baseline_check
- baseline_publish
@@ -81,11 +79,7 @@ stages:
# TODO: updating tests and tpls is not necessary anymore since pipelines are
# now using unique directories so repo are never shared with another pipeline.
# This is not memory efficient (we keep a lot of data), hence this reminder.
# Setup
setup:
tags:
- shell
- quartz
.setup:
stage: setup
variables:
GIT_STRATEGY: none
@@ -106,15 +100,6 @@ setup:
before_script:
- module load gcc/6.1.0
# On lassen
.with_gcc_8_3_1:
variables:
TOOLCHAIN: gcc_8_3_1
CXX: g++
CC: gcc
before_script:
- module load gcc/8.3.1
.with_gcc_4_9_3:
variables:
TOOLCHAIN: gcc_4_9_3
@@ -305,4 +290,3 @@ setup:
# The list on jobs is defined in machine-specific files.
include:
- local: .gitlab/quartz.yml
- local: .gitlab/lassen.yml
-57
View File
@@ -1,57 +0,0 @@
# Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# GitLab pipelines configurations for the Lassen machine at LLNL
.on_lassen:
tags:
- shell
- lassen
variables:
PLAT: lassen
# Build MFEM
build_mfem_ser_lassen:
extends: [.with_gcc_8_3_1, .on_lassen]
needs: [setup]
stage: lassen_build
script:
- mkdir -p ${BUILD_PATH}
- cp -r ${CI_PROJECT_DIR} ${BUILD_PATH}/${CI_PROJECT_NAME}_lassen_ser
- cd ${BUILD_PATH}/${CI_PROJECT_NAME}_lassen_ser
- lalloc 1 -W 5 -q pdebug make -j cuda CUDA_ARCH=sm_70
build_mfem_debug_ser_lassen:
extends: [.with_gcc_8_3_1, .on_lassen]
needs: [setup]
stage: lassen_build
script:
- mkdir -p ${BUILD_PATH}
- cp -r ${CI_PROJECT_DIR} ${BUILD_PATH}/${CI_PROJECT_NAME}_lassen_ser_debug
- cd ${BUILD_PATH}/${CI_PROJECT_NAME}_lassen_ser_debug
- lalloc 1 -W 5 -q pdebug make -j cuda MFEM_DEBUG="YES" CUDA_ARCH=sm_70
# Sanity check
sanitycheck_mfem_ser_lassen:
extends: [.with_gcc_8_3_1, .on_lassen]
stage: lassen_test
needs: [build_mfem_ser_lassen]
script:
- cd ${BUILD_PATH}/${CI_PROJECT_NAME}_lassen_ser
- lalloc 1 -W 15 -q pdebug make -j test
sanitycheck_mfem_debug_ser_lassen:
extends: [.with_gcc_8_3_1, .on_lassen]
stage: lassen_test
needs: [build_mfem_debug_ser_lassen]
script:
- cd ${BUILD_PATH}/${CI_PROJECT_NAME}_lassen_ser_debug
- lalloc 1 -W 30 -q pdebug make -j test
+4
View File
@@ -22,6 +22,10 @@
MAKE_PAR: 6
BASELINE_PAR: 18
# Setup
setup_quartz:
extends: [.setup, .on_quartz]
# Allocate
allocate_quartz:
variables:
+13 -77
View File
@@ -11,9 +11,6 @@
language: cpp
os: linux
dist: bionic
stages:
- checks
- tests
@@ -37,7 +34,6 @@ jobs:
- stage: checks
os: linux
dist: xenial
name: "code-style"
addons:
apt:
@@ -56,6 +52,9 @@ jobs:
packages:
- doxygen
- graphviz
- mpich
- libmpich-dev
env: MPI=YES
script:
- cd ${TRAVIS_BUILD_DIR}
- cd tests/scripts
@@ -70,24 +69,13 @@ jobs:
- mpich
- libmpich-dev
env: MPI=YES
before_script:
script:
- cd ${TRAVIS_BUILD_DIR}
- mpicxx -v
- make config MFEM_USE_MPI=YES MFEM_MPI_NP=2
- make all -j3
- make test-noclean
script:
- cd tests/scripts
- ./runtest gitignore
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
before_cache:
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
mv libmetis.a Lib ..; rm -rf * ; mv ../libmetis.a ../Lib .;
rm -f Lib/*.{c,o}
# ========================
# Optional Checks/Tests
@@ -96,7 +84,6 @@ jobs:
- stage: optional
name: "branch-history"
if: branch != next
# need full git history for the binary/big files check
git:
depth: false
@@ -125,8 +112,6 @@ jobs:
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
cache:
ccache: true
- os: linux
compiler: gcc
@@ -135,8 +120,6 @@ jobs:
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
cache:
ccache: true
- os: linux
compiler: gcc
@@ -160,7 +143,6 @@ jobs:
MFEM_TEST_TARGET=check
NPROCS=2
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
@@ -191,7 +173,6 @@ jobs:
MFEM_TEST_TARGET=test
NPROCS=2
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
@@ -223,7 +204,6 @@ jobs:
- make -j3
- ctest --output-on-failure
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
@@ -241,43 +221,27 @@ jobs:
# - parallel
- os: osx
osx_image: xcode11.2
# osx_image: xcode7.3
compiler: clang
name: "Mac: Serial + Debug"
addons:
homebrew:
packages:
- ccache
env: DEBUG=YES
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=check
cache:
ccache: true
- os: osx
osx_image: xcode11.2
# osx_image: xcode7.3
compiler: clang
name: "Mac: Serial"
addons:
homebrew:
packages:
- ccache
env: DEBUG=NO
MPI=NO
CODECOV=NO
MFEM_TEST_TARGET=test
cache:
ccache: true
- os: osx
osx_image: xcode11.2
# osx_image: xcode7.3
compiler: clang
name: "Mac: Parallel + Debug"
addons:
homebrew:
packages:
- ccache
env: DEBUG=YES
MPI=YES
CODECOV=NO
@@ -285,7 +249,6 @@ jobs:
NPROCS=4
TMPDIR=/tmp
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
@@ -296,13 +259,9 @@ jobs:
rm -f Lib/*.{c,o}
- os: osx
osx_image: xcode11.2
# osx_image: xcode7.3
compiler: clang
name: "Mac: Parallel"
addons:
homebrew:
packages:
- ccache
env: DEBUG=NO
MPI=YES
CODECOV=YES
@@ -310,7 +269,6 @@ jobs:
NPROCS=4
TMPDIR=/tmp
cache:
ccache: true
directories:
- $TRAVIS_BUILD_DIR/../$HYPRE_TOP_DIR/src/hypre
- $TRAVIS_BUILD_DIR/../metis-4.0
@@ -326,19 +284,14 @@ before_install:
# brew install open-mpi;
# fi
# Disable ccache while building dependencies that are cached:
- echo "before \$PATH = $PATH";
export PATH=${PATH//\/usr\/lib\/ccache:/};
echo "after \$PATH = $PATH"
# On Mac OS X, build and cache OpenMPI 2.1.6:
# On Mac OS X, build and cache OpenMPI 2.1.1:
- if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
if [ ! -e $HOME/local-cached/bin/mpicc ]; then
mkdir -p $HOME/builds && cd $HOME/builds &&
wget https://download.open-mpi.org/release/open-mpi/v2.1/openmpi-2.1.6.tar.bz2 &&
tar jxf openmpi-2.1.6.tar.bz2 &&
wget https://www.open-mpi.org/software/ompi/v2.1/downloads/openmpi-2.1.1.tar.bz2 &&
tar jxf openmpi-2.1.1.tar.bz2 &&
mkdir openmpi-build && cd openmpi-build &&
../openmpi-2.1.6/configure --prefix=$HOME/local-cached &&
../openmpi-2.1.1/configure --prefix=$HOME/local-cached &&
make -j3 all && make install;
fi;
PATH=$HOME/local-cached/bin:$PATH;
@@ -399,9 +352,7 @@ install:
echo "Serial build, not using hypre";
fi
# METIS, use a mirror because the original source server is not always up.
# Original url:
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz
# METIS
- if [ $MPI == "YES" ]; then
if [ ! -e metis-4.0/libmetis.a ]; then
wget https://mfem.github.io/tpls/metis-4.0.3.tar.gz;
@@ -414,18 +365,6 @@ install:
fi;
fi
# Re-enable ccache on linux; enable ccache on mac os:
- if [ $TRAVIS_OS_NAME == "linux" ]; then
export PATH="/usr/lib/ccache:$PATH";
else
if [ $TRAVIS_OS_NAME == "osx" ]; then
export PATH="/usr/local/opt/ccache/libexec:$PATH";
fi;
fi
- printf "which \$CC = "; which $CC;
printf "which \$CXX = "; which $CXX
script:
# Compiler
- if [ $MPI == "YES" ]; then
@@ -446,9 +385,6 @@ script:
if [ "$CODECOV" == "YES" ]; then
CPPFLAGS="--coverage -g";
fi;
if [ "$TRAVIS_OS_NAME" != "linux" ] || [ "$DEBUG" == "YES" ]; then
CPPFLAGS+=" -pedantic -Wall -Werror";
fi
# Configure the library
- make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG $MAKE_CXX_FLAG
+14 -61
View File
@@ -16,12 +16,7 @@ Meshing improvements
- The graph linear ordering library Gecko, previously an external dependency, is
now included directly in MFEM. As a result, Mesh::GetGeckoElementOrdering is
always available. The interface has also been improved, see for example the
Mesh Explorer miniapp.
- Improved Gmsh reader (version 2.2), which now supports both high-order and
periodic meshes. Segments, triangles, quadrilaterals, and tetrahedra are
supported up to order 10. Wedges and hexahedra are supported up to order 9.
For sample periodic meshes, see the periodic*.msh files in the data directory.
mesh-explorer miniapp.
- Added support for finite difference-based gradient and Hessian approximation
in the TMOP mesh optimization algorithms. This improves the accuracy of the
@@ -32,17 +27,15 @@ Meshing improvements
the user to specify different discrete functions for controlling the
size, aspect-ratio, orientation, and skew of elements in the mesh.
- Added TMOP capability for approximate tangential mesh relaxation. Added
support and examples for using TMOP on mixed meshes.
- Added TMOP capability for approximate tangential mesh relaxation.
- Added support for reading periodic meshes in Gmsh format (version 2.2). See
for example the periodic-annulus-sector and periodic-torus-sector files in
the data directory.
- Added complete action of the TMOP Integrator to account for the spatial
derivatives of discrete and analytic targets.
- Added support for initialization of (serial) non-conforming meshes. Hanging
nodes can be marked with Mesh::AddVertexParents when building the mesh with
the "init" constructor. The usage is demonstrated in a new meshing miniapp
(polar-nc) which generates meshes that are non-conforming from the start.
Performance improvements
------------------------
- Added support for explicit vectorization in the high-performance templated
@@ -51,7 +44,7 @@ Performance improvements
- x86 (SSE/AVX/AVX2/AVX512),
- Power8 & Power9 (VSX),
- BG/Q (QPX).
These are disabled by default, and can be enabled with MFEM_USE_SIMD=YES.
These are now enabled by default, and can be disabled with MFEM_USE_SIMD=NO.
See the new file linalg/simd.hpp and the new directory linalg/simd.
Improved GPU capabilities
@@ -65,13 +58,11 @@ Improved GPU capabilities
compute a global sparse matrix. All integrators supported by element assembly
are also supported by full assembly. See the '-fa' option in Example 9.
- Added CUDA support for sparse matrix-vector multiplication with cuSPARSE.
- Added support for BlockOperator on GPU. See the updated Example 5.
- Added partial assembly and GPU support for complex operators, including the
classes ComplexOperator, [Par]ComplexGridFunction, [Par]ComplexLinearForm, and
[Par]SesquilinearForm. See the updated Example 22.
- Added partial assembly and GPU support for ComplexOperator,
[Par]ComplexGridFunction, [Par]ComplexLinearForm, and [Par]SesquilinearForm.
See the updated Example 22.
Discretization improvements
---------------------------
@@ -101,14 +92,6 @@ Discretization improvements
- Added support face integrals on the boundaries of NURBS meshes.
- Added support for interpolation of functions in L2, H(div) and H(curl)
spaces using GSLIB-FindPoints.
- Added support for computing asymptotic error estimates and convergence rates
for the whole de Rham sequence based on the new class ConvergenceStudy and new
member methods in GridFunction and ParGridFunction. See the rates.cpp file in
the tests/convergence directory for sample usage.
Linear and nonlinear solvers
----------------------------
- Added power method to iteratively estimate the largest eigenvalue and the
@@ -134,12 +117,6 @@ Linear and nonlinear solvers
- Added support for the SLEPc eigensolver package.
- Added partially assembled convergent diagonal preconditioner for adaptively
refined meshes (i.e. non-conforming finite element spaces), see Example 6/6p.
- Added an interface to the Intel MKL Parallel Direct Sparse Solver for
Clusters. An example usage of the interface is shown in Example 11p.
New and updated examples and miniapps
-------------------------------------
- Added a new example, Example 25/25p, to demonstrate the use of a Perfectly
@@ -176,12 +153,6 @@ New and updated examples and miniapps
- Added a new meshing miniapp, Minimal Surface, which solves Plateau's problem:
the Dirichlet problem for the minimal surface equation.
- Added a new meshing miniapp, Polar NC, which demonstrates the construction of
polar non-conforming meshes.
- Added partial assembly support to Example 4/4p and Example 5/5p, with diagonal
preconditioning.
- Added full assembly support in Example 9/9p.
- Added a new test problem in Example 24/24p, demonstrating a mixed bilinear
@@ -193,47 +164,29 @@ New and updated examples and miniapps
mesh based on element attributes. Any newly exposed boundary elements are
assigned attribute numbers related to the trimmed element attributes.
- Added a new miniapp (field-interp) that demonstrates transfer of grid function
between different meshes using GSLIB-FindPoints.
- Added diagonal preconditioner in Example 6/6p for partial assembly with AMR.
- Added device support in Example 5/5p.
- Added partial assembly and device support to Example 22/22p, with diagonal
preconditioning.
- Added the option to plot a function in Mesh Explorer.
- Added partial assembly and device support to Example 4/4p, Example 5/5p,
Example 22/22p, and Example 25/25p, with diagonal preconditioning.
Improved testing
----------------
- Upgraded the Catch unit test framework from version 1.6.1 to version 2.13.0.
- Added a GitLab pipeline that automates PR testing on supercomputing systems
and Linux clusters at Lawrence Livermore National Lab (LLNL). This can be
triggered only by LLNL developers, see .gitlab-ci.yml, the .gitlab directory
and the updated CONTRIBUTING.md file.
- Added testing of the parallel mesh format in tests/par-mesh-format.
Miscellaneous
-------------
- Added support for ADIOS2 for parallel I/O with ParaView visualization. The
classes adios2stream and ADIOS2DataCollection are introduced in mfem as the
interfaces to generate ADIOS2 Binary Pack (BP4) directory datasets for the
entire spatial and temporal node data. Cell centered data is accessible by
ADIOS2 data readers (e.g. Python), but currently not yet implement as of
ParaView v5.8.1. In addition, ADIOS2 allows for setting a user-defined number
of data substreams/subfiles at scale. See examples 5, 9, 12, 16.
entire spatial and temporal data. In addition, ADIOS2 allows for setting a
user-defined number of data substreams/subfiles. See examples 5, 9, 12, 16.
- The integration order used in the ComputeLpError and ComputeElementLpError
methods of class GridFunction has been increased.
- Various other simplifications, extensions, and bugfixes in the code.
- Renamed "Backend::DEBUG" to "Backend::DEBUG_DEVICE" to avoid conflicts,
as DEBUG is sometimes used as a macro.
Version 4.1, released on March 10, 2020
=======================================
+17 -38
View File
@@ -89,38 +89,8 @@ enable_language(CXX)
if (MFEM_USE_CUDA)
# MFEM_USE_CUDA requires CMake 3.8 or newer (for direct CUDA support)
cmake_minimum_required(VERSION 3.8 FATAL_ERROR)
# Use ${CMAKE_CXX_COMPILER} as the cuda host compiler.
if (NOT CMAKE_CUDA_HOST_COMPILER)
set(CMAKE_CUDA_HOST_COMPILER ${CMAKE_CXX_COMPILER})
endif()
enable_language(CUDA)
set(CMAKE_CUDA_STANDARD 11)
set(CMAKE_CUDA_STANDARD_REQUIRED ON)
set(CMAKE_CUDA_EXTENSIONS OFF)
set(CUDA_FLAGS "--expt-extended-lambda")
if (CMAKE_VERSION VERSION_LESS 3.18.0)
set(CUDA_FLAGS "-arch=${CUDA_ARCH} ${CUDA_FLAGS}")
elseif (NOT CMAKE_CUDA_ARCHITECTURES)
string(REGEX REPLACE "^sm_" "" ARCH_NUMBER "${CUDA_ARCH}")
if ("${CUDA_ARCH}" STREQUAL "sm_${ARCH_NUMBER}")
set(CMAKE_CUDA_ARCHITECTURES "${ARCH_NUMBER}")
else()
message(FATAL_ERROR "Unknown CUDA_ARCH: ${CUDA_ARCH}")
endif()
else()
set(CUDA_ARCH "CMAKE_CUDA_ARCHITECTURES: ${CMAKE_CUDA_ARCHITECTURES}")
endif()
message(STATUS "Using CUDA architecture: ${CUDA_ARCH}")
if (CMAKE_VERSION VERSION_LESS 3.12.0)
# CMake versions 3.8 and 3.9 require this to work; 3.10 and 3.11 are not
# tested and may not actually need this (but should be ok to keep).
set(CUDA_FLAGS "-ccbin=${CMAKE_CXX_COMPILER} ${CUDA_FLAGS}")
set(CMAKE_CUDA_HOST_LINK_LAUNCHER ${CMAKE_CXX_COMPILER})
endif()
set(CMAKE_CUDA_FLAGS "${CUDA_FLAGS}" CACHE STRING
"CUDA flags set for MFEM" FORCE)
set(CUSPARSE_FOUND TRUE)
set(CUSPARSE_LIBRARIES "cusparse")
endif()
if (XSDK_ENABLE_C)
@@ -326,6 +296,22 @@ if (MFEM_USE_HIOP)
# find_package updates HIOP_FOUND, HIOP_INCLUDE_DIRS, HIOP_LIBRARIES
endif()
# CUDA
if (MFEM_USE_CUDA)
set(CMAKE_CUDA_STANDARD 11)
set(CMAKE_CUDA_STANDARD_REQUIRED ON)
set(CMAKE_CUDA_EXTENSIONS OFF)
set(CMAKE_CUDA_FLAGS "-arch=${CUDA_ARCH} --expt-extended-lambda"
CACHE STRING "CUDA flags set for MFEM" FORCE)
if (MFEM_USE_MPI)
set(CUDA_CCBIN_COMPILER ${MPI_CXX_COMPILER})
else()
set(CUDA_CCBIN_COMPILER ${CMAKE_CXX_COMPILER})
endif()
string(APPEND CMAKE_CUDA_FLAGS " -ccbin ${CUDA_CCBIN_COMPILER}")
set(CMAKE_CUDA_HOST_LINK_LAUNCHER ${CUDA_CCBIN_COMPILER})
endif()
# OCCA
if (MFEM_USE_OCCA)
find_package(OCCA REQUIRED)
@@ -346,12 +332,6 @@ if (MFEM_USE_ADIOS2)
find_package(ADIOS2 REQUIRED)
endif()
if (MFEM_USE_MKL_CPARDISO)
if (MFEM_USE_MPI)
find_package(MKL_CPARDISO REQUIRED MKL_SEQUENTIAL MKL_LP64 MKL_MPI_WRAPPER)
endif()
endif()
# MFEM_TIMER_TYPE
if (NOT DEFINED MFEM_TIMER_TYPE)
if (APPLE)
@@ -377,8 +357,7 @@ endif()
# be before SuiteSparse.
set(MFEM_TPLS MPI_CXX OPENMP BLAS LAPACK METIS HYPRE SuiteSparse SUNDIALS PETSC
SLEPC MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT Ginkgo GNUTLS GSLIB NETCDF
MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE ADIOS2
CUSPARSE MKL_CPARDISO)
MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE ADIOS2)
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
set(TPL_LIBRARIES "")
set(TPL_INCLUDE_DIRS "")
+1 -8
View File
@@ -486,13 +486,6 @@ MFEM_USE_CEED = YES/NO
library for performant high-order operator evaluation developed by the Center
for Efficient Exascale Discretizations in the Exascale Computing Project.
MFEM_USE_MKL_CPARDISO = YES/NO
Enables the interface to the Intel MKL Parallel Direct Sparse Solver for
Clusters. Make sure to set the correct values for MKL_MPI_WRAPPER and
MKL_LIBRARY_SUBDIR as shown in defaults.mk. If you configure MFEM with
MFEM_USE_LAPACK=YES, verify that the MKL LAPACK libraries are used. The
OpenMP capabilities are disabled at link time.
MFEM_BUILD_TAG = (any value)
An optional tag to characterize the build. Exported to config/config.mk.
Can be used to identify the MFEM build from other makefiles.
@@ -670,7 +663,7 @@ The specific libraries and their options are:
URL: https://github.com/CEED/libCEED
https://ceed.exascaleproject.org/libceed
Options: CEED_DIR, CEED_OPT, CEED_LIB.
Versions: libCEED > 0.6, git-hash bdfed75.
Versions: libCEED >= 0.6, git-hash a970f63.
- RAJA (optional), used when MFEM_USE_RAJA = YES.
Beginning with MFEM v4.1, only RAJA v0.10.0+ is supported.
-3
View File
@@ -156,7 +156,4 @@
// library.
#cmakedefine MFEM_USE_SIMMETRIX
// Enable interface to the MKL CPardiso library.
#cmakedefine MFEM_USE_MKL_CPARDISO
#endif // MFEM_CONFIG_HEADER
+1 -13
View File
@@ -38,19 +38,7 @@ if(NOT ADIOS2_FOUND)
endif()
find_path(ADIOS2_INCLUDE_DIR adios2.h ${ADIOS2_INCLUDE_OPTS})
# adios2 version 2.5.0
find_library(ADIOS2_LIBRARY NAMES adios2 ${ADIOS2_LIBRARY_OPTS})
# adios2 version 2.6.0 and onwards
if(NOT ADIOS2_LIBRARY)
find_library(ADIOS2_CXX11_MPI_LIBRARY NAMES adios2_cxx11_mpi ${ADIOS2_LIBRARY_OPTS})
find_library(ADIOS2_CXX11_LIBRARY NAMES adios2_cxx11 ${ADIOS2_LIBRARY_OPTS})
set(ADIOS2_LIBRARY ${ADIOS2_CXX11_MPI_LIBRARY} ${ADIOS2_CXX11_LIBRARY})
if(MFEM_USE_MPI)
add_definitions(-DADIOS2_USE_MPI)
endif()
endif()
find_library(ADIOS2_LIBRARY NAMES adios2 ${ADIOS2_LIBRARY_OPTS})
include(FindPackageHandleStandardArgs)
find_package_handle_standard_args(ADIOS2
-106
View File
@@ -1,106 +0,0 @@
# Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Defines the following variables:
# - MKL_CPARDISO_FOUND
# - MKL_CPARDISO_LIBRARIES
# - MKL_CPARDISO_INCLUDE_DIRS
if(NOT MKL_MPI_WRAPPER_LIB)
message(FATAL_ERROR "MKL CPardiso enabled but no MKL MPI Wrapper lib specified")
endif()
if(NOT MKL_LIBRARY_DIR)
message(WARNING "Using default MKL library path. Double check the variable MKL_LIBRARY_DIR")
set(MKL_LIBRARY_DIR "lib")
endif()
include(MfemCmakeUtilities)
mfem_find_package(MKL_CPARDISO MKL_CPARDISO
MKL_CPARDISO_DIR "include" mkl_cluster_sparse_solver.h ${MKL_LIBRARY_DIR} mkl_core
"Paths to headers required by MKL CPardiso." "Libraries required by MKL CPARDISO."
ADD_COMPONENT MKL_LP64 "include" "" ${MKL_LIBRARY_DIR} mkl_intel_lp64
ADD_COMPONENT MKL_SEQUENTIAL "include" "" ${MKL_LIBRARY_DIR} mkl_sequential
ADD_COMPONENT MKL_MPI_WRAPPER "include" "" ${MKL_LIBRARY_DIR} ${MKL_MPI_WRAPPER_LIB}
CHECK_BUILD MKL_CPARDISO_VERSION_OK TRUE
"
#include <mpi.h>
#include <mkl.h>
#include <mkl_cluster_sparse_solver.h>
int main (void)
{
MKL_INT n = 5;
MKL_INT ia[6] = { 1, 4, 6, 9, 12, 14};
MKL_INT ja[13] = { 1, 2, 4, /* index of non-zeros in 1 row*/
1, 2, /* index of non-zeros in 2 row*/
3, 4, 5, /* index of non-zeros in 3 row*/
1, 3, 4, /* index of non-zeros in 4 row*/
2, 5 /* index of non-zeros in 5 row*/
};
double a[13] = {
1.0, -1.0, /*0*/ -3.0, /*0*/
-2.0, 5.0, /*0*/ /*0*/ /*0*/
/*0*/ 4.0, 6.0, 4.0, /*0*/
-4.0, /*0*/ 2.0, 7.0, /*0*/
/*0*/ 8.0, /*0*/ /*0*/ -5.0
};
MKL_INT mtype = 11; /* set matrix type to \"real unsymmetric matrix\" */
MKL_INT nrhs = 1; /* Number of right hand sides. */
double b[5], x[5], bs[5], res, res0; /* RHS and solution vectors. */
/* Internal solver memory pointer pt
* 32-bit: int pt[64] or void *pt[64];
* 64-bit: long int pt[64] or void *pt[64]; */
void *pt[64] = { 0 };
/* Cluster Sparse Solver control parameters. */
MKL_INT iparm[64] = { 0 };
MKL_INT maxfct, mnum, phase, msglvl, error;
/* Auxiliary variables. */
double ddum; /* Double dummy */
MKL_INT idum; /* Integer dummy. */
MKL_INT i, j;
int mpi_stat = 0;
int argc = 0;
int comm, rank;
char* uplo;
char** argv;
mpi_stat = MPI_Init( &argc, &argv );
mpi_stat = MPI_Comm_rank( MPI_COMM_WORLD, &rank );
comm = MPI_Comm_c2f( MPI_COMM_WORLD );
iparm[ 0] = 1; /* Solver default parameters overriden with provided by iparm */
iparm[ 1] = 2; /* Use METIS for fill-in reordering */
iparm[ 5] = 0; /* Write solution into x */
iparm[ 7] = 2; /* Max number of iterative refinement steps */
iparm[ 9] = 13; /* Perturb the pivot elements with 1E-13 */
iparm[10] = 1; /* Use nonsymmetric permutation and scaling MPS */
iparm[12] = 1; /* Switch on Maximum Weighted Matching algorithm (default for non-symmetric) */
iparm[17] = -1; /* Output: Number of nonzeros in the factor LU */
iparm[18] = -1; /* Output: Mflops for LU factorization */
iparm[26] = 1; /* Check input data for correctness */
iparm[39] = 0; /* Input: matrix/rhs/solution stored on master */
maxfct = 1; /* Maximum number of numerical factorizations. */
mnum = 1; /* Which factorization to use. */
msglvl = 1; /* Print statistical information in file */
error = 0; /* Initialize error flag */
phase = 11;
cluster_sparse_solver ( pt, &maxfct, &mnum, &mtype, &phase,
&n, a, ia, ja, &idum, &nrhs, iparm, &msglvl, &ddum, &ddum, &comm, &error );
mpi_stat = MPI_Finalize();
return error;
}
")
@@ -128,15 +128,7 @@ function(add_mfem_miniapp MFEM_EXE_NAME)
if (MFEM_USE_CUDA)
set_property(SOURCE ${MAIN_LIST} ${EXTRA_SOURCES_LIST}
PROPERTY LANGUAGE CUDA)
if (CMAKE_VERSION VERSION_GREATER_EQUAL 3.12.0)
list(TRANSFORM EXTRA_OPTIONS_LIST PREPEND "-Xcompiler=")
else()
set(LIST_)
foreach(item IN LISTS EXTRA_OPTIONS_LIST)
list(APPEND LIST_ "-Xcompiler=${item}")
endforeach()
set(EXTRA_OPTIONS_LIST ${LIST_})
endif()
list(TRANSFORM EXTRA_OPTIONS_LIST PREPEND "-Xcompiler=")
endif()
# Actually add the executable
-6
View File
@@ -42,15 +42,9 @@
#ifdef MFEM_USE_SUPERLU
#error Building with SuperLU_DIST (MFEM_USE_SUPERLU=YES) requires MPI (MFEM_USE_MPI=YES)
#endif
#ifdef MFEM_USE_MUMPS
#error Building with MUMPS (MFEM_USE_MUMPS=YES) requires MPI (MFEM_USE_MPI=YES)
#endif
#ifdef MFEM_USE_STRUMPACK
#error Building with STRUMPACK (MFEM_USE_STRUMPACK=YES) requires MPI (MFEM_USE_MPI=YES)
#endif
#ifdef MFEM_USE_MKL_CPARDISO
#error Building with MKL CPARDISO (MFEM_USE_MKL_CPARDISO=YES) requires MPI (MFEM_USE_MPI=YES)
#endif
#ifdef MFEM_USE_PETSC
#error Building with PETSc (MFEM_USE_PETSC=YES) requires MPI (MFEM_USE_MPI=YES)
#endif
-7
View File
@@ -94,10 +94,6 @@
// Enable MFEM functionality based on the SuperLU library.
// #define MFEM_USE_SUPERLU
// Enable MFEM functionality based on the MUMPS library.
// #define MFEM_USE_MUMPS
// #define MFEM_MUMPS_VERSION @MFEM_MUMPS_VERSION@
// Enable MFEM functionality based on the STRUMPACK library.
// #define MFEM_USE_STRUMPACK
@@ -167,7 +163,4 @@
// library.
// #define MFEM_USE_SIMMETRIX
// Enable interface to the MKL CPardiso library.
// #define MFEM_USE_MKL_CPARDISO
#endif // MFEM_CONFIG_HEADER
-2
View File
@@ -32,7 +32,6 @@ MFEM_USE_SUNDIALS = @MFEM_USE_SUNDIALS@
MFEM_USE_MESQUITE = @MFEM_USE_MESQUITE@
MFEM_USE_SUITESPARSE = @MFEM_USE_SUITESPARSE@
MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
MFEM_USE_MUMPS = @MFEM_USE_MUMPS@
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
MFEM_USE_GNUTLS = @MFEM_USE_GNUTLS@
@@ -53,7 +52,6 @@ MFEM_USE_CEED = @MFEM_USE_CEED@
MFEM_USE_UMPIRE = @MFEM_USE_UMPIRE@
MFEM_USE_SIMD = @MFEM_USE_SIMD@
MFEM_USE_ADIOS2 = @MFEM_USE_ADIOS2@
MFEM_USE_MKL_CPARDISO = @MFEM_USE_MKL_CPARDISO@
# Compiler, compile options, and link options
MFEM_CXX = @MFEM_CXX@
+1 -6
View File
@@ -50,9 +50,8 @@ option(MFEM_USE_OCCA "Enable OCCA" OFF)
option(MFEM_USE_RAJA "Enable RAJA" OFF)
option(MFEM_USE_CEED "Enable CEED" OFF)
option(MFEM_USE_UMPIRE "Enable Umpire" OFF)
option(MFEM_USE_SIMD "Enable use of SIMD intrinsics" OFF)
option(MFEM_USE_SIMD "Enable use of SIMD intrinsics" ON)
option(MFEM_USE_ADIOS2 "Enable ADIOS2" OFF)
option(MFEM_USE_MKL_CPARDISO "Enable MKL CPardiso" OFF)
set(MFEM_MPI_NP 4 CACHE STRING "Number of processes used for MPI tests")
@@ -181,10 +180,6 @@ set(HIOP_DIR "${MFEM_DIR}/../hiop/install" CACHE STRING
set(HIOP_REQUIRED_PACKAGES "BLAS" "LAPACK" CACHE STRING
"Packages that HiOp depends on.")
set(MKL_CPARDISO_DIR "" CACHE STRING "MKL installation path.")
set(MKL_MPI_WRAPPER_LIB "mkl_blacs_mpich_lp64" CACHE STRING "MKL MPI wrapper library")
set(MKL_LIBRARY_DIR "" CACHE STRING "Custom library subdirectory")
set(OCCA_DIR "${MFEM_DIR}/../occa" CACHE PATH "Path to OCCA")
set(RAJA_DIR "${MFEM_DIR}/../raja" CACHE PATH "Path to RAJA")
set(CEED_DIR "${MFEM_DIR}/../libCEED" CACHE PATH "Path to libCEED")
+5 -21
View File
@@ -120,7 +120,6 @@ MFEM_USE_SUNDIALS = NO
MFEM_USE_MESQUITE = NO
MFEM_USE_SUITESPARSE = NO
MFEM_USE_SUPERLU = NO
MFEM_USE_MUMPS = NO
MFEM_USE_STRUMPACK = NO
MFEM_USE_GINKGO = NO
MFEM_USE_GNUTLS = NO
@@ -139,9 +138,8 @@ MFEM_USE_RAJA = NO
MFEM_USE_OCCA = NO
MFEM_USE_CEED = NO
MFEM_USE_UMPIRE = NO
MFEM_USE_SIMD = NO
MFEM_USE_SIMD = YES
MFEM_USE_ADIOS2 = NO
MFEM_USE_MKL_CPARDISO = NO
# Compile and link options for zlib.
ZLIB_DIR =
@@ -158,7 +156,7 @@ HYPRE_OPT = -I$(HYPRE_DIR)/include
HYPRE_LIB = -L$(HYPRE_DIR)/lib -lHYPRE
# METIS library configuration
ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK)$(MFEM_USE_MUMPS),NONONO)
ifeq ($(MFEM_USE_SUPERLU)$(MFEM_USE_STRUMPACK),NONO)
ifeq ($(MFEM_USE_METIS_5),NO)
METIS_DIR = @MFEM_DIR@/../metis-4.0
METIS_OPT =
@@ -234,7 +232,7 @@ SCALAPACK_DIR = @MFEM_DIR@/../scalapack-2.0.2
SCALAPACK_OPT = -I$(SCALAPACK_DIR)/SRC
SCALAPACK_LIB = -L$(SCALAPACK_DIR)/lib -lscalapack $(LAPACK_LIB)
# MPI Fortran library, needed e.g. by STRUMPACK or MUMPS
# MPI Fortran library, needed e.g. by STRUMPACK
# MPICH:
MPI_FORTRAN_LIB = -lmpifort
# OpenMPI:
@@ -242,11 +240,6 @@ MPI_FORTRAN_LIB = -lmpifort
# Additional Fortan library:
# MPI_FORTRAN_LIB += -lgfortran
# MUMPS library configuration
MUMPS_DIR =
MUMPS_OPT = -I$(MUMPS_DIR)/include
MUMPS_LIB = -Wl,-rpath,$(MUMPS_DIR)/lib -L$(MUMPS_DIR)/lib -ldmumps -lmumps_common -lpord $(SCALAPACK_LIB) $(LAPACK_LIB) $(MPI_FORTRAN_LIB)
# STRUMPACK library configuration
STRUMPACK_DIR = @MFEM_DIR@/../STRUMPACK-build
STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
@@ -348,9 +341,9 @@ GSLIB_DIR = @MFEM_DIR@/../gslib/build
GSLIB_OPT = -I$(GSLIB_DIR)/include
GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
# CUDA library configuration
# CUDA library configuration (currently not needed)
CUDA_OPT =
CUDA_LIB = -lcusparse
CUDA_LIB =
# HIP library configuration (currently not needed)
HIP_OPT =
@@ -379,15 +372,6 @@ UMPIRE_DIR = @MFEM_DIR@/../umpire
UMPIRE_OPT = -I$(UMPIRE_DIR)/include
UMPIRE_LIB = -L$(UMPIRE_DIR)/lib -lumpire
# MKL CPardiso library configuration
MKL_CPARDISO_DIR ?=
MKL_MPI_WRAPPER ?= mkl_blacs_mpich_lp64
MKL_LIBRARY_SUBDIR ?= lib
MKL_CPARDISO_OPT = -I$(MKL_CPARDISO_DIR)/include
MKL_CPARDISO_LIB = -Wl,-rpath,$(MKL_CPARDISO_DIR)/$(MKL_LIBRARY_SUBDIR)\
-L$(MKL_CPARDISO_DIR)/$(MKL_LIBRARY_SUBDIR) -l$(MKL_MPI_WRAPPER)\
-lmkl_intel_lp64 -lmkl_sequential -lmkl_core
# If YES, enable some informational messages
VERBOSE = NO
-33
View File
@@ -1,33 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "dmumps_c.h"
#include <string>
#include <iostream>
#include <algorithm>
// Macros to expand a macro as a string
#define STR_EXPAND(s) #s
#define STR(s) STR_EXPAND(s)
int main()
{
#ifdef MUMPS_VERSION
const char *ptr = STR(MUMPS_VERSION);
std::string s(ptr);
s.erase(std::remove(s.begin(), s.end(), '"'), s.end());
s.erase(std::remove(s.begin(), s.end(), '.'), s.end());
std::cout << s << "\n";
return 0;
#else
return -1;
#endif
}
+2 -19
View File
@@ -42,10 +42,6 @@ GHV_FLAGS = $(subst @MFEM_DIR@,$(if $(MFEM_DIR),$(MFEM_DIR),..),$(HYPRE_OPT))
SMX = $(if $(MFEM_USE_PUMI:NO=),MFEM_USE_SIMMETRIX)
SMX_PATH = $(PUMI_DIR)/include/gmi_sim.h
SMX_FILE = $(subst @MFEM_DIR@,$(if $(MFEM_DIR),$(MFEM_DIR),..),$(SMX_PATH))
MUMPS = $(MFEM_USE_MUMPS:NO=)
GMV_CXX ?= $(MFEM_CXX)
GMV = get_mumps_version
GMV_FLAGS = $(subst @MFEM_DIR@,$(if $(MFEM_DIR),$(MFEM_DIR),..),$(MUMPS_OPT))
$(GHV): $(SRC)$(GHV).cpp
$(call mfem-info, Determining HYPRE version ...)
@@ -54,13 +50,6 @@ $(GHV).out: $(GHV)
./$(GHV) > $(GHV).out
.INTERMEDIATE: $(GHV) $(GHV).out
$(GMV): $(SRC)$(GMV).cpp
$(call mfem-info, Determining MUMPS version ...)
$(GMV_CXX) ${GMV_FLAGS} $(SRC)$(GMV).cpp -o $(GMV)
$(GMV).out: $(GMV)
./$(GMV) > $(GMV).out
.INTERMEDIATE: $(GMV) $(GMV).out
get-hypre-version: $(GHV).out
$(eval MFEM_HYPRE_VERSION:=$(shell cat $(GHV).out))
$(if $(MFEM_HYPRE_VERSION),$(eval export MFEM_HYPRE_VERSION)\
@@ -73,16 +62,10 @@ check-smx:
$(call mfem-info, MFEM_USE_SIMMETRIX = $(MFEM_USE_SIMMETRIX))
$(eval export MFEM_USE_SIMMETRIX)
get-mumps-version: $(GMV).out
$(eval MFEM_MUMPS_VERSION:=$(shell cat $(GMV).out))
$(if $(MFEM_MUMPS_VERSION),$(eval export MFEM_MUMPS_VERSION)\
$(info MUMPS version: $(MFEM_MUMPS_VERSION)),\
$(error Unable to determine MUMPS version))
header: $(if $(MPI),get-hypre-version,) $(if $(SMX),check-smx,) $(if $(MUMPS),get-mumps-version,)
header: $(if $(MPI),get-hypre-version,) $(if $(SMX),check-smx)
$(call mfem-info, Writing $(CONFIG_HPP) ...)
@set -- && \
for def in $${MFEM_DEFINES} $(if $(MPI),MFEM_HYPRE_VERSION) $(SMX) $(if $(MUMPS),MFEM_MUMPS_VERSION); do \
for def in $${MFEM_DEFINES} $(if $(MPI),MFEM_HYPRE_VERSION) $(SMX); do \
eval var=\$$$$def && \
if [ "NO" != "$${var}" ]; then \
set -- "$$@" -e "s|// \(#define $${def} \)|\1|" && \
+7 -50
View File
@@ -78,14 +78,6 @@ groups_parallel=(
"miniapps/electromagnetics"
"joule.cpp"'
# "{volta,tesla,joule}.cpp"' # todo: multiline sample runs
'"convergence"
"Convergence tests:"
"tests/convergence"
"diffusion.cpp"'
'"par-mesh-format"
"Parallel mesh tests:"
"tests/par-mesh-format"
"ex1p.cpp"'
)
# All groups serial + parallel runs mixed in the same group:
groups_all=(
@@ -115,14 +107,6 @@ groups_all=(
"miniapps/electromagnetics"
"joule.cpp"'
# "{volta,tesla,joule}.cpp"' # todo: multiline sample runs
'"convergence"
"Convergence tests:"
"tests/convergence"
"diffusion.cpp"'
'"par-mesh-format"
"Parallel mesh tests:"
"tests/par-mesh-format"
"ex1p.cpp"'
)
make_all="all"
base_timeformat=$'real: %3Rs user: %3Us sys: %3Ss %%cpu: %P'
@@ -396,15 +380,10 @@ function timed_run()
# This function is used to execute the sample runs
function go()
{
# Strip leading and trailing spaces from $1 and store the result in cmd_line
shopt -s extglob
local cmd_line="${1##+( )}"
cmd_line="${cmd_line%%+( )}"
shopt -u extglob
eval local cmd=(${cmd_line})
local cmd=("$@")
local res=""
echo $sep
echo "<${group}>" "${cmd_line}"
echo "<${group}>" "${cmd[@]}"
echo $sep
if [ "${timing}" == "yes" ]; then
timed_run "${cmd[@]}"
@@ -416,15 +395,15 @@ function go()
else
res="${red}FAILED${none}"
fi
printf "[${res}] <${group}> ${cmd_line}\n"
printf "[${res}] <${group}> ${cmd[*]}\n"
if [ "${timing}" == "yes" ]; then
printf "Run time: %s\n" "${timer}"
timer=(${timer})
timer="${timer[1]}"
printf -v line "[$res](%8s) ${cmd_line}" "$timer"
printf -v line "[$res](%8s) ${cmd[*]}" "$timer"
summary=("${summary[@]}" "$line")
else
summary=("${summary[@]}" "[${res}] ${cmd_line}")
summary=("${summary[@]}" "[${res}] ${cmd[*]}")
fi
echo $sep
}
@@ -459,7 +438,7 @@ function go_group()
fi
for run in "${runs[@]}"; do
if [ "${run}" == "" ]; then continue; fi
eval go \"\${run_prefix} \${run} \${run_suffix}\" $output
eval go \${run_prefix} \${run} \${run_suffix} $output
done
done
${make} clean-exec
@@ -525,7 +504,7 @@ function echo_run()
{
echo " $@"
{ echo " $@"; echo "$sep";
eval "$@"
"$@"
echo "$sep"; } >> "$echo_log" 2>&1
}
@@ -545,28 +524,6 @@ function build_all()
echo_run ${make} config ${mfem_config} || exit 1
echo_run ${make} ${make_j} || exit 1
echo_run ${make} ${make_all} ${make_j} || exit 1
# Build groups in directories other than the directories built by 'make all':
for group_params in "${groups[@]}"; do
eval params=(${group_params})
group_dir="${params[2]}"
case "$group_dir" in
(examples*|miniapps*)
# Built by 'make all'
;;
(*)
if [ "${mfem_dir}" != "${mfem_build_dir}" ]; then
echo_run mkdir -p "${group_dir}" || exit 1
echo_run cd "${group_dir}" || exit 1
echo_run cp -af "${mfem_dir}/${group_dir}/makefile" . || exit 1
else
echo_run cd "${group_dir}" || exit 1
fi
echo_run ${make} clean || exit 1
echo_run ${make} MFEM_DIR="${mfem_dir}" ${make_j} || exit 1
echo_run cd "${mfem_build_dir}" || exit 1
;;
esac
done
}
# Function that runs all sample runs, given by the array variable "groups".
+8 -57
View File
@@ -1,38 +1,13 @@
SetFactory("OpenCASCADE");
// Select periodic mesh by setting this to either 0 - standard, 1 - periodic
periodic = 1;
// Set the geometry order (1, 2, ..., 9)
order = 3;
// Set the element type (3 - triangles, 4 - quadrilaterals)
type = 3;
// Number of radial elements
nrad = 2;
// Number of azimuthal elements on inner arc
nazm1 = 3;
// Number of azimuthal elements on outer arc
nazm2 = 5;
// Note: Using type = 4 with nazm1 != nazm2 can lead to mixed meshes
// containing both triangles and quadrilaterals.
// Inner and outer radii
R1 = 1.0;
R2 = 2.0;
// Angular size of the sector
Phi = Pi/3.0;
Point(1) = {0.0, 0, 0, 1.0};
Point(2) = {R1, 0, 0, 1.0};
Point(3) = {R2, 0, 0, 1.0};
Point(4) = {R1*Cos(Phi), R1*Sin(Phi), 0, 1.0};
Point(5) = {R2*Cos(Phi), R2*Sin(Phi), 0, 1.0};
Point(4) = {R1*Cos(Pi/3), R1*Sin(Pi/3), 0, 1.0};
Point(5) = {R2*Cos(Pi/3), R2*Sin(Pi/3), 0, 1.0};
Line(1) = {2, 3};
Line(2) = {4, 5};
Circle(3) = {2, 1, 4};
@@ -40,23 +15,13 @@ Circle(4) = {3, 1, 5};
Curve Loop(5) = {1, 4, -2, -3};
Plane Surface(1) = {5};
Transfinite Curve{1} = nrad+1;
Transfinite Curve{2} = nrad+1;
Transfinite Curve{3} = nazm1+1;
Transfinite Curve{4} = nazm2+1;
If (nazm1 == nazm2)
Transfinite Surface{1};
EndIf
If (type == 4)
Recombine Surface {1};
EndIf
Transfinite Curve{1} = 7;
Transfinite Curve{2} = 7;
Transfinite Curve{3} = 4;
Transfinite Curve{4} = 10;
// Set a rotation periodicity constraint:
If (periodic)
Periodic Line{1} = {2} Rotate{{0,0,1}, {0,0,0}, -Phi};
EndIf
Periodic Line{1} = {2} Rotate{{0,0,1}, {0,0,0}, -Pi/3};
// Tag surfaces and volumes with positive integers
Physical Curve(1) = {3};
@@ -65,22 +30,8 @@ Physical Curve(3) = {1};
Physical Curve(4) = {2};
Physical Surface(1) = {1};
// Optimize the high-order mesh
// See https://gmsh.info/doc/texinfo/gmsh.html#index-Mesh_002eHighOrderOptimize
// Mesh.ElementOrder = order;
// Mesh.HighOrderOptimize = 1;
// Generate 2D mesh
Mesh 2;
SetOrder order;
Mesh.MshFileVersion = 2.2;
// Check the element quality (the Plugin may be called AnalyseCurvedMesh)
// Plugin(AnalyseMeshQuality).JacobianDeterminant = 1;
// Plugin(AnalyseMeshQuality).Run;
If (periodic)
Save Sprintf("periodic-annulus-sector-t%01g-o%01g.msh", type, order);
Else
Save Sprintf("annulus-sector-t%01g-o%01g.msh", type, order);
EndIf
Save "periodic-annulus-sector.msh";
+161 -168
View File
@@ -2,191 +2,184 @@ $MeshFormat
2.2 0 8
$EndMeshFormat
$Nodes
136
55
1 1 0 0
2 2 0 0
3 0.5000000000000001 0.8660254037844386 0
4 1 1.732050807568877 0
5 1.5 0 0
6 1.166666666666667 0 0
7 1.333333333333333 0 0
5 1.166666666666667 0 0
6 1.333333333333333 0 0
7 1.5 0 0
8 1.666666666666667 0 0
9 1.833333333333333 0 0
10 0.7500000000000002 1.299038105676658 0
11 0.5833333333333335 1.010362971081845 0
12 0.6666666666666667 1.154700538379251 0
10 0.5833333333333335 1.010362971081845 0
11 0.6666666666666667 1.154700538379251 0
12 0.7500000000000002 1.299038105676658 0
13 0.8333333333333335 1.443375672974064 0
14 0.9166666666666669 1.587713240271471 0
15 0.9396926207859085 0.3420201433256683 0
16 0.7660444431189786 0.6427876096865386 0
17 0.993238357741943 0.1160929141252301 0
18 0.9730448705798238 0.2306158707424401 0
19 0.8936326403234125 0.4487991802004617 0
20 0.8354878114129367 0.5495089780708056 0
21 0.6862416378687343 0.7273736415730481 0
22 0.597158591702787 0.8021231927550432 0
23 1.956295201467611 0.4158233816355181 0
24 1.827090915285202 0.8134732861515996 0
25 1.618033988749896 1.175570504584944 0
26 1.338261212717719 1.486289650954786 0
27 1.995128100519648 0.1395129474882505 0
28 1.980536137483141 0.278346201920131 0
29 1.922523391876638 0.551274711633998 0
30 1.879385241571817 0.6840402866513373 0
31 1.765895185717855 0.9389431255717802 0
32 1.696096192312853 1.059838528466408 0
33 1.532088886237958 1.285575219373077 0
34 1.438679600677305 1.389316740917992 0
35 1.231322950651319 1.576021507213442 0
36 1.118385806941496 1.658075145110082 0
37 1.162276263405681 0.6710405135499813 0
38 1.248615852873337 1.079531485311822 0
39 1.559209616901855 0.5415673055003691 0
40 1.478306597054007 0.8535007117539289 0
41 0.9210953433941653 0.9653302893212266 0
42 1.296548225291847 0.3150268220262836 0
43 1.055002035226811 1.358510675893086 0
44 1.704005774249187 0.2344032256041583 0
45 0.6403651144647218 0.8991270322967013 0
46 0.7807302289294435 0.9322286608089638 0
47 0.864063562262777 1.07656622810637 0
48 0.8070317811313885 1.187802166891514 0
49 0.7236984477980553 1.043464599594108 0
50 1.432182741763949 0.1050089406754279 0
51 1.364365483527898 0.2100178813508558 0
52 1.197698816861231 0.2100178813508558 0
53 1.098849408430616 0.1050089406754279 0
54 1.265516075097282 0.1050089406754278 0
55 0.8177280765440409 0.7503018362314346 0
56 0.869411709969103 0.8578160627763305 0
57 0.7348318576552288 0.8316090412164392 0
58 1.177596357123201 0.3240245957927452 0
59 1.058644488954555 0.3330223695592067 0
60 1.087610484537871 0.2205785356313179 0
61 1.267619707955123 0.7318605796179638 0
62 1.372963152504565 0.7926806456859463 0
63 1.40174301566045 0.9288443029398934 0
64 1.325179434266893 1.004187894125858 0
65 1.219835989717452 0.9433678280578752 0
66 1.191056126561566 0.8072041708039283 0
67 1.296399571111008 0.8680242368719107 0
68 1.532241943619239 0.645545107584889 0
69 1.505274270336623 0.7495229096694089 0
70 1.294587381237739 0.6278827775334439 0
71 1.426898499069797 0.5847250415169065 0
72 1.399930825787181 0.6887028436014264 0
73 1.139442349713613 1.041464419981624 0
74 1.030268846553889 1.003397354651425 0
75 1.001488983398004 0.8672336973974781 0
76 1.081882623401843 0.7691371054737297 0
77 1.110662486557728 0.9053007627276766 0
78 1.207033584034403 0.5523692830420821 0
79 1.251790904663125 0.4336980525341829 0
80 1.384102022495183 0.3905403165176455 0
81 1.471655819698519 0.4660538110090073 0
82 1.339344701866461 0.5092115470255447 0
83 1.73779714915742 0.7228379592678562 0
84 1.648503383029637 0.6322026323841127 0
85 1.691571478423774 0.4996526642120855 0
86 1.823933339945692 0.4577380229238018 0
87 1.787039416783436 0.5922941012619014 0
88 1.308379426102925 1.350703595740465 0
89 1.278497639488131 1.215117540526144 0
90 1.371755231498856 1.111544491736196 0
91 1.494894610124376 1.14355749816057 0
92 1.4064614465962 1.25147448186763 0
93 1.013887168325833 0.451693600067106 0
94 1.088081715865757 0.5613670568085436 0
95 1.03019898997678 0.6616228789288338 0
96 0.8981217165478794 0.6522052443076862 0
97 0.9637989050473432 0.5564495572737495 0
98 1.432367408277627 0.2881522898855752 0
99 1.568186591263407 0.2612777577448668 0
100 1.655740388466743 0.3367912522362286 0
101 1.607475002684299 0.4391792788682989 0
102 1.519921205480963 0.3636657843769371 0
103 1.184077913657828 1.17252454883891 0
104 1.119539974442319 1.265517612365998 0
105 1.010366471282595 1.2274505470358 0
106 0.9657309073383804 1.096390418178513 0
107 1.074904410498104 1.134457483508712 0
108 1.901335258083062 0.07813440853471942 0
109 1.802670516166124 0.1562688170694388 0
110 1.636003849499458 0.156268817069439 0
111 1.568001924749729 0.07813440853471942 0
112 1.734668591416396 0.07813440853471944 0
113 0.8516673450756037 1.318862295748801 0
114 0.9533346901512071 1.338686485820944 0
115 1.03666802348454 1.48302405311835 0
116 1.01833401174227 1.607537430343614 0
117 0.9350006784089369 1.463199863046207 0
118 1.710829475874804 0.8268157613523761 0
119 1.594568036464405 0.8401582365531526 0
120 1.621535709747021 0.7361804344686326 0
121 1.52488239428597 0.9608573093642675 0
122 1.571458191517933 1.068213906974606 0
123 1.448318812892413 1.036200900550232 0
124 0.908699126206992 1.207626356963657 0
125 1.500184666513678 0.1831433492101474 0
126 1.646765991694905 0.9507607885973723 0
127 0.9498053499729417 0.7597194708525821 0
128 1.132839036494479 0.4426958263006443 0
129 1.149421761057113 1.40110366758032 0
130 1.243841486887416 1.443696659267553 0
131 1.134903597606542 1.530869109405537 0
132 1.872198725728137 0.3553499962917315 0
133 1.788102249988661 0.2948766109479449 0
134 1.893223337417325 0.2174207916708467 0
135 1.213959700272622 1.308110604053232 0
136 1.739836864206217 0.3972646375800152 0
17 1.986476715483886 0.2321858282504602 0
18 1.946089741159648 0.4612317414848793 0
19 1.879385241571817 0.6840402866513365 0
20 1.787265280646825 0.8975983604009234 0
21 1.670975622825874 1.09901795614161 0
22 1.532088886237958 1.285575219373077 0
23 1.372483275737469 1.454747283146095 0
24 1.194317183405575 1.604246385510085 0
25 1.425989114816062 0.1915326920916892 0
26 0.8788667344146573 1.13917645290495 0
27 1.630372059110754 0.7154531062316609 0
28 1.436395769298814 1.053728612482506 0
29 1.081023776188756 0.6241293681829633 0
30 1.168737372335971 1.428012728596308 0
31 1.821063986059922 0.298149890497067 0
32 1.234707097211386 0.3469796339295647 0
33 1.377747393186519 0.6200150626754309 0
34 1.457047681210906 0.3890895843559762 0
35 0.917846726184522 0.8957978954532204 0
36 1.218335619030348 0.9017812086952638 0
37 1.066623110765233 1.061857005744772 0
38 1.587029716281926 0.1355955181472859 0
39 1.744445799211916 0.1441515753740107 0
40 1.25 0.1443375672974065 0
41 1.453660070628011 0.8435769396609902 0
42 1.741367044061892 0.499612708014486 0
43 1.30550638526547 1.257610469847477 0
44 1.118213276932792 0.1666674689105279 0
45 0.9109440214958271 1.306610291787315 0
46 0.9970618258753989 1.438658589955562 0
47 0.7499999999999998 1.010362971081845 0
48 0.7034449005273667 0.8850673702175776 0
49 1.605449512513618 0.9269067082200894 0
50 1.561654019115059 0.5298592532912715 0
51 1.229782222487711 1.096820457143683 0
52 1.617066998712459 0.3090202662210922 0
53 1.079645953234324 1.246963713711438 0
54 1.877063966817811 0.1348974588243076 0
55 1.055356609656722 1.558136350380461 0
$EndNodes
$Elements
38
1 26 2 3 1 1 5 6 7
2 26 2 3 1 5 2 8 9
3 26 2 4 2 3 10 11 12
4 26 2 4 2 10 4 13 14
5 26 2 1 3 1 15 17 18
6 26 2 1 3 15 16 19 20
7 26 2 1 3 16 3 21 22
8 26 2 2 4 2 23 27 28
9 26 2 2 4 23 24 29 30
10 26 2 2 4 24 25 31 32
11 26 2 2 4 25 26 33 34
12 26 2 2 4 26 4 35 36
13 21 2 1 1 3 41 10 45 46 47 48 12 11 49
14 21 2 1 1 5 42 1 50 51 52 53 6 7 54
15 21 2 1 1 16 41 3 55 56 46 45 22 21 57
16 21 2 1 1 1 42 15 53 52 58 59 18 17 60
17 21 2 1 1 37 40 38 61 62 63 64 65 66 67
18 21 2 1 1 39 40 37 68 69 62 61 70 71 72
19 21 2 1 1 38 41 37 73 74 75 76 66 65 77
20 21 2 1 1 37 42 39 78 79 80 81 71 70 82
21 21 2 1 1 24 39 23 83 84 85 86 29 30 87
22 21 2 1 1 26 38 25 88 89 90 91 33 34 92
23 21 2 1 1 15 37 16 93 94 95 96 20 19 97
24 21 2 1 1 42 44 39 98 99 100 101 81 80 102
25 21 2 1 1 38 43 41 103 104 105 106 74 73 107
26 21 2 1 1 2 44 5 108 109 110 111 8 9 112
27 21 2 1 1 10 43 4 113 114 115 116 14 13 117
28 21 2 1 1 24 40 39 118 119 69 68 84 83 120
29 21 2 1 1 38 40 25 64 63 121 122 91 90 123
30 21 2 1 1 41 43 10 106 105 114 113 48 47 124
31 21 2 1 1 5 44 42 111 110 99 98 51 50 125
32 21 2 1 1 25 40 24 122 121 119 118 31 32 126
33 21 2 1 1 37 41 16 76 75 56 55 96 95 127
34 21 2 1 1 15 42 37 59 58 79 78 94 93 128
35 21 2 1 1 4 43 26 116 115 129 130 35 36 131
36 21 2 1 1 23 44 2 132 133 109 108 27 28 134
37 21 2 1 1 26 43 38 130 129 104 103 89 88 135
38 21 2 1 1 39 44 23 101 100 133 132 86 85 136
108
1 1 2 3 1 1 5
2 1 2 3 1 5 6
3 1 2 3 1 6 7
4 1 2 3 1 7 8
5 1 2 3 1 8 9
6 1 2 3 1 9 2
7 1 2 4 2 3 10
8 1 2 4 2 10 11
9 1 2 4 2 11 12
10 1 2 4 2 12 13
11 1 2 4 2 13 14
12 1 2 4 2 14 4
13 1 2 1 3 1 15
14 1 2 1 3 15 16
15 1 2 1 3 16 3
16 1 2 2 4 2 17
17 1 2 2 4 17 18
18 1 2 2 4 18 19
19 1 2 2 4 19 20
20 1 2 2 4 20 21
21 1 2 2 4 21 22
22 1 2 2 4 22 23
23 1 2 2 4 23 24
24 1 2 2 4 24 4
25 2 2 1 1 32 40 25
26 2 2 1 1 25 34 32
27 2 2 1 1 33 41 36
28 2 2 1 1 38 52 25
29 2 2 1 1 33 36 29
30 2 2 1 1 26 47 35
31 2 2 1 1 35 37 26
32 2 2 1 1 25 52 34
33 2 2 1 1 32 44 40
34 2 2 1 1 15 32 29
35 2 2 1 1 15 29 16
36 2 2 1 1 36 41 28
37 2 2 1 1 32 33 29
38 2 2 1 1 50 52 42
39 2 2 1 1 32 34 33
40 2 2 1 1 42 52 31
41 2 2 1 1 43 53 51
42 2 2 1 1 27 41 33
43 2 2 1 1 26 53 45
44 2 2 1 1 18 31 17
45 2 2 1 1 29 35 16
46 2 2 1 1 29 36 35
47 2 2 1 1 24 30 23
48 2 2 1 1 30 53 43
49 2 2 1 1 17 54 2
50 2 2 1 1 4 55 24
51 2 2 1 1 28 51 36
52 2 2 1 1 47 48 35
53 2 2 1 1 36 37 35
54 2 2 1 1 37 53 26
55 2 2 1 1 22 28 21
56 2 2 1 1 20 27 19
57 2 2 1 1 33 50 27
58 2 2 1 1 15 44 32
59 2 2 1 1 18 42 31
60 2 2 1 1 30 43 23
61 2 2 1 1 35 48 16
62 2 2 1 1 31 54 17
63 2 2 1 1 9 39 8
64 2 2 1 1 8 38 7
65 2 2 1 1 7 25 6
66 2 2 1 1 22 43 28
67 2 2 1 1 23 43 22
68 2 2 1 1 39 54 31
69 2 2 1 1 19 42 18
70 2 2 1 1 24 55 30
71 2 2 1 1 27 42 19
72 2 2 1 1 13 46 14
73 2 2 1 1 51 53 37
74 2 2 1 1 39 52 38
75 2 2 1 1 6 40 5
76 2 2 1 1 34 52 50
77 2 2 1 1 12 45 13
78 2 2 1 1 30 55 46
79 2 2 1 1 10 47 11
80 2 2 1 1 8 39 38
81 2 2 1 1 28 49 21
82 2 2 1 1 7 38 25
83 2 2 1 1 41 49 28
84 2 2 1 1 20 49 27
85 2 2 1 1 11 26 12
86 2 2 1 1 27 49 41
87 2 2 1 1 31 52 39
88 2 2 1 1 25 40 6
89 2 2 1 1 2 54 9
90 2 2 1 1 14 55 4
91 2 2 1 1 45 53 46
92 2 2 1 1 45 46 13
93 2 2 1 1 5 44 1
94 2 2 1 1 21 49 20
95 2 2 1 1 46 53 30
96 2 2 1 1 3 48 10
97 2 2 1 1 34 50 33
98 2 2 1 1 36 51 37
99 2 2 1 1 26 45 12
100 2 2 1 1 11 47 26
101 2 2 1 1 27 50 42
102 2 2 1 1 40 44 5
103 2 2 1 1 43 51 28
104 2 2 1 1 10 48 47
105 2 2 1 1 9 54 39
106 2 2 1 1 46 55 14
107 2 2 1 1 1 44 15
108 2 2 1 1 16 48 3
$EndElements
$Periodic
1
1 1 2
Affine 0.5000000000000001 0.8660254037844386 0 0 -0.8660254037844386 0.5000000000000001 0 0 0 0 1 0 0 0 0 1
3
7
9 14
6 11
8 13
5 10
1 3
7 12
2 4
1 3
$EndPeriodic
+13 -129
View File
@@ -1,141 +1,25 @@
// Select periodic mesh by setting this to either 0 - standard, 1 - periodic
periodic = 1;
SetFactory("OpenCASCADE");
// Set the geometry order (1, 2, ..., 10 for tetrahedra or 9 for other types)
order = 3;
R = 1.5;
r = 0.5;
// Set the element type (4 - tetrahedra, 6 - wedges, 8 - hexahedra)
type = 8;
Torus(1) = {0,0,0, R, r, Pi/3};
// Minor and major radii
R1 = 1.0;
R2 = 2.0;
pts() = PointsOf{ Volume{1}; };
// Side length of interior square
A1 = 0.8;
// Angular size of the sector
Phi = Pi/3.0;
// Number of azimuthal elements
nazm = 3;
// Number of elements around a quarter of the circle
narc = 2;
// Number of elements between surface and interior square
nshl = 1;
lc = 0.5;
a1 = A1 / Sqrt(2.0);
Point(1) = {R2+R1, 0, 0, lc};
Point(2) = {R2, 0, R1, lc};
Point(3) = {R2-R1, 0, 0, lc};
Point(4) = {R2, 0, -R1, lc};
Point(5) = {R2, 0, 0, lc};
Point(6) = {R2+a1, 0, 0, lc};
Point(7) = {R2, 0, a1, lc};
Point(8) = {R2-a1, 0, 0, lc};
Point(9) = {R2, 0, -a1, lc};
Circle(1) = {1,5,2};
Circle(2) = {2,5,3};
Circle(3) = {3,5,4};
Circle(4) = {4,5,1};
Line(5) = {6,1};
Line(6) = {7,2};
Line(7) = {8,3};
Line(8) = {9,4};
Line(9) = {6, 7};
Line(10) = {7, 8};
Line(11) = {8, 9};
Line(12) = {9, 6};
Line Loop(101) = {1, -6, -9, 5};
Line Loop(102) = {2, -7, -10, 6};
Line Loop(103) = {3, -8, -11, 7};
Line Loop(104) = {4, -5, -12, 8};
Line Loop(105) = {9, 10, 11, 12};
Plane Surface(201) = {101};
Plane Surface(202) = {102};
Plane Surface(203) = {103};
Plane Surface(204) = {104};
Plane Surface(205) = {105};
Transfinite Curve{1} = narc+1;
Transfinite Curve{2} = narc+1;
Transfinite Curve{3} = narc+1;
Transfinite Curve{4} = narc+1;
Transfinite Curve{5} = nshl+1;
Transfinite Curve{6} = nshl+1;
Transfinite Curve{7} = nshl+1;
Transfinite Curve{8} = nshl+1;
Transfinite Curve{9} = narc+1;
Transfinite Curve{10} = narc+1;
Transfinite Curve{11} = narc+1;
Transfinite Curve{12} = narc+1;
If (type == 8)
Recombine Surface {201};
Recombine Surface {202};
Recombine Surface {203};
Recombine Surface {204};
Recombine Surface {205};
Transfinite Surface {201} = {1,2,7,6};
Transfinite Surface {202} = {2,3,8,7};
Transfinite Surface {203} = {3,4,9,8};
Transfinite Surface {204} = {4,1,6,9};
Transfinite Surface {205} = {6,7,8,9};
EndIf
If (type == 4)
Extrude { {0,0,1} , {0,0,0} , Phi} {
Surface{201,202,203,204,205}; Layers{nazm};
}
Else
Extrude { {0,0,1} , {0,0,0} , Phi} {
Surface{201,202,203,204,205}; Layers{nazm}; Recombine;
}
EndIf
Characteristic Length{ pts() } = 0.25;
// Set a rotation periodicity constraint:
If (periodic)
Periodic Surface{227} = {201} Rotate{{0,0,1}, {0,0,0}, Phi};
Periodic Surface{249} = {202} Rotate{{0,0,1}, {0,0,0}, Phi};
Periodic Surface{271} = {203} Rotate{{0,0,1}, {0,0,0}, Phi};
Periodic Surface{293} = {204} Rotate{{0,0,1}, {0,0,0}, Phi};
Periodic Surface{315} = {205} Rotate{{0,0,1}, {0,0,0}, Phi};
EndIf
Periodic Surface{3} = {2} Rotate{{0,0,1}, {0,0,0}, Pi/3};
// Tag surfaces and volumes with positive integers
Physical Surface(1) = {201,202,203,204,205};
Physical Surface(2) = {227,249,271,293,315};
Physical Surface(3) = {214,236,258,280};
Physical Volume(1) = {1,2,3,4,5};
// Optimize the high-order mesh
// See https://gmsh.info/doc/texinfo/gmsh.html#index-Mesh_002eHighOrderOptimize
// Mesh.ElementOrder = order;
// Mesh.HighOrderOptimize = 1;
Physical Surface(1) = {1};
Physical Surface(2) = {2};
Physical Surface(3) = {3};
Physical Volume(1) = {1};
// Generate 3D mesh
Mesh 3;
SetOrder order;
Mesh.MshFileVersion = 2.2;
// Check the element quality (the Plugin may be called AnalyseCurvedMesh)
// Plugin(AnalyseMeshQuality).JacobianDeterminant = 1;
// Plugin(AnalyseMeshQuality).Run;
If (periodic)
Save Sprintf("periodic-torus-sector-t%01g-o%01g.msh", type, order);
Else
Save Sprintf("torus-sector-t%01g-o%01g.msh", type, order);
EndIf
Save "periodic-torus-sector.msh";
File diff suppressed because it is too large Load Diff
-118
View File
@@ -1,118 +0,0 @@
MFEM mesh v1.0
#
# MFEM Geometry Types (see mesh/geom.hpp):
#
# POINT = 0
# SEGMENT = 1
# TRIANGLE = 2
# SQUARE = 3
# TETRAHEDRON = 4
# CUBE = 5
# PRISM = 6
#
dimension
2
elements
20
1 3 0 1 6 5
1 3 1 2 7 6
1 3 2 3 8 7
1 3 3 4 9 8
1 3 5 6 11 10
1 2 6 7 11
1 2 7 12 11
1 2 7 8 13
1 2 7 13 12
1 3 8 9 14 13
1 3 10 11 16 15
1 2 11 12 17
1 2 11 17 16
1 2 12 13 17
1 2 13 18 17
1 3 13 14 19 18
1 3 15 16 21 20
1 3 16 17 22 21
1 3 17 18 23 22
1 3 18 19 24 23
boundary
16
2 1 0 1
2 1 1 2
2 1 2 3
2 1 3 4
2 1 21 20
2 1 22 21
2 1 23 22
2 1 24 23
1 1 5 0
1 1 10 5
1 1 15 10
1 1 20 15
1 1 4 9
1 1 9 14
1 1 14 19
1 1 19 24
vertices
25
nodes
FiniteElementSpace
FiniteElementCollection: H1_2D_P1
VDim: 2
Ordering: 0
0
0.25
0.5
0.75
1
0
0.25
0.5
0.75
1
0
0.25
0.5
0.75
1
0
0.25
0.5
0.75
1
0
0.25
0.5
0.75
1
0
0
0
0
0
0.25
0.25
0.25
0.25
0.25
0.5
0.5
0.5
0.5
0.5
0.75
0.75
0.75
0.75
0.75
1
1
1
1
1
+2 -3
View File
@@ -144,12 +144,12 @@ namespace mfem {
* - <a class="el" href="maxwell_8cpp_source.html">Maxwell</a>: simple transient full-wave electromagnetics simulation code
* - <a class="el" href="joule_8cpp_source.html">Joule</a>: transient magnetics and Joule heating miniapp
* - <a class="el" href="classmfem_1_1navier_1_1NavierSolver.html">Navier</a>: solve the transient incompressible Navier-Stokes equations
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
* - <a class="el" href="mobius-strip_8cpp_source.html">Mobius Strip</a>: generate various Mobius strip-like meshes
* - <a class="el" href="klein-bottle_8cpp_source.html">Klein Bottle</a>: generate three types of Klein bottle surfaces
* - <a class="el" href="toroid_8cpp_source.html">Toroid</a>: generate simple toroidal meshes
* - <a class="el" href="twist_8cpp_source.html">Twist</a>: generate simple periodic meshes
* - <a class="el" href="minimal-surface_8cpp_source.html">Minimal Surface</a>: compute minimal surfaces, <a class="el" href="minimal-surface_8cpp_source.html">serial</a> and <a class="el" href="pminimal-surface_8cpp_source.html">parallel</a> versions
* - <a class="el" href="polar-nc_8cpp_source.html">Polar NC</a>: generate polar non-conforming meshes
* - <a class="el" href="shaper_8cpp_source.html">Shaper</a>: resolve material interfaces by mesh refinement
* - <a class="el" href="extruder_8cpp_source.html">Extruder</a>: extrude a low-dimensional mesh into a higher dimension
* - <a class="el" href="mesh-explorer_8cpp_source.html">Mesh Explorer</a>: visualize and manipulate meshes
@@ -162,7 +162,6 @@ namespace mfem {
* - <a class="el" href="lor-transfer_8cpp_source.html">LOR Transfer</a>: map functions between high-order and low-order refined spaces
* - <a class="el" href="findpts_8cpp_source.html">Find Points</a>: evaluate grid function in physical space, <a class="el" href="findpts_8cpp_source.html">serial</a> and <a class="el" href="pfindpts_8cpp_source.html">parallel</a> versions
* - <a class="el" href="field-diff_8cpp_source.html">Field Diff</a>: compare grid functions on different meshes
* - <a class="el" href="field-interp_8cpp_source.html">Field Interp</a>: transfer a grid functions betwen meshes
* - <a class="el" href="miniapps_2performance_2ex1_8cpp_source.html">HPC Example 1</a>: high-performance nodal H1 FEM for the Laplace problem
* - <a class="el" href="miniapps_2performance_2ex1p_8cpp_source.html">HPC Example 1p</a>: high-performance parallel nodal H1 FEM for the Laplace problem
*
+1 -1
View File
@@ -19,7 +19,7 @@ html: $(DOXYGEN_CONF)
@# Generate the html documentation
@doxygen $(DOXYGEN_CONF)
@echo "<meta http-equiv=\"REFRESH\" content=\"0;URL=CodeDocumentation/html/index.html\">" > CodeDocumentation.html
@cat warnings.log 1>&2
@cat warnings.log
@# Generate the log of undocumented methods
@( cat $(DOXYGEN_CONF) ; echo "GENERATE_HTML=NO" ; echo "EXTRACT_ALL=NO" ; echo "WARN_LOGFILE=undoc.log" ; echo "QUIET=YES" ) | doxygen - &> /dev/null
+2 -20
View File
@@ -72,7 +72,6 @@ int main(int argc, char *argv[])
int seed = 75;
bool slu_solver = false;
bool sp_solver = false;
bool pardiso_solver = false;
bool visualization = 1;
OptionsParser args(argc, argv);
@@ -96,14 +95,6 @@ int main(int argc, char *argv[])
#ifdef MFEM_USE_STRUMPACK
args.AddOption(&sp_solver, "-sp", "--strumpack", "-no-sp",
"--no-strumpack", "Use the STRUMPACK Solver.");
#endif
#ifdef MFEM_USE_MKL_CPARDISO
args.AddOption(&pardiso_solver,
"-pardiso",
"--pardiso",
"-no-pardiso",
"--no-pardiso",
"Use the MKL Cluster Pardiso Solver.");
#endif
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
@@ -245,7 +236,7 @@ int main(int argc, char *argv[])
// preconditioner for A to be used within the solver. Set the matrices
// which define the generalized eigenproblem A x = lambda M x.
Solver * precond = NULL;
if (!slu_solver && !sp_solver && !pardiso_solver)
if (!slu_solver && !sp_solver)
{
HypreBoomerAMG * amg = new HypreBoomerAMG(*A);
amg->SetPrintLevel(0);
@@ -277,19 +268,10 @@ int main(int argc, char *argv[])
strumpack->SetFromCommandLine();
precond = strumpack;
}
#endif
#ifdef MFEM_USE_MKL_CPARDISO
if (pardiso_solver)
{
auto pardiso = new CPardisoSolver(A->GetComm());
pardiso->SetMatrixType(CPardisoSolver::MatType::REAL_STRUCTURE_SYMMETRIC);
pardiso->SetPrintLevel(1);
pardiso->SetOperator(*A);
precond = pardiso;
}
#endif
}
HypreLOBPCG * lobpcg = new HypreLOBPCG(MPI_COMM_WORLD);
lobpcg->SetNumModes(nev);
lobpcg->SetRandomSeed(seed);
-234
View File
@@ -1,234 +0,0 @@
// MFEM Example 1 - Parallel Version
//
// Compile with: make ex1p
//
// Sample runs: mpirun -np 4 ex1p -m ../data/square-disc.mesh
//
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
int main(int argc, char *argv[])
{
// 1. Initialize MPI.
int num_procs, myid;
MPI_Init(&argc, &argv);
MPI_Comm_size(MPI_COMM_WORLD, &num_procs);
MPI_Comm_rank(MPI_COMM_WORLD, &myid);
// 2. Parse command-line options.
const char *mesh_file = "../data/inline-quad.mesh";
int order = 1;
bool static_cond = false;
bool visualization = true;
int sr = 1;
int pr = 1;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
"Mesh file to use.");
args.AddOption(&order, "-o", "--order",
"Finite element order (polynomial degree) or -1 for"
" isoparametric space.");
args.AddOption(&sr, "-sr", "--serial_ref",
"Number of serial refinements");
args.AddOption(&pr, "-pr", "--parallel_ref",
"Number of parallel refinements");
args.AddOption(&static_cond, "-sc", "--static-condensation", "-no-sc",
"--no-static-condensation", "Enable static condensation.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.Parse();
if (!args.Good())
{
if (myid == 0)
{
args.PrintUsage(cout);
}
MPI_Finalize();
return 1;
}
if (myid == 0)
{
args.PrintOptions(cout);
}
// 4. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
// and volume meshes with the same code.
Mesh mesh(mesh_file, 1, 1);
int dim = mesh.Dimension();
// 5. Refine the serial mesh on all processors to increase the resolution. In
// this example we do 'ref_levels' of uniform refinement. We choose
// 'ref_levels' to be the largest number that gives a final mesh with no
// more than 10,000 elements.
{
for (int l = 0; l < sr; l++)
{
mesh.UniformRefinement();
}
}
// 6. Define a parallel mesh by a partitioning of the serial mesh. Refine
// this mesh further in parallel to increase the resolution. Once the
// parallel mesh is defined, the serial mesh can be deleted.
ParMesh pmesh(MPI_COMM_WORLD, mesh);
mesh.Clear();
{
for (int l = 0; l < pr; l++)
{
pmesh.UniformRefinement();
}
}
// 7. Define a parallel finite element space on the parallel mesh. Here we
// use continuous Lagrange finite elements of the specified order. If
// order < 1, we instead use an isoparametric/isogeometric space.
FiniteElementCollection *fec = new H1_FECollection(order, dim);
ParFiniteElementSpace fespace(&pmesh, fec);
HYPRE_Int size = fespace.GlobalTrueVSize();
if (myid == 0)
{
cout << "Number of finite element unknowns: " << size << endl;
}
// 8. Determine the list of true (i.e. parallel conforming) essential
// boundary dofs. In this example, the boundary conditions are defined
// by marking all the boundary attributes from the mesh as essential
// (Dirichlet) and converting them to a list of true dofs.
Array<int> ess_tdof_list;
if (pmesh.bdr_attributes.Size())
{
Array<int> ess_bdr(pmesh.bdr_attributes.Max());
ess_bdr = 1;
fespace.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
}
// 9. Set up the parallel linear form b(.) which corresponds to the
// right-hand side of the FEM linear system, which in this case is
// (1,phi_i) where phi_i are the basis functions in fespace.
ParLinearForm b(&fespace);
ConstantCoefficient one(1.0);
b.AddDomainIntegrator(new DomainLFIntegrator(one));
b.Assemble();
// 10. Define the solution vector x as a parallel finite element grid function
// corresponding to fespace. Initialize x with initial guess of zero,
// which satisfies the boundary conditions.
ParGridFunction x(&fespace);
x = 0.0;
// 11. Set up the parallel bilinear form a(.,.) on the finite element space
// corresponding to the Laplacian operator -Delta, by adding the Diffusion
// domain integrator.
ParBilinearForm a(&fespace);
a.AddDomainIntegrator(new DiffusionIntegrator(one));
if (static_cond) { a.EnableStaticCondensation(); }
a.Assemble();
HypreParMatrix A;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
// // 13. Solve the linear system A X = B.
// // * With full assembly, use the BoomerAMG preconditioner from hypre.
// // * With partial assembly, use Jacobi smoothing, for now.
StopWatch chrono;
chrono.Clear();
chrono.Start();
HypreBoomerAMG *prec = new HypreBoomerAMG;
prec->SetPrintLevel(0);
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-13);
cg.SetMaxIter(2000);
cg.SetPrintLevel(0);
if (prec) { cg.SetPreconditioner(*prec); }
cg.SetOperator(A);
cg.Mult(B, X);
delete prec;
if (myid == 0)
{
cout << "PCG-AMG time: " << chrono.RealTime() << endl;
}
chrono.Clear();
chrono.Start();
{
MUMPSSolver MA;
MA.SetMatrixSymType(0);
MA.SetOperator(A);
Vector Y(X.Size());
MA.Mult(B,Y);
Y-=X;
cout << "Mumps Diff norm = " << Y.Norml2() << endl;
}
if (myid == 0)
{
cout << "mumps time: " << chrono.RealTime() << endl;
}
chrono.Clear();
chrono.Start();
{
CPardisoSolver pardiso(A.GetComm());
// pardiso.SetMatrixType(CPardisoSolver::MatType::REAL_STRUCTURE_SYMMETRIC);
pardiso.SetMatrixType(CPardisoSolver::MatType::REAL_UNSYMMETRIC);
pardiso.SetPrintLevel(0);
pardiso.SetOperator(A);
Vector Y(X.Size());
pardiso.Mult(B, Y);
Y-=X;
cout << "Pardiso Diff norm = " << Y.Norml2() << endl;
}
if (myid == 0)
{
cout << "pardiso time: " << chrono.RealTime() << endl;
}
{
SuperLURowLocMatrix SA(A);
SuperLUSolver superlu(MPI_COMM_WORLD);
superlu.SetPrintStatistics(false);
superlu.SetSymmetricPattern(false);
superlu.SetColumnPermutation(superlu::PARMETIS);
superlu.SetOperator(SA);
Vector Y(X.Size());
superlu.Mult(B, Y);
Y-=X;
cout << "Superlu Diff norm = " << Y.Norml2() << endl;
}
if (myid == 0)
{
cout << "superlu time: " << chrono.RealTime() << endl;
}
a.RecoverFEMSolution(X, b, x);
// 16. Send the solution by socket to a GLVis server.
if (visualization)
{
char vishost[] = "localhost";
int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock << "parallel " << num_procs << " " << myid << "\n";
sol_sock.precision(8);
sol_sock << "solution\n" << pmesh << x << flush;
}
// 17. Free the used memory.
delete fec;
MPI_Finalize();
return 0;
}
+2 -2
View File
@@ -6,13 +6,11 @@
// ex22 -m ../data/inline-tri.mesh -o 3
// ex22 -m ../data/inline-quad.mesh -o 3
// ex22 -m ../data/inline-quad.mesh -o 3 -p 1
// ex22 -m ../data/inline-quad.mesh -o 3 -p 1 -pa
// ex22 -m ../data/inline-quad.mesh -o 3 -p 2
// ex22 -m ../data/inline-tet.mesh -o 2
// ex22 -m ../data/inline-hex.mesh -o 2
// ex22 -m ../data/inline-hex.mesh -o 2 -p 1
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2
// ex22 -m ../data/inline-hex.mesh -o 2 -p 2 -pa
// ex22 -m ../data/star.mesh -r 1 -o 2 -sigma 10.0
//
// Device sample runs:
@@ -431,6 +429,8 @@ int main(int argc, char *argv[])
// 12. Recover the solution as a finite element grid function and compute the
// errors if the exact solution is known.
a->RecoverFEMSolution(U, b, u);
u.real().SyncMemory(u);
u.imag().SyncMemory(u);
if (exact_sol)
{
+3 -3
View File
@@ -7,12 +7,10 @@
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3 -p 1
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 3 -p 2
// mpirun -np 4 ex22p -m ../data/inline-quad.mesh -o 1 -p 1 -pa
// mpirun -np 4 ex22p -m ../data/inline-tet.mesh -o 2
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 1
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 2 -p 2
// mpirun -np 4 ex22p -m ../data/inline-hex.mesh -o 1 -p 2 -pa
// mpirun -np 4 ex22p -m ../data/star.mesh -o 2 -sigma 10.0
//
// Device sample runs:
@@ -167,7 +165,7 @@ int main(int argc, char *argv[])
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
if (myid == 0) { device.Print(); }
device.Print();
// 4. Read the (serial) mesh from the given mesh file on all processors. We
// can handle triangular, quadrilateral, tetrahedral, hexahedral, surface
@@ -471,6 +469,8 @@ int main(int argc, char *argv[])
// 14. Recover the parallel grid function corresponding to U. This is the
// local finite element solution on each processor.
a->RecoverFEMSolution(U, b, u);
u.real().SyncMemory(u);
u.imag().SyncMemory(u);
if (exact_sol)
{
File diff suppressed because it is too large Load Diff
+69 -31
View File
@@ -10,6 +10,10 @@
// ex25 -o 2 -f 8.0 -ref 3 -prob 4 -m ../data/inline-quad.mesh
// ex25 -o 2 -f 2.0 -ref 1 -prob 4 -m ../data/inline-hex.mesh
//
// Device sample runs:
// ex25 -o 2 -f 8.0 -ref 3 -prob 4 -m ../data/inline-quad.mesh -pa -d cuda
// ex25 -o 2 -f 2.0 -ref 1 -prob 4 -m ../data/inline-hex.mesh -pa -d cuda
//
// Description: This example code solves a simple electromagnetic wave
// propagation problem corresponding to the second order
// indefinite Maxwell equation
@@ -153,6 +157,8 @@ int main(int argc, char *argv[])
double freq = 5.0;
bool herm_conv = true;
bool visualization = 1;
bool pa = false;
const char *device_config = "cpu";
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -174,12 +180,21 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.Parse();
if (iprob > 4) { iprob = 4; }
prob = (prob_type)iprob;
// 2. Setup the mesh
// 2. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 3. Setup the mesh
if (!mesh_file)
{
exact_known = true;
@@ -220,7 +235,7 @@ int main(int argc, char *argv[])
// Setup PML length
Array2D<double> length(dim, 2); length = 0.0;
// 3. Setup the Cartesian PML region.
// 4. Setup the Cartesian PML region.
switch (prob)
{
case disc:
@@ -246,19 +261,19 @@ int main(int argc, char *argv[])
comp_domain_bdr = pml->GetCompDomainBdr();
domain_bdr = pml->GetDomainBdr();
// 4. Refine the mesh to increase the resolution.
// 5. Refine the mesh to increase the resolution.
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
}
// 5. Reorient mesh in case of a tet mesh
// 6. Reorient mesh in case of a tet mesh
mesh->ReorientTetMesh();
// Set element attributes in order to distinguish elements in the PML region
pml->SetAttributes(mesh);
// 6. Define a finite element space on the mesh. Here we use the Nedelec
// 7. Define a finite element space on the mesh. Here we use the Nedelec
// finite elements of the specified order.
FiniteElementCollection *fec = new ND_FECollection(order, dim);
FiniteElementSpace *fespace = new FiniteElementSpace(mesh, fec);
@@ -266,7 +281,7 @@ int main(int argc, char *argv[])
cout << "Number of finite element unknowns: " << size << endl;
// 7. Determine the list of true essential boundary dofs. In this example,
// 8. Determine the list of true essential boundary dofs. In this example,
// the boundary conditions are defined based on the specific mesh and the
// problem type.
Array<int> ess_tdof_list;
@@ -308,12 +323,12 @@ int main(int argc, char *argv[])
}
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
// 8. Setup Complex Operator convention
// 9. Setup Complex Operator convention
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 9. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system.
// 10. Set up the linear form b(.) which corresponds to the right-hand side of
// the FEM linear system.
VectorFunctionCoefficient f(dim, source);
ComplexLinearForm b(fespace, conv);
if (prob == load_src)
@@ -323,7 +338,7 @@ int main(int argc, char *argv[])
b.Vector::operator=(0.0);
b.Assemble();
// 10. Define the solution vector x as a complex finite element grid function
// 11. Define the solution vector x as a complex finite element grid function
// corresponding to fespace.
ComplexGridFunction x(fespace);
x = 0.0;
@@ -331,7 +346,7 @@ int main(int argc, char *argv[])
VectorFunctionCoefficient E_Im(dim, E_bdr_data_Im);
x.ProjectBdrCoefficientTangent(E_Re, E_Im, ess_bdr);
// 11. Set up the sesquilinear form a(.,.)
// 12. Set up the sesquilinear form a(.,.)
//
// In Comp
// Domain: 1/mu (Curl E, Curl F) - omega^2 * epsilon (E,F)
@@ -385,26 +400,30 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(restr_c2_Re),
new VectorFEMassIntegrator(restr_c2_Im));
// 12. Assemble the bilinear form and the corresponding linear system,
// 13. Assemble the bilinear form and the corresponding linear system,
// applying any necessary transformations such as: assembly, eliminating
// boundary conditions, applying conforming constraints for
// non-conforming AMR, etc.
#ifndef MFEM_USE_SUITESPARSE
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
#endif
a.Assemble(0);
OperatorPtr A;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
// 13. Solve using a direct or an iterative solver
// 14. Solve using a direct or an iterative solver
#ifdef MFEM_USE_SUITESPARSE
{
if (pa) { cout << "PA not available with MFEM_USE_SUITESPARSE" << endl; }
ComplexUMFPackSolver csolver(*A.As<ComplexSparseMatrix>());
csolver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
csolver.SetPrintLevel(1);
csolver.Mult(B, X);
}
#else
// 13a. Set up the Bilinear form a(.,.) for the preconditioner
// 14a. Set up the Bilinear form a(.,.) for the preconditioner
//
// In Comp
// Domain: 1/mu (Curl E, Curl F) + omega^2 * epsilon (E,F)
@@ -430,39 +449,58 @@ int main(int argc, char *argv[])
prec.AddDomainIntegrator(new CurlCurlIntegrator(restr_c1_abs));
prec.AddDomainIntegrator(new VectorFEMassIntegrator(restr_c2_abs));
if (pa) { prec.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
prec.Assemble();
OperatorPtr PCOpAh;
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// 13b. Define and apply a GMRES solver for AU=B with a block diagonal
// preconditioner based on the Gauss-Seidel sparse smoother.
// 14b. Define and apply a GMRES solver for AU=B with a block diagonal
// preconditioner based on the Gauss-Seidel or Jacobi sparse smoother.
Array<int> offsets(3);
offsets[0] = 0;
offsets[1] = fespace->GetTrueVSize();
offsets[2] = fespace->GetTrueVSize();
offsets.PartialSum();
GSSmoother gs00(*PCOpAh.As<SparseMatrix>());
BlockDiagonalPreconditioner BlockGS(offsets);
ScaledOperator gs11(&gs00,
(conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0);
BlockGS.SetDiagonalBlock(0,&gs00);
BlockGS.SetDiagonalBlock(1,&gs11);
Operator *pc_r = nullptr;
Operator *pc_i = nullptr;
int s = (conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0;
if (pa)
{
// Jacobi Smoother
OperatorJacobiSmoother *d00 = new OperatorJacobiSmoother(prec, ess_tdof_list);
ScaledOperator *d11 = new ScaledOperator(d00, s);
pc_r = d00;
pc_i = d11;
}
else
{
OperatorPtr PCOpAh;
prec.SetDiagonalPolicy(mfem::Operator::DIAG_ONE);
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// Gauss-Seidel Smoother
GSSmoother *gs00 = new GSSmoother(*PCOpAh.As<SparseMatrix>());
ScaledOperator *gs11 = new ScaledOperator(gs00, s);
pc_r = gs00;
pc_i = gs11;
}
BlockDiagonalPreconditioner BlockDP(offsets);
BlockDP.SetDiagonalBlock(0, pc_r);
BlockDP.SetDiagonalBlock(1, pc_i);
GMRESSolver gmres;
gmres.SetPrintLevel(1);
gmres.SetKDim(200);
gmres.SetMaxIter(2000);
gmres.SetMaxIter(pa ? 5000 : 2000);
gmres.SetRelTol(1e-5);
gmres.SetAbsTol(0.0);
gmres.SetOperator(*A);
gmres.SetPreconditioner(BlockGS);
gmres.SetPreconditioner(BlockDP);
gmres.Mult(B, X);
}
#endif
// 14. Recover the solution as a finite element grid function and compute the
// 15. Recover the solution as a finite element grid function and compute the
// errors if the exact solution is known.
a.RecoverFEMSolution(X, b, x);
@@ -499,7 +537,7 @@ int main(int argc, char *argv[])
<< sqrt(L2Error_Re*L2Error_Re + L2Error_Im*L2Error_Im) << "\n\n";
}
// 15. Save the refined mesh and the solution. This output can be viewed
// 16. Save the refined mesh and the solution. This output can be viewed
// later using GLVis: "glvis -m mesh -g sol".
{
ofstream mesh_ofs("ex25.mesh");
@@ -514,7 +552,7 @@ int main(int argc, char *argv[])
x.imag().Save(sol_i_ofs);
}
// 16. Send the solution by socket to a GLVis server.
// 17. Send the solution by socket to a GLVis server.
if (visualization)
{
// Define visualization keys for GLVis (see GLVis documentation)
@@ -565,7 +603,7 @@ int main(int argc, char *argv[])
}
}
// 17. Free the used memory.
// 18. Free the used memory.
delete pml;
delete fespace;
delete fec;
+64 -27
View File
@@ -10,6 +10,10 @@
// mpirun -np 4 ex25p -o 2 -f 8.0 -rs 2 -rp 2 -prob 4 -m ../data/inline-quad.mesh
// mpirun -np 4 ex25p -o 2 -f 2.0 -rs 1 -rp 1 -prob 4 -m ../data/inline-hex.mesh
//
// Device sample runs:
// mpirun -np 4 ex25p -o 1 -f 3.0 -rs 3 -rp 1 -prob 2 -pa -d cuda
// mpirun -np 4 ex25p -o 2 -f 1.0 -rs 1 -rp 1 -prob 3 -pa -d cuda
//
// Description: This example code solves a simple electromagnetic wave
// propagation problem corresponding to the second order
// indefinite Maxwell equation
@@ -160,6 +164,8 @@ int main(int argc, char *argv[])
double freq = 5.0;
bool herm_conv = true;
bool visualization = 1;
bool pa = false;
const char *device_config = "cpu";
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -183,12 +189,21 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly", "Enable Partial Assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.Parse();
if (iprob > 4) { iprob = 4; }
prob = (prob_type)iprob;
// 3. Setup the (serial) mesh on all processors.
// 3. Enable hardware devices such as GPUs, and programming models such as
// CUDA, OCCA, RAJA and OpenMP based on command line options.
Device device(device_config);
device.Print();
// 4. Setup the (serial) mesh on all processors.
if (!mesh_file)
{
exact_known = true;
@@ -236,7 +251,7 @@ int main(int argc, char *argv[])
// Setup PML length
Array2D<double> length(dim, 2); length = 0.0;
// 4. Setup the Cartesian PML region.
// 5. Setup the Cartesian PML region.
switch (prob)
{
case disc:
@@ -262,13 +277,13 @@ int main(int argc, char *argv[])
comp_domain_bdr = pml->GetCompDomainBdr();
domain_bdr = pml->GetDomainBdr();
// 5. Refine the serial mesh on all processors to increase the resolution.
// 6. Refine the serial mesh on all processors to increase the resolution.
for (int l = 0; l < ref_levels; l++)
{
mesh->UniformRefinement();
}
// 6. Define a parallel mesh by a partitioning of the serial mesh.
// 7. Define a parallel mesh by a partitioning of the serial mesh.
ParMesh *pmesh = new ParMesh(MPI_COMM_WORLD, *mesh);
delete mesh;
{
@@ -278,13 +293,13 @@ int main(int argc, char *argv[])
}
}
// 6a. Reorient mesh in case of a tet mesh
// 7a. Reorient mesh in case of a tet mesh
pmesh->ReorientTetMesh();
// 7. Set element attributes in order to distinguish elements in the PML
// 8. Set element attributes in order to distinguish elements in the PML
pml->SetAttributes(pmesh);
// 8. Define a parallel finite element space on the parallel mesh. Here we
// 9. Define a parallel finite element space on the parallel mesh. Here we
// use the Nedelec finite elements of the specified order.
FiniteElementCollection *fec = new ND_FECollection(order, dim);
ParFiniteElementSpace *fespace = new ParFiniteElementSpace(pmesh, fec);
@@ -294,9 +309,9 @@ int main(int argc, char *argv[])
cout << "Number of finite element unknowns: " << size << endl;
}
// 9. Determine the list of true (i.e. parallel conforming) essential
// boundary dofs. In this example, the boundary conditions are defined
// based on the specific mesh and the problem type.
// 10. Determine the list of true (i.e. parallel conforming) essential
// boundary dofs. In this example, the boundary conditions are defined
// based on the specific mesh and the problem type.
Array<int> ess_tdof_list;
Array<int> ess_bdr;
if (pmesh->bdr_attributes.Size())
@@ -336,11 +351,11 @@ int main(int argc, char *argv[])
}
fespace->GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
// 10. Setup Complex Operator convention
// 11. Setup Complex Operator convention
ComplexOperator::Convention conv =
herm_conv ? ComplexOperator::HERMITIAN : ComplexOperator::BLOCK_SYMMETRIC;
// 11. Set up the parallel linear form b(.) which corresponds to the
// 12. Set up the parallel linear form b(.) which corresponds to the
// right-hand side of the FEM linear system.
VectorFunctionCoefficient f(dim, source);
ParComplexLinearForm b(fespace, conv);
@@ -351,7 +366,7 @@ int main(int argc, char *argv[])
b.Vector::operator=(0.0);
b.Assemble();
// 12. Define the solution vector x as a parallel complex finite element grid
// 13. Define the solution vector x as a parallel complex finite element grid
// function corresponding to fespace.
ParComplexGridFunction x(fespace);
x = 0.0;
@@ -359,7 +374,7 @@ int main(int argc, char *argv[])
VectorFunctionCoefficient E_Im(dim, E_bdr_data_Im);
x.ProjectBdrCoefficientTangent(E_Re, E_Im, ess_bdr);
// 13. Set up the parallel sesquilinear form a(.,.)
// 14. Set up the parallel sesquilinear form a(.,.)
//
// In Comp
// Domain: 1/mu (Curl E, Curl F) - omega^2 * epsilon (E,F)
@@ -413,19 +428,23 @@ int main(int argc, char *argv[])
a.AddDomainIntegrator(new VectorFEMassIntegrator(restr_c2_Re),
new VectorFEMassIntegrator(restr_c2_Im));
// 14. Assemble the parallel bilinear form and the corresponding linear
// 15. Assemble the parallel bilinear form and the corresponding linear
// system, applying any necessary transformations such as: parallel
// assembly, eliminating boundary conditions, applying conforming
// constraints for non-conforming AMR, etc.
#ifndef MFEM_USE_SUPERLU
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
#endif
a.Assemble();
OperatorPtr Ah;
Vector B, X;
a.FormLinearSystem(ess_tdof_list, x, b, Ah, X, B);
// 15. Solve using a direct or an iterative solver
// 16. Solve using a direct or an iterative solver
#ifdef MFEM_USE_SUPERLU
{
if (pa) { cout << "PA not available with MFEM_USE_SUPERLU" << endl; }
// Transform to monolithic HypreParMatrix
HypreParMatrix *A = Ah.As<ComplexHypreParMatrix>()->GetSystemMatrix();
SuperLURowLocMatrix SA(*A);
@@ -464,11 +483,9 @@ int main(int argc, char *argv[])
prec.AddDomainIntegrator(new CurlCurlIntegrator(restr_c1_abs));
prec.AddDomainIntegrator(new VectorFEMassIntegrator(restr_c2_abs));
if (pa) { prec.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
prec.Assemble();
OperatorPtr PCOpAh;
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// 16b. Define and apply a parallel GMRES solver for AU=B with a block
// diagonal preconditioner based on hypre's AMS preconditioner.
Array<int> offsets(3);
@@ -477,21 +494,41 @@ int main(int argc, char *argv[])
offsets[2] = fespace->GetTrueVSize();
offsets.PartialSum();
HypreAMS ams00(*PCOpAh.As<HypreParMatrix>(),fespace);
BlockDiagonalPreconditioner BlockAMS(offsets);
ScaledOperator ams11(&ams00,
(conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0);
BlockAMS.SetDiagonalBlock(0,&ams00);
BlockAMS.SetDiagonalBlock(1,&ams11);
Operator *pc_r = nullptr;
Operator *pc_i = nullptr;
int s = (conv == ComplexOperator::HERMITIAN) ? -1.0 : 1.0;
if (pa)
{
// Jacobi Smoother
OperatorJacobiSmoother *d00 = new OperatorJacobiSmoother(prec, ess_tdof_list);
ScaledOperator *d11 = new ScaledOperator(d00, s);
pc_r = d00;
pc_i = d11;
}
else
{
OperatorPtr PCOpAh;
prec.FormSystemMatrix(ess_tdof_list, PCOpAh);
// Hypre AMS
HypreAMS *ams00 = new HypreAMS(*PCOpAh.As<HypreParMatrix>(), fespace);
ScaledOperator *ams11 = new ScaledOperator(ams00, s);
pc_r = ams00;
pc_i = ams11;
}
BlockDiagonalPreconditioner BlockDP(offsets);
BlockDP.SetDiagonalBlock(0, pc_r);
BlockDP.SetDiagonalBlock(1, pc_i);
GMRESSolver gmres(MPI_COMM_WORLD);
gmres.SetPrintLevel(1);
gmres.SetKDim(200);
gmres.SetMaxIter(2000);
gmres.SetMaxIter(pa ? 5000 : 2000);
gmres.SetRelTol(1e-5);
gmres.SetAbsTol(0.0);
gmres.SetOperator(*Ah);
gmres.SetPreconditioner(BlockAMS);
gmres.SetPreconditioner(BlockDP);
gmres.Mult(B, X);
}
#endif
+2 -12
View File
@@ -60,7 +60,6 @@ int main(int argc, char *argv[])
bool static_cond = false;
bool visualization = 1;
bool amg_elast = 0;
bool reorder_space = false;
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh",
@@ -76,8 +75,6 @@ int main(int argc, char *argv[])
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&reorder_space, "-nodes", "--by-nodes", "-vdim", "--by-vdim",
"Use byNODES ordering of vector space instead of byVDIM");
args.Parse();
if (!args.Good())
{
@@ -159,14 +156,7 @@ int main(int argc, char *argv[])
else
{
fec = new H1_FECollection(order, dim);
if (reorder_space)
{
fespace = new ParFiniteElementSpace(pmesh, fec, dim, Ordering::byNODES);
}
else
{
fespace = new ParFiniteElementSpace(pmesh, fec, dim, Ordering::byVDIM);
}
fespace = new ParFiniteElementSpace(pmesh, fec, dim, Ordering::byVDIM);
}
HYPRE_Int size = fespace->GlobalTrueVSize();
if (myid == 0)
@@ -259,7 +249,7 @@ int main(int argc, char *argv[])
}
else
{
amg->SetSystemsOptions(dim, reorder_space);
amg->SetSystemsOptions(dim);
}
HyprePCG *pcg = new HyprePCG(A);
pcg->SetTol(1e-8);
+3 -8
View File
@@ -108,11 +108,7 @@ int main(int argc, char *argv[])
// the Laplace problem -\Delta u = 1. We don't assemble the discrete
// problem yet, this will be done in the main loop.
BilinearForm a(&fespace);
if (pa)
{
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.SetDiagonalPolicy(Operator::DIAG_ONE);
}
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
LinearForm b(&fespace);
ConstantCoefficient one(1.0);
@@ -203,10 +199,9 @@ int main(int argc, char *argv[])
umf_solver.Mult(B, X);
#endif
}
else // Diagonal preconditioning in partial assembly mode.
else // No preconditioning for now in partial assembly mode.
{
OperatorJacobiSmoother M(a, ess_tdof_list);
PCG(*A, M, B, X, 3, 2000, 1e-12, 0.0);
CG(*A, B, X, 3, 2000, 1e-12, 0.0);
}
// 18. After solving the linear system, reconstruct the solution as a
+6 -19
View File
@@ -129,11 +129,7 @@ int main(int argc, char *argv[])
// the Laplace problem -\Delta u = 1. We don't assemble the discrete
// problem yet, this will be done in the main loop.
ParBilinearForm a(&fespace);
if (pa)
{
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.SetDiagonalPolicy(Operator::DIAG_ONE);
}
if (pa) { a.SetAssemblyLevel(AssemblyLevel::PARTIAL); }
ParLinearForm b(&fespace);
ConstantCoefficient one(1.0);
@@ -224,26 +220,17 @@ int main(int argc, char *argv[])
// 17. Solve the linear system A X = B.
// * With full assembly, use the BoomerAMG preconditioner from hypre.
// * With partial assembly, use a diagonal preconditioner.
Solver *M = NULL;
if (pa)
{
M = new OperatorJacobiSmoother(a, ess_tdof_list);
}
else
{
HypreBoomerAMG *amg = new HypreBoomerAMG;
amg->SetPrintLevel(0);
M = amg;
}
// * With partial assembly, use no preconditioner, for now.
HypreBoomerAMG *amg = NULL;
if (!pa) { amg = new HypreBoomerAMG; amg->SetPrintLevel(0); }
CGSolver cg(MPI_COMM_WORLD);
cg.SetRelTol(1e-6);
cg.SetMaxIter(2000);
cg.SetPrintLevel(3); // print the first and the last iterations only
cg.SetPreconditioner(*M);
if (amg) { cg.SetPreconditioner(*amg); }
cg.SetOperator(*A);
cg.Mult(B, X);
delete M;
delete amg;
// 18. Switch back to the host and extract the parallel grid function
// corresponding to the finite element approximation X. This is the
-5
View File
@@ -119,11 +119,6 @@ ex11p-test-superlu: ex11p
@$(call mfem-test,$<, $(RUN_MPI), SuperLU_DIST example,--superlu)
test-par-YES: ex11p-test-superlu
endif
ifeq ($(MFEM_USE_MKL_CPARDISO),YES)
ex11p-test-superlu: ex11p
@$(call mfem-test,$<, $(RUN_MPI), MKL_CPARDISO example,--pardiso)
test-par-YES: ex11p-test-pardiso
endif
# Testing: "test" target and mfem-test* variables are defined in config/test.mk
-2
View File
@@ -31,7 +31,6 @@ set(SRCS
bilininteg_vecmass.cpp
coefficient.cpp
complex_fem.cpp
convergence.cpp
datacollection.cpp
eltrans.cpp
estimators.cpp
@@ -66,7 +65,6 @@ set(HDRS
bilininteg.hpp
coefficient.hpp
complex_fem.hpp
convergence.hpp
datacollection.hpp
eltrans.hpp
estimators.hpp
-27
View File
@@ -627,33 +627,6 @@ void BilinearForm::AssembleDiagonal(Vector &diag) const
MFEM_ASSERT(diag.Size() == fes->GetTrueVSize(),
"Vector for holding diagonal has wrong size!");
const Operator *P = fes->GetProlongationMatrix();
// For an AMR mesh, a convergent diagonal is assembled with |P^T| d_e,
// where |P^T| has the entry-wise absolute values of the conforming
// prolongation transpose operator.
if (P && !fes->Conforming())
{
Vector local_diag(P->Height());
ext->AssembleDiagonal(local_diag);
const SparseMatrix *SP = dynamic_cast<const SparseMatrix*>(P);
#ifdef MFEM_USE_MPI
const HypreParMatrix *HP = dynamic_cast<const HypreParMatrix*>(P);
#endif
if (SP)
{
SP->AbsMultTranspose(local_diag, diag);
}
#ifdef MFEM_USE_MPI
else if (HP)
{
HP->AbsMultTranspose(1.0, local_diag, 0.0, diag);
}
#endif
else
{
MFEM_ABORT("Prolongation matrix has unexpected type.");
}
return;
}
if (!IsIdentityProlongation(P))
{
Vector local_diag(P->Height());
+7 -17
View File
@@ -96,9 +96,6 @@ void PABilinearFormExtension::Assemble()
integrators[i]->AssemblePA(*a->FESpace());
}
MFEM_VERIFY(a->GetBBFI()->Size() == 0,
"Partial assembly does not support AddBoundaryIntegrator yet.");
Array<BilinearFormIntegrator*> &intFaceIntegrators = *a->GetFBFI();
const int intFaceIntegratorCount = intFaceIntegrators.Size();
for (int i = 0; i < intFaceIntegratorCount; ++i)
@@ -119,7 +116,7 @@ void PABilinearFormExtension::AssembleDiagonal(Vector &y) const
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int iSz = integrators.Size();
if (elem_restrict && !DeviceCanUseCeed())
if (elem_restrict)
{
localY = 0.0;
for (int i = 0; i < iSz; ++i)
@@ -310,21 +307,19 @@ void EABilinearFormExtension::Assemble()
ea_data.SetSize(ne*elemDofs*elemDofs, Device::GetMemoryType());
ea_data.UseDevice(true);
ea_data = 0.0;
Array<BilinearFormIntegrator*> &integrators = *a->GetDBFI();
const int integratorCount = integrators.Size();
for (int i = 0; i < integratorCount; ++i)
{
integrators[i]->AssembleEA(*a->FESpace(), ea_data, i);
integrators[i]->AssembleEA(*a->FESpace(), ea_data);
}
faceDofs = trialFes ->
GetTraceElement(0, trialFes->GetMesh()->GetFaceBaseGeometry(0)) ->
GetDof();
MFEM_VERIFY(a->GetBBFI()->Size() == 0,
"Element assembly does not support AddBoundaryIntegrator yet.");
Array<BilinearFormIntegrator*> &intFaceIntegrators = *a->GetFBFI();
const int intFaceIntegratorCount = intFaceIntegrators.Size();
if (intFaceIntegratorCount>0)
@@ -332,13 +327,14 @@ void EABilinearFormExtension::Assemble()
nf_int = trialFes->GetNFbyType(FaceType::Interior);
ea_data_int.SetSize(2*nf_int*faceDofs*faceDofs, Device::GetMemoryType());
ea_data_ext.SetSize(2*nf_int*faceDofs*faceDofs, Device::GetMemoryType());
ea_data_int = 0.0;
ea_data_ext = 0.0;
}
for (int i = 0; i < intFaceIntegratorCount; ++i)
{
intFaceIntegrators[i]->AssembleEAInteriorFaces(*a->FESpace(),
ea_data_int,
ea_data_ext,
i);
ea_data_ext);
}
Array<BilinearFormIntegrator*> &bdrFaceIntegrators = *a->GetBFBFI();
@@ -351,7 +347,7 @@ void EABilinearFormExtension::Assemble()
}
for (int i = 0; i < boundFaceIntegratorCount; ++i)
{
bdrFaceIntegrators[i]->AssembleEABoundaryFaces(*a->FESpace(),ea_data_bdr,i);
bdrFaceIntegrators[i]->AssembleEABoundaryFaces(*a->FESpace(),ea_data_bdr);
}
if (factorize_face_terms && int_face_restrict_lex)
@@ -798,12 +794,6 @@ void PAMixedBilinearFormExtension::Assemble()
{
integrators[i]->AssemblePA(*trialFes, *testFes);
}
MFEM_VERIFY(a->GetBBFI()->Size() == 0,
"Partial assembly does not support AddBoundaryIntegrator yet.");
MFEM_VERIFY(a->GetTFBFI()->Size() == 0,
"Partial assembly does not support AddTraceFaceIntegrator yet.");
MFEM_VERIFY(a->GetBTFBFI()->Size() == 0,
"Partial assembly does not support AddBdrTraceFaceIntegrator yet.");
}
void PAMixedBilinearFormExtension::Update()
+3 -6
View File
@@ -52,8 +52,7 @@ void BilinearFormIntegrator::AssembleDiagonalPA(Vector &)
}
void BilinearFormIntegrator::AssembleEA(const FiniteElementSpace &fes,
Vector &emat,
const bool add)
Vector &emat)
{
mfem_error ("BilinearFormIntegrator::AssembleEA(...)\n"
" is not implemented for this class.");
@@ -62,8 +61,7 @@ void BilinearFormIntegrator::AssembleEA(const FiniteElementSpace &fes,
void BilinearFormIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace
&fes,
Vector &ea_data_int,
Vector &ea_data_ext,
const bool add)
Vector &ea_data_ext)
{
mfem_error ("BilinearFormIntegrator::AssembleEAInteriorFaces(...)\n"
" is not implemented for this class.");
@@ -71,8 +69,7 @@ void BilinearFormIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace
void BilinearFormIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace
&fes,
Vector &ea_data_bdr,
const bool add)
Vector &ea_data_bdr)
{
mfem_error ("BilinearFormIntegrator::AssembleEABoundaryFaces(...)\n"
" is not implemented for this class.");
+15 -26
View File
@@ -86,10 +86,9 @@ public:
virtual void AddMultTransposePA(const Vector &x, Vector &y) const;
/// Method defining element assembly.
/** The result of the element assembly is added to the @a emat Vector if
@a add is true. Otherwise, if @a add is false, we set @a emat. */
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
const bool add = true);
/** The result of the element assembly is added and stored in the @a emat
Vector. */
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat);
/** Used with BilinearFormIntegrators that have different spaces. */
// virtual void AssembleEA(const FiniteElementSpace &trial_fes,
// const FiniteElementSpace &test_fes,
@@ -97,12 +96,10 @@ public:
virtual void AssembleEAInteriorFaces(const FiniteElementSpace &fes,
Vector &ea_data_int,
Vector &ea_data_ext,
const bool add = true);
Vector &ea_data_ext);
virtual void AssembleEABoundaryFaces(const FiniteElementSpace &fes,
Vector &ea_data_bdr,
const bool add = true);
Vector &ea_data_bdr);
/// Given a particular Finite Element computes the element matrix elmat.
virtual void AssembleElementMatrix(const FiniteElement &el,
@@ -265,17 +262,14 @@ public:
bfi->AddMultTransposePA(x, y);
}
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
const bool add);
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat);
virtual void AssembleEAInteriorFaces(const FiniteElementSpace &fes,
Vector &ea_data_int,
Vector &ea_data_ext,
const bool add);
Vector &ea_data_ext);
virtual void AssembleEABoundaryFaces(const FiniteElementSpace &fes,
Vector &ea_data_bdr,
const bool add);
Vector &ea_data_bdr);
virtual ~TransposeIntegrator() { if (own_bfi) { delete bfi; } }
};
@@ -1958,8 +1952,7 @@ public:
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
const bool add);
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat);
virtual void AssembleDiagonalPA(Vector &diag);
@@ -1968,7 +1961,7 @@ public:
static const IntegrationRule &GetRule(const FiniteElement &trial_fe,
const FiniteElement &test_fe);
void SetupPA(const FiniteElementSpace &fes);
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
};
/** Class for local mass matrix assembling a(u,v) := (Q u, v) */
@@ -2034,8 +2027,7 @@ public:
virtual void AssemblePA(const FiniteElementSpace &fes);
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
const bool add);
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat);
virtual void AssembleDiagonalPA(Vector &diag);
@@ -2045,7 +2037,7 @@ public:
const FiniteElement &test_fe,
ElementTransformation &Trans);
void SetupPA(const FiniteElementSpace &fes);
void SetupPA(const FiniteElementSpace &fes, const bool force = false);
};
/** Mass integrator (u, v) restricted to the boundary of a domain */
@@ -2091,8 +2083,7 @@ public:
virtual void AssemblePA(const FiniteElementSpace&);
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat,
const bool add);
virtual void AssembleEA(const FiniteElementSpace &fes, Vector &emat);
virtual void AddMultPA(const Vector&, Vector&) const;
@@ -2669,12 +2660,10 @@ public:
virtual void AssembleEAInteriorFaces(const FiniteElementSpace& fes,
Vector &ea_data_int,
Vector &ea_data_ext,
const bool add);
Vector &ea_data_ext);
virtual void AssembleEABoundaryFaces(const FiniteElementSpace& fes,
Vector &ea_data_bdr,
const bool add);
Vector &ea_data_bdr);
static const IntegrationRule &GetRule(Geometry::Type geom, int order,
FaceElementTransformations &T);
+30 -58
View File
@@ -22,7 +22,6 @@ static void EAConvectionAssemble1D(const int NE,
const Array<double> &g,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -55,14 +54,7 @@ static void EAConvectionAssemble1D(const int NE,
{
val += r_Bj[k1] * D(k1, e) * r_Gi[k1];
}
if (add)
{
A(i1, j1, e) += val;
}
else
{
A(i1, j1, e) = val;
}
A(i1, j1, e) += val;
}
}
});
@@ -74,7 +66,6 @@ static void EAConvectionAssemble2D(const int NE,
const Array<double> &g,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -130,14 +121,7 @@ static void EAConvectionAssemble2D(const int NE,
* r_B[k1][j1]* r_B[k2][j2];
}
}
if (add)
{
A(i1, i2, j1, j2, e) += val;
}
else
{
A(i1, i2, j1, j2, e) = val;
}
A(i1, i2, j1, j2, e) += val;
}
}
}
@@ -151,7 +135,6 @@ static void EAConvectionAssemble3D(const int NE,
const Array<double> &g,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -208,14 +191,7 @@ static void EAConvectionAssemble3D(const int NE,
}
}
}
if (add)
{
A(i1, i2, i3, j1, j2, j3, e) += val;
}
else
{
A(i1, i2, i3, j1, j2, j3, e) = val;
}
A(i1, i2, i3, j1, j2, j3, e) += val;
}
}
}
@@ -226,8 +202,7 @@ static void EAConvectionAssemble3D(const int NE,
}
void ConvectionIntegrator::AssembleEA(const FiniteElementSpace &fes,
Vector &ea_data,
const bool add)
Vector &ea_data)
{
AssemblePA(fes);
const int ne = fes.GetMesh()->GetNE();
@@ -237,47 +212,44 @@ void ConvectionIntegrator::AssembleEA(const FiniteElementSpace &fes,
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x22: return EAConvectionAssemble1D<2,2>(ne,B,G,pa_data,ea_data,add);
case 0x33: return EAConvectionAssemble1D<3,3>(ne,B,G,pa_data,ea_data,add);
case 0x44: return EAConvectionAssemble1D<4,4>(ne,B,G,pa_data,ea_data,add);
case 0x55: return EAConvectionAssemble1D<5,5>(ne,B,G,pa_data,ea_data,add);
case 0x66: return EAConvectionAssemble1D<6,6>(ne,B,G,pa_data,ea_data,add);
case 0x77: return EAConvectionAssemble1D<7,7>(ne,B,G,pa_data,ea_data,add);
case 0x88: return EAConvectionAssemble1D<8,8>(ne,B,G,pa_data,ea_data,add);
case 0x99: return EAConvectionAssemble1D<9,9>(ne,B,G,pa_data,ea_data,add);
default: return EAConvectionAssemble1D(ne,B,G,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x22: return EAConvectionAssemble1D<2,2>(ne,B,G,pa_data,ea_data);
case 0x33: return EAConvectionAssemble1D<3,3>(ne,B,G,pa_data,ea_data);
case 0x44: return EAConvectionAssemble1D<4,4>(ne,B,G,pa_data,ea_data);
case 0x55: return EAConvectionAssemble1D<5,5>(ne,B,G,pa_data,ea_data);
case 0x66: return EAConvectionAssemble1D<6,6>(ne,B,G,pa_data,ea_data);
case 0x77: return EAConvectionAssemble1D<7,7>(ne,B,G,pa_data,ea_data);
case 0x88: return EAConvectionAssemble1D<8,8>(ne,B,G,pa_data,ea_data);
case 0x99: return EAConvectionAssemble1D<9,9>(ne,B,G,pa_data,ea_data);
default: return EAConvectionAssemble1D(ne,B,G,pa_data,ea_data,dofs1D,quad1D);
}
}
else if (dim == 2)
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x22: return EAConvectionAssemble2D<2,2>(ne,B,G,pa_data,ea_data,add);
case 0x33: return EAConvectionAssemble2D<3,3>(ne,B,G,pa_data,ea_data,add);
case 0x44: return EAConvectionAssemble2D<4,4>(ne,B,G,pa_data,ea_data,add);
case 0x55: return EAConvectionAssemble2D<5,5>(ne,B,G,pa_data,ea_data,add);
case 0x66: return EAConvectionAssemble2D<6,6>(ne,B,G,pa_data,ea_data,add);
case 0x77: return EAConvectionAssemble2D<7,7>(ne,B,G,pa_data,ea_data,add);
case 0x88: return EAConvectionAssemble2D<8,8>(ne,B,G,pa_data,ea_data,add);
case 0x99: return EAConvectionAssemble2D<9,9>(ne,B,G,pa_data,ea_data,add);
default: return EAConvectionAssemble2D(ne,B,G,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x22: return EAConvectionAssemble2D<2,2>(ne,B,G,pa_data,ea_data);
case 0x33: return EAConvectionAssemble2D<3,3>(ne,B,G,pa_data,ea_data);
case 0x44: return EAConvectionAssemble2D<4,4>(ne,B,G,pa_data,ea_data);
case 0x55: return EAConvectionAssemble2D<5,5>(ne,B,G,pa_data,ea_data);
case 0x66: return EAConvectionAssemble2D<6,6>(ne,B,G,pa_data,ea_data);
case 0x77: return EAConvectionAssemble2D<7,7>(ne,B,G,pa_data,ea_data);
case 0x88: return EAConvectionAssemble2D<8,8>(ne,B,G,pa_data,ea_data);
case 0x99: return EAConvectionAssemble2D<9,9>(ne,B,G,pa_data,ea_data);
default: return EAConvectionAssemble2D(ne,B,G,pa_data,ea_data,dofs1D,quad1D);
}
}
else if (dim == 3)
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x23: return EAConvectionAssemble3D<2,3>(ne,B,G,pa_data,ea_data,add);
case 0x34: return EAConvectionAssemble3D<3,4>(ne,B,G,pa_data,ea_data,add);
case 0x45: return EAConvectionAssemble3D<4,5>(ne,B,G,pa_data,ea_data,add);
case 0x56: return EAConvectionAssemble3D<5,6>(ne,B,G,pa_data,ea_data,add);
case 0x67: return EAConvectionAssemble3D<6,7>(ne,B,G,pa_data,ea_data,add);
case 0x78: return EAConvectionAssemble3D<7,8>(ne,B,G,pa_data,ea_data,add);
case 0x89: return EAConvectionAssemble3D<8,9>(ne,B,G,pa_data,ea_data,add);
default: return EAConvectionAssemble3D(ne,B,G,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x23: return EAConvectionAssemble3D<2,3>(ne,B,G,pa_data,ea_data);
case 0x34: return EAConvectionAssemble3D<3,4>(ne,B,G,pa_data,ea_data);
case 0x45: return EAConvectionAssemble3D<4,5>(ne,B,G,pa_data,ea_data);
case 0x56: return EAConvectionAssemble3D<5,6>(ne,B,G,pa_data,ea_data);
case 0x67: return EAConvectionAssemble3D<6,7>(ne,B,G,pa_data,ea_data);
case 0x78: return EAConvectionAssemble3D<7,8>(ne,B,G,pa_data,ea_data);
case 0x89: return EAConvectionAssemble3D<8,9>(ne,B,G,pa_data,ea_data);
default: return EAConvectionAssemble3D(ne,B,G,pa_data,ea_data,dofs1D,quad1D);
}
}
MFEM_ABORT("Unknown kernel.");
+55 -114
View File
@@ -20,8 +20,7 @@ static void EADGTraceAssemble1DInt(const int NF,
const Array<double> &basis,
const Vector &padata,
Vector &eadata_int,
Vector &eadata_ext,
const bool add)
Vector &eadata_ext)
{
auto D = Reshape(padata.Read(), 2, 2, NF);
auto A_int = Reshape(eadata_int.ReadWrite(), 2, NF);
@@ -33,41 +32,23 @@ static void EADGTraceAssemble1DInt(const int NF,
val_ext10 = D(1, 0, f);
val_ext01 = D(0, 1, f);
val_int1 = D(1, 1, f);
if (add)
{
A_int(0, f) += val_int0;
A_int(1, f) += val_int1;
A_ext(0, f) += val_ext01;
A_ext(1, f) += val_ext10;
}
else
{
A_int(0, f) = val_int0;
A_int(1, f) = val_int1;
A_ext(0, f) = val_ext01;
A_ext(1, f) = val_ext10;
}
A_int(0, f) += val_int0;
A_int(1, f) += val_int1;
A_ext(0, f) += val_ext01;
A_ext(1, f) += val_ext10;
});
}
static void EADGTraceAssemble1DBdr(const int NF,
const Array<double> &basis,
const Vector &padata,
Vector &eadata_bdr,
const bool add)
Vector &eadata_bdr)
{
auto D = Reshape(padata.Read(), 2, 2, NF);
auto A_bdr = Reshape(eadata_bdr.ReadWrite(), NF);
MFEM_FORALL(f, NF,
{
if (add)
{
A_bdr(f) += D(0, 0, f);
}
else
{
A_bdr(f) = D(0, 0, f);
}
A_bdr(f) += D(0, 0, f);
});
}
@@ -77,7 +58,6 @@ static void EADGTraceAssemble2DInt(const int NF,
const Vector &padata,
Vector &eadata_int,
Vector &eadata_ext,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -108,20 +88,10 @@ static void EADGTraceAssemble2DInt(const int NF,
val_ext10 += B(k1,i1) * B(k1,j1) * D(k1, 1, 0, f);
val_int1 += B(k1,i1) * B(k1,j1) * D(k1, 1, 1, f);
}
if (add)
{
A_int(i1, j1, 0, f) += val_int0;
A_int(i1, j1, 1, f) += val_int1;
A_ext(i1, j1, 0, f) += val_ext01;
A_ext(i1, j1, 1, f) += val_ext10;
}
else
{
A_int(i1, j1, 0, f) = val_int0;
A_int(i1, j1, 1, f) = val_int1;
A_ext(i1, j1, 0, f) = val_ext01;
A_ext(i1, j1, 1, f) = val_ext10;
}
A_int(i1, j1, 0, f) += val_int0;
A_int(i1, j1, 1, f) += val_int1;
A_ext(i1, j1, 0, f) += val_ext01;
A_ext(i1, j1, 1, f) += val_ext10;
}
}
});
@@ -132,7 +102,6 @@ static void EADGTraceAssemble2DBdr(const int NF,
const Array<double> &basis,
const Vector &padata,
Vector &eadata_bdr,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -156,14 +125,7 @@ static void EADGTraceAssemble2DBdr(const int NF,
{
val_bdr += B(k1,i1) * B(k1,j1) * D(k1, 0, 0, f);
}
if (add)
{
A_bdr(i1, j1, f) += val_bdr;
}
else
{
A_bdr(i1, j1, f) = val_bdr;
}
A_bdr(i1, j1, f) += val_bdr;
}
}
});
@@ -175,7 +137,6 @@ static void EADGTraceAssemble3DInt(const int NF,
const Vector &padata,
Vector &eadata_int,
Vector &eadata_ext,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -246,20 +207,10 @@ static void EADGTraceAssemble3DInt(const int NF,
* s_D[k1][k2][1][0];
}
}
if (add)
{
A_int(i1, i2, j1, j2, 0, f) += val_int0;
A_int(i1, i2, j1, j2, 1, f) += val_int1;
A_ext(i1, i2, j1, j2, 0, f) += val_ext01;
A_ext(i1, i2, j1, j2, 1, f) += val_ext10;
}
else
{
A_int(i1, i2, j1, j2, 0, f) = val_int0;
A_int(i1, i2, j1, j2, 1, f) = val_int1;
A_ext(i1, i2, j1, j2, 0, f) = val_ext01;
A_ext(i1, i2, j1, j2, 1, f) = val_ext10;
}
A_int(i1, i2, j1, j2, 0, f) += val_int0;
A_int(i1, i2, j1, j2, 1, f) += val_int1;
A_ext(i1, i2, j1, j2, 0, f) += val_ext01;
A_ext(i1, i2, j1, j2, 1, f) += val_ext10;
}
}
}
@@ -272,7 +223,6 @@ static void EADGTraceAssemble3DBdr(const int NF,
const Array<double> &basis,
const Vector &padata,
Vector &eadata_bdr,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -330,14 +280,7 @@ static void EADGTraceAssemble3DBdr(const int NF,
* s_D[k1][k2][0][0];
}
}
if (add)
{
A_bdr(i1, i2, j1, j2, f) += val_bdr;
}
else
{
A_bdr(i1, i2, j1, j2, f) = val_bdr;
}
A_bdr(i1, i2, j1, j2, f) += val_bdr;
}
}
}
@@ -347,8 +290,7 @@ static void EADGTraceAssemble3DBdr(const int NF,
void DGTraceIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace& fes,
Vector &ea_data_int,
Vector &ea_data_ext,
const bool add)
Vector &ea_data_ext)
{
SetupPA(fes, FaceType::Interior);
nf = fes.GetNFbyType(FaceType::Interior);
@@ -356,7 +298,7 @@ void DGTraceIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace& fes,
const Array<double> &B = maps->B;
if (dim == 1)
{
return EADGTraceAssemble1DInt(nf,B,pa_data,ea_data_int,ea_data_ext,add);
return EADGTraceAssemble1DInt(nf,B,pa_data,ea_data_int,ea_data_ext);
}
else if (dim == 2)
{
@@ -364,31 +306,31 @@ void DGTraceIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace& fes,
{
case 0x22:
return EADGTraceAssemble2DInt<2,2>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x33:
return EADGTraceAssemble2DInt<3,3>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x44:
return EADGTraceAssemble2DInt<4,4>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x55:
return EADGTraceAssemble2DInt<5,5>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x66:
return EADGTraceAssemble2DInt<6,6>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x77:
return EADGTraceAssemble2DInt<7,7>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x88:
return EADGTraceAssemble2DInt<8,8>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x99:
return EADGTraceAssemble2DInt<9,9>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
default:
return EADGTraceAssemble2DInt(nf,B,pa_data,ea_data_int,
ea_data_ext,add,dofs1D,quad1D);
ea_data_ext,dofs1D,quad1D);
}
}
else if (dim == 3)
@@ -397,36 +339,35 @@ void DGTraceIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace& fes,
{
case 0x23:
return EADGTraceAssemble3DInt<2,3>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x34:
return EADGTraceAssemble3DInt<3,4>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x45:
return EADGTraceAssemble3DInt<4,5>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x56:
return EADGTraceAssemble3DInt<5,6>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x67:
return EADGTraceAssemble3DInt<6,7>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x78:
return EADGTraceAssemble3DInt<7,8>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
case 0x89:
return EADGTraceAssemble3DInt<8,9>(nf,B,pa_data,ea_data_int,
ea_data_ext,add);
ea_data_ext);
default:
return EADGTraceAssemble3DInt(nf,B,pa_data,ea_data_int,
ea_data_ext,add,dofs1D,quad1D);
ea_data_ext,dofs1D,quad1D);
}
}
MFEM_ABORT("Unknown kernel.");
}
void DGTraceIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace& fes,
Vector &ea_data_bdr,
const bool add)
Vector &ea_data_bdr)
{
SetupPA(fes, FaceType::Boundary);
nf = fes.GetNFbyType(FaceType::Boundary);
@@ -434,37 +375,37 @@ void DGTraceIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace& fes,
const Array<double> &B = maps->B;
if (dim == 1)
{
return EADGTraceAssemble1DBdr(nf,B,pa_data,ea_data_bdr,add);
return EADGTraceAssemble1DBdr(nf,B,pa_data,ea_data_bdr);
}
else if (dim == 2)
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x22: return EADGTraceAssemble2DBdr<2,2>(nf,B,pa_data,ea_data_bdr,add);
case 0x33: return EADGTraceAssemble2DBdr<3,3>(nf,B,pa_data,ea_data_bdr,add);
case 0x44: return EADGTraceAssemble2DBdr<4,4>(nf,B,pa_data,ea_data_bdr,add);
case 0x55: return EADGTraceAssemble2DBdr<5,5>(nf,B,pa_data,ea_data_bdr,add);
case 0x66: return EADGTraceAssemble2DBdr<6,6>(nf,B,pa_data,ea_data_bdr,add);
case 0x77: return EADGTraceAssemble2DBdr<7,7>(nf,B,pa_data,ea_data_bdr,add);
case 0x88: return EADGTraceAssemble2DBdr<8,8>(nf,B,pa_data,ea_data_bdr,add);
case 0x99: return EADGTraceAssemble2DBdr<9,9>(nf,B,pa_data,ea_data_bdr,add);
case 0x22: return EADGTraceAssemble2DBdr<2,2>(nf,B,pa_data,ea_data_bdr);
case 0x33: return EADGTraceAssemble2DBdr<3,3>(nf,B,pa_data,ea_data_bdr);
case 0x44: return EADGTraceAssemble2DBdr<4,4>(nf,B,pa_data,ea_data_bdr);
case 0x55: return EADGTraceAssemble2DBdr<5,5>(nf,B,pa_data,ea_data_bdr);
case 0x66: return EADGTraceAssemble2DBdr<6,6>(nf,B,pa_data,ea_data_bdr);
case 0x77: return EADGTraceAssemble2DBdr<7,7>(nf,B,pa_data,ea_data_bdr);
case 0x88: return EADGTraceAssemble2DBdr<8,8>(nf,B,pa_data,ea_data_bdr);
case 0x99: return EADGTraceAssemble2DBdr<9,9>(nf,B,pa_data,ea_data_bdr);
default:
return EADGTraceAssemble2DBdr(nf,B,pa_data,ea_data_bdr,add,dofs1D,quad1D);
return EADGTraceAssemble2DBdr(nf,B,pa_data,ea_data_bdr,dofs1D,quad1D);
}
}
else if (dim == 3)
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x23: return EADGTraceAssemble3DBdr<2,3>(nf,B,pa_data,ea_data_bdr,add);
case 0x34: return EADGTraceAssemble3DBdr<3,4>(nf,B,pa_data,ea_data_bdr,add);
case 0x45: return EADGTraceAssemble3DBdr<4,5>(nf,B,pa_data,ea_data_bdr,add);
case 0x56: return EADGTraceAssemble3DBdr<5,6>(nf,B,pa_data,ea_data_bdr,add);
case 0x67: return EADGTraceAssemble3DBdr<6,7>(nf,B,pa_data,ea_data_bdr,add);
case 0x78: return EADGTraceAssemble3DBdr<7,8>(nf,B,pa_data,ea_data_bdr,add);
case 0x89: return EADGTraceAssemble3DBdr<8,9>(nf,B,pa_data,ea_data_bdr,add);
case 0x23: return EADGTraceAssemble3DBdr<2,3>(nf,B,pa_data,ea_data_bdr);
case 0x34: return EADGTraceAssemble3DBdr<3,4>(nf,B,pa_data,ea_data_bdr);
case 0x45: return EADGTraceAssemble3DBdr<4,5>(nf,B,pa_data,ea_data_bdr);
case 0x56: return EADGTraceAssemble3DBdr<5,6>(nf,B,pa_data,ea_data_bdr);
case 0x67: return EADGTraceAssemble3DBdr<6,7>(nf,B,pa_data,ea_data_bdr);
case 0x78: return EADGTraceAssemble3DBdr<7,8>(nf,B,pa_data,ea_data_bdr);
case 0x89: return EADGTraceAssemble3DBdr<8,9>(nf,B,pa_data,ea_data_bdr);
default:
return EADGTraceAssemble3DBdr(nf,B,pa_data,ea_data_bdr,add,dofs1D,quad1D);
return EADGTraceAssemble3DBdr(nf,B,pa_data,ea_data_bdr,dofs1D,quad1D);
}
}
MFEM_ABORT("Unknown kernel.");
+31 -59
View File
@@ -22,7 +22,6 @@ static void EADiffusionAssemble1D(const int NE,
const Array<double> &g,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -54,14 +53,7 @@ static void EADiffusionAssemble1D(const int NE,
{
val += r_Gj[k1] * D(k1, e) * r_Gi[k1];
}
if (add)
{
A(i1, j1, e) += val;
}
else
{
A(i1, j1, e) = val;
}
A(i1, j1, e) += val;
}
}
});
@@ -73,7 +65,6 @@ static void EADiffusionAssemble2D(const int NE,
const Array<double> &g,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -129,14 +120,7 @@ static void EADiffusionAssemble2D(const int NE,
+ gbi * D11 * gbj;
}
}
if (add)
{
A(i1, i2, j1, j2, e) += val;
}
else
{
A(i1, i2, j1, j2, e) = val;
}
A(i1, i2, j1, j2, e) += val;
}
}
}
@@ -146,11 +130,10 @@ static void EADiffusionAssemble2D(const int NE,
template<int T_D1D = 0, int T_Q1D = 0>
static void EADiffusionAssemble3D(const int NE,
const Array<double> &b,
const Array<double> &g,
const Array<double> &b,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -225,14 +208,7 @@ static void EADiffusionAssemble3D(const int NE,
}
}
}
if (add)
{
A(i1, i2, i3, j1, j2, j3, e) += val;
}
else
{
A(i1, i2, i3, j1, j2, j3, e) = val;
}
A(i1, i2, i3, j1, j2, j3, e) += val;
}
}
}
@@ -243,8 +219,7 @@ static void EADiffusionAssemble3D(const int NE,
}
void DiffusionIntegrator::AssembleEA(const FiniteElementSpace &fes,
Vector &ea_data,
const bool add)
Vector &ea_data)
{
AssemblePA(fes);
const int ne = fes.GetMesh()->GetNE();
@@ -254,47 +229,44 @@ void DiffusionIntegrator::AssembleEA(const FiniteElementSpace &fes,
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x22: return EADiffusionAssemble1D<2,2>(ne,B,G,pa_data,ea_data,add);
case 0x33: return EADiffusionAssemble1D<3,3>(ne,B,G,pa_data,ea_data,add);
case 0x44: return EADiffusionAssemble1D<4,4>(ne,B,G,pa_data,ea_data,add);
case 0x55: return EADiffusionAssemble1D<5,5>(ne,B,G,pa_data,ea_data,add);
case 0x66: return EADiffusionAssemble1D<6,6>(ne,B,G,pa_data,ea_data,add);
case 0x77: return EADiffusionAssemble1D<7,7>(ne,B,G,pa_data,ea_data,add);
case 0x88: return EADiffusionAssemble1D<8,8>(ne,B,G,pa_data,ea_data,add);
case 0x99: return EADiffusionAssemble1D<9,9>(ne,B,G,pa_data,ea_data,add);
default: return EADiffusionAssemble1D(ne,B,G,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x22: return EADiffusionAssemble1D<2,2>(ne,B,G,pa_data,ea_data);
case 0x33: return EADiffusionAssemble1D<3,3>(ne,B,G,pa_data,ea_data);
case 0x44: return EADiffusionAssemble1D<4,4>(ne,B,G,pa_data,ea_data);
case 0x55: return EADiffusionAssemble1D<5,5>(ne,B,G,pa_data,ea_data);
case 0x66: return EADiffusionAssemble1D<6,6>(ne,B,G,pa_data,ea_data);
case 0x77: return EADiffusionAssemble1D<7,7>(ne,B,G,pa_data,ea_data);
case 0x88: return EADiffusionAssemble1D<8,8>(ne,B,G,pa_data,ea_data);
case 0x99: return EADiffusionAssemble1D<9,9>(ne,B,G,pa_data,ea_data);
default: return EADiffusionAssemble1D(ne,B,G,pa_data,ea_data,dofs1D,quad1D);
}
}
else if (dim == 2)
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x22: return EADiffusionAssemble2D<2,2>(ne,B,G,pa_data,ea_data,add);
case 0x33: return EADiffusionAssemble2D<3,3>(ne,B,G,pa_data,ea_data,add);
case 0x44: return EADiffusionAssemble2D<4,4>(ne,B,G,pa_data,ea_data,add);
case 0x55: return EADiffusionAssemble2D<5,5>(ne,B,G,pa_data,ea_data,add);
case 0x66: return EADiffusionAssemble2D<6,6>(ne,B,G,pa_data,ea_data,add);
case 0x77: return EADiffusionAssemble2D<7,7>(ne,B,G,pa_data,ea_data,add);
case 0x88: return EADiffusionAssemble2D<8,8>(ne,B,G,pa_data,ea_data,add);
case 0x99: return EADiffusionAssemble2D<9,9>(ne,B,G,pa_data,ea_data,add);
default: return EADiffusionAssemble2D(ne,B,G,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x22: return EADiffusionAssemble2D<2,2>(ne,B,G,pa_data,ea_data);
case 0x33: return EADiffusionAssemble2D<3,3>(ne,B,G,pa_data,ea_data);
case 0x44: return EADiffusionAssemble2D<4,4>(ne,B,G,pa_data,ea_data);
case 0x55: return EADiffusionAssemble2D<5,5>(ne,B,G,pa_data,ea_data);
case 0x66: return EADiffusionAssemble2D<6,6>(ne,B,G,pa_data,ea_data);
case 0x77: return EADiffusionAssemble2D<7,7>(ne,B,G,pa_data,ea_data);
case 0x88: return EADiffusionAssemble2D<8,8>(ne,B,G,pa_data,ea_data);
case 0x99: return EADiffusionAssemble2D<9,9>(ne,B,G,pa_data,ea_data);
default: return EADiffusionAssemble2D(ne,B,G,pa_data,ea_data,dofs1D,quad1D);
}
}
else if (dim == 3)
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x23: return EADiffusionAssemble3D<2,3>(ne,B,G,pa_data,ea_data,add);
case 0x34: return EADiffusionAssemble3D<3,4>(ne,B,G,pa_data,ea_data,add);
case 0x45: return EADiffusionAssemble3D<4,5>(ne,B,G,pa_data,ea_data,add);
case 0x56: return EADiffusionAssemble3D<5,6>(ne,B,G,pa_data,ea_data,add);
case 0x67: return EADiffusionAssemble3D<6,7>(ne,B,G,pa_data,ea_data,add);
case 0x78: return EADiffusionAssemble3D<7,8>(ne,B,G,pa_data,ea_data,add);
case 0x89: return EADiffusionAssemble3D<8,9>(ne,B,G,pa_data,ea_data,add);
default: return EADiffusionAssemble3D(ne,B,G,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x23: return EADiffusionAssemble3D<2,3>(ne,B,G,pa_data,ea_data);
case 0x34: return EADiffusionAssemble3D<3,4>(ne,B,G,pa_data,ea_data);
case 0x45: return EADiffusionAssemble3D<4,5>(ne,B,G,pa_data,ea_data);
case 0x56: return EADiffusionAssemble3D<5,6>(ne,B,G,pa_data,ea_data);
case 0x67: return EADiffusionAssemble3D<6,7>(ne,B,G,pa_data,ea_data);
case 0x78: return EADiffusionAssemble3D<7,8>(ne,B,G,pa_data,ea_data);
case 0x89: return EADiffusionAssemble3D<8,9>(ne,B,G,pa_data,ea_data);
default: return EADiffusionAssemble3D(ne,B,G,pa_data,ea_data,dofs1D,quad1D);
}
}
MFEM_ABORT("Unknown kernel.");
+111 -104
View File
@@ -96,28 +96,26 @@ void PADiffusionSetup2D<2>(const int Q1D,
const Vector &c,
Vector &d)
{
const int NQ = Q1D*Q1D;
const bool const_c = c.Size() == 1;
const auto W = Reshape(w.Read(), Q1D,Q1D);
const auto J = Reshape(j.Read(), Q1D,Q1D,2,2,NE);
const auto C = const_c ? Reshape(c.Read(), 1,1,1) :
Reshape(c.Read(), Q1D,Q1D,NE);
auto D = Reshape(d.Write(), Q1D,Q1D, 3, NE);
MFEM_FORALL_2D(e, NE, Q1D,Q1D,1,
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto C = const_c ? Reshape(c.Read(), 1, 1) : Reshape(c.Read(), NQ, NE);
auto D = Reshape(d.Write(), NQ, 3, NE);
MFEM_FORALL(e, NE,
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
const double J11 = J(qx,qy,0,0,e);
const double J21 = J(qx,qy,1,0,e);
const double J12 = J(qx,qy,0,1,e);
const double J22 = J(qx,qy,1,1,e);
const double coeff = const_c ? C(0,0,0) : C(qx,qy,e);
const double c_detJ = W(qx,qy) * coeff / ((J11*J22)-(J21*J12));
D(qx,qy,0,e) = c_detJ * (J12*J12 + J22*J22); // 1,1
D(qx,qy,1,e) = -c_detJ * (J12*J11 + J22*J21); // 1,2
D(qx,qy,2,e) = c_detJ * (J11*J11 + J21*J21); // 2,2
}
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double coeff = const_c ? C(0,0) : C(q,e);
const double c_detJ = W[q] * coeff / ((J11*J22)-(J21*J12));
D(q,0,e) = c_detJ * (J12*J12 + J22*J22); // 1,1
D(q,1,e) = -c_detJ * (J12*J11 + J22*J21); // 1,2
D(q,2,e) = c_detJ * (J11*J11 + J21*J21); // 2,2
}
});
}
@@ -133,35 +131,33 @@ void PADiffusionSetup2D<3>(const int Q1D,
{
constexpr int DIM = 2;
constexpr int SDIM = 3;
const int NQ = Q1D*Q1D;
const bool const_c = c.Size() == 1;
const auto W = Reshape(w.Read(), Q1D,Q1D);
const auto J = Reshape(j.Read(), Q1D,Q1D,SDIM,DIM,NE);
const auto C = const_c ? Reshape(c.Read(), 1,1,1) :
Reshape(c.Read(), Q1D,Q1D,NE);
auto D = Reshape(d.Write(), Q1D,Q1D, 3, NE);
MFEM_FORALL_2D(e, NE, Q1D,Q1D,1,
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, SDIM, DIM, NE);
auto C = const_c ? Reshape(c.Read(), 1, 1) : Reshape(c.Read(), NQ, NE);
auto D = Reshape(d.Write(), NQ, 3, NE);
MFEM_FORALL(e, NE,
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
const double wq = W(qx,qy);
const double J11 = J(qx,qy,0,0,e);
const double J21 = J(qx,qy,1,0,e);
const double J31 = J(qx,qy,2,0,e);
const double J12 = J(qx,qy,0,1,e);
const double J22 = J(qx,qy,1,1,e);
const double J32 = J(qx,qy,2,1,e);
const double E = J11*J11 + J21*J21 + J31*J31;
const double G = J12*J12 + J22*J22 + J32*J32;
const double F = J11*J12 + J21*J22 + J31*J32;
const double iw = 1.0 / sqrt(E*G - F*F);
const double coeff = const_c ? C(0,0,0) : C(qx,qy,e);
const double alpha = wq * coeff * iw;
D(qx,qy,0,e) = alpha * G; // 1,1
D(qx,qy,1,e) = -alpha * F; // 1,2
D(qx,qy,2,e) = alpha * E; // 2,2
}
const double wq = W[q];
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double E = J11*J11 + J21*J21 + J31*J31;
const double G = J12*J12 + J22*J22 + J32*J32;
const double F = J11*J12 + J21*J22 + J31*J32;
const double iw = 1.0 / sqrt(E*G - F*F);
const double coeff = const_c ? C(0,0) : C(q,e);
const double alpha = wq * coeff * iw;
D(q,0,e) = alpha * G; // 1,1
D(q,1,e) = -alpha * F; // 1,2
D(q,2,e) = alpha * E; // 2,2
}
});
}
@@ -174,53 +170,47 @@ static void PADiffusionSetup3D(const int Q1D,
const Vector &c,
Vector &d)
{
const int NQ = Q1D*Q1D*Q1D;
const bool const_c = c.Size() == 1;
const auto W = Reshape(w.Read(), Q1D,Q1D,Q1D);
const auto J = Reshape(j.Read(), Q1D,Q1D,Q1D,3,3,NE);
const auto C = const_c ? Reshape(c.Read(), 1,1,1,1) :
Reshape(c.Read(), Q1D,Q1D,Q1D,NE);
auto D = Reshape(d.Write(), Q1D,Q1D,Q1D, 6, NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto C = const_c ? Reshape(c.Read(), 1, 1) : Reshape(c.Read(), NQ, NE);
auto D = Reshape(d.Write(), NQ, 6, NE);
MFEM_FORALL(e, NE,
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
const double J11 = J(qx,qy,qz,0,0,e);
const double J21 = J(qx,qy,qz,1,0,e);
const double J31 = J(qx,qy,qz,2,0,e);
const double J12 = J(qx,qy,qz,0,1,e);
const double J22 = J(qx,qy,qz,1,1,e);
const double J32 = J(qx,qy,qz,2,1,e);
const double J13 = J(qx,qy,qz,0,2,e);
const double J23 = J(qx,qy,qz,1,2,e);
const double J33 = J(qx,qy,qz,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
const double c_detJ = W(qx,qy,qz) * coeff / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
D(qx,qy,qz,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
D(qx,qy,qz,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
D(qx,qy,qz,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
D(qx,qy,qz,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
D(qx,qy,qz,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
D(qx,qy,qz,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
}
}
const double J11 = J(q,0,0,e);
const double J21 = J(q,1,0,e);
const double J31 = J(q,2,0,e);
const double J12 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double J32 = J(q,2,1,e);
const double J13 = J(q,0,2,e);
const double J23 = J(q,1,2,e);
const double J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double coeff = const_c ? C(0,0) : C(q,e);
const double c_detJ = W[q] * coeff / detJ;
// adj(J)
const double A11 = (J22 * J33) - (J23 * J32);
const double A12 = (J32 * J13) - (J12 * J33);
const double A13 = (J12 * J23) - (J22 * J13);
const double A21 = (J31 * J23) - (J21 * J33);
const double A22 = (J11 * J33) - (J13 * J31);
const double A23 = (J21 * J13) - (J11 * J23);
const double A31 = (J21 * J32) - (J31 * J22);
const double A32 = (J31 * J12) - (J11 * J32);
const double A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
D(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
D(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
D(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
D(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
D(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
D(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
}
});
}
@@ -263,7 +253,8 @@ static void PADiffusionSetup(const int dim,
}
}
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes,
const bool force)
{
// Assuming the same element type
fespace = &fes;
@@ -272,7 +263,7 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
const FiniteElement &el = *fes.GetFE(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
if (DeviceCanUseCeed() && !force)
{
if (ceedDataPtr) { delete ceedDataPtr; }
CeedData* ptr = new CeedData();
@@ -280,6 +271,8 @@ void DiffusionIntegrator::SetupPA(const FiniteElementSpace &fes)
InitCeedCoeff(Q, ptr);
return CeedPADiffusionAssemble(fes, *ir, *ptr);
}
#else
MFEM_CONTRACT_VAR(force);
#endif
const int dims = el.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
@@ -756,17 +749,9 @@ static void PADiffusionAssembleDiagonal(const int dim,
void DiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
{
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
CeedAssembleDiagonalPA(ceedDataPtr, diag);
}
else
#endif
{
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->G, pa_data, diag);
}
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
PADiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->G, pa_data, diag);
}
@@ -1680,7 +1665,7 @@ static void PADiffusionApply(const int dim,
MFEM_ABORT("OCCA PADiffusionApply unknown kernel!");
}
#endif // MFEM_USE_OCCA
const int ID = (D1D << 4) | Q1D;
const int ID = (D1D << 4 ) | Q1D;
if (dim == 2)
{
@@ -1723,7 +1708,29 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
CeedAddMultPA(ceedDataPtr, x, y);
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorSyncArray(ceedDataPtr->v, mem);
}
else
#endif
+298 -1830
View File
File diff suppressed because it is too large Load Diff
+30 -58
View File
@@ -21,7 +21,6 @@ static void EAMassAssemble1D(const int NE,
const Array<double> &basis,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -53,14 +52,7 @@ static void EAMassAssemble1D(const int NE,
{
val += r_Bi[k1] * r_Bj[k1] * D(k1, e);
}
if (add)
{
M(i1, j1, e) += val;
}
else
{
M(i1, j1, e) = val;
}
M(i1, j1, e) += val;
}
}
});
@@ -71,7 +63,6 @@ static void EAMassAssemble2D(const int NE,
const Array<double> &basis,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -123,14 +114,7 @@ static void EAMassAssemble2D(const int NE,
* s_D[k1][k2];
}
}
if (add)
{
M(i1, i2, j1, j2, e) += val;
}
else
{
M(i1, i2, j1, j2, e) = val;
}
M(i1, i2, j1, j2, e) += val;
}
}
}
@@ -143,7 +127,6 @@ static void EAMassAssemble3D(const int NE,
const Array<double> &basis,
const Vector &padata,
Vector &eadata,
const bool add,
const int d1d = 0,
const int q1d = 0)
{
@@ -206,14 +189,7 @@ static void EAMassAssemble3D(const int NE,
}
}
}
if (add)
{
M(i1, i2, i3, j1, j2, j3, e) += val;
}
else
{
M(i1, i2, i3, j1, j2, j3, e) = val;
}
M(i1, i2, i3, j1, j2, j3, e) += val;
}
}
}
@@ -224,8 +200,7 @@ static void EAMassAssemble3D(const int NE,
}
void MassIntegrator::AssembleEA(const FiniteElementSpace &fes,
Vector &ea_data,
const bool add)
Vector &ea_data)
{
AssemblePA(fes);
const int ne = fes.GetMesh()->GetNE();
@@ -234,47 +209,44 @@ void MassIntegrator::AssembleEA(const FiniteElementSpace &fes,
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x22: return EAMassAssemble1D<2,2>(ne,B,pa_data,ea_data,add);
case 0x33: return EAMassAssemble1D<3,3>(ne,B,pa_data,ea_data,add);
case 0x44: return EAMassAssemble1D<4,4>(ne,B,pa_data,ea_data,add);
case 0x55: return EAMassAssemble1D<5,5>(ne,B,pa_data,ea_data,add);
case 0x66: return EAMassAssemble1D<6,6>(ne,B,pa_data,ea_data,add);
case 0x77: return EAMassAssemble1D<7,7>(ne,B,pa_data,ea_data,add);
case 0x88: return EAMassAssemble1D<8,8>(ne,B,pa_data,ea_data,add);
case 0x99: return EAMassAssemble1D<9,9>(ne,B,pa_data,ea_data,add);
default: return EAMassAssemble1D(ne,B,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x22: return EAMassAssemble1D<2,2>(ne,B,pa_data,ea_data);
case 0x33: return EAMassAssemble1D<3,3>(ne,B,pa_data,ea_data);
case 0x44: return EAMassAssemble1D<4,4>(ne,B,pa_data,ea_data);
case 0x55: return EAMassAssemble1D<5,5>(ne,B,pa_data,ea_data);
case 0x66: return EAMassAssemble1D<6,6>(ne,B,pa_data,ea_data);
case 0x77: return EAMassAssemble1D<7,7>(ne,B,pa_data,ea_data);
case 0x88: return EAMassAssemble1D<8,8>(ne,B,pa_data,ea_data);
case 0x99: return EAMassAssemble1D<9,9>(ne,B,pa_data,ea_data);
default: return EAMassAssemble1D(ne,B,pa_data,ea_data,dofs1D,quad1D);
}
}
else if (dim == 2)
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x22: return EAMassAssemble2D<2,2>(ne,B,pa_data,ea_data,add);
case 0x33: return EAMassAssemble2D<3,3>(ne,B,pa_data,ea_data,add);
case 0x44: return EAMassAssemble2D<4,4>(ne,B,pa_data,ea_data,add);
case 0x55: return EAMassAssemble2D<5,5>(ne,B,pa_data,ea_data,add);
case 0x66: return EAMassAssemble2D<6,6>(ne,B,pa_data,ea_data,add);
case 0x77: return EAMassAssemble2D<7,7>(ne,B,pa_data,ea_data,add);
case 0x88: return EAMassAssemble2D<8,8>(ne,B,pa_data,ea_data,add);
case 0x99: return EAMassAssemble2D<9,9>(ne,B,pa_data,ea_data,add);
default: return EAMassAssemble2D(ne,B,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x22: return EAMassAssemble2D<2,2>(ne,B,pa_data,ea_data);
case 0x33: return EAMassAssemble2D<3,3>(ne,B,pa_data,ea_data);
case 0x44: return EAMassAssemble2D<4,4>(ne,B,pa_data,ea_data);
case 0x55: return EAMassAssemble2D<5,5>(ne,B,pa_data,ea_data);
case 0x66: return EAMassAssemble2D<6,6>(ne,B,pa_data,ea_data);
case 0x77: return EAMassAssemble2D<7,7>(ne,B,pa_data,ea_data);
case 0x88: return EAMassAssemble2D<8,8>(ne,B,pa_data,ea_data);
case 0x99: return EAMassAssemble2D<9,9>(ne,B,pa_data,ea_data);
default: return EAMassAssemble2D(ne,B,pa_data,ea_data,dofs1D,quad1D);
}
}
else if (dim == 3)
{
switch ((dofs1D << 4 ) | quad1D)
{
case 0x23: return EAMassAssemble3D<2,3>(ne,B,pa_data,ea_data,add);
case 0x34: return EAMassAssemble3D<3,4>(ne,B,pa_data,ea_data,add);
case 0x45: return EAMassAssemble3D<4,5>(ne,B,pa_data,ea_data,add);
case 0x56: return EAMassAssemble3D<5,6>(ne,B,pa_data,ea_data,add);
case 0x67: return EAMassAssemble3D<6,7>(ne,B,pa_data,ea_data,add);
case 0x78: return EAMassAssemble3D<7,8>(ne,B,pa_data,ea_data,add);
case 0x89: return EAMassAssemble3D<8,9>(ne,B,pa_data,ea_data,add);
default: return EAMassAssemble3D(ne,B,pa_data,ea_data,add,
dofs1D,quad1D);
case 0x23: return EAMassAssemble3D<2,3>(ne,B,pa_data,ea_data);
case 0x34: return EAMassAssemble3D<3,4>(ne,B,pa_data,ea_data);
case 0x45: return EAMassAssemble3D<4,5>(ne,B,pa_data,ea_data);
case 0x56: return EAMassAssemble3D<5,6>(ne,B,pa_data,ea_data);
case 0x67: return EAMassAssemble3D<6,7>(ne,B,pa_data,ea_data);
case 0x78: return EAMassAssemble3D<7,8>(ne,B,pa_data,ea_data);
case 0x89: return EAMassAssemble3D<8,9>(ne,B,pa_data,ea_data);
default: return EAMassAssemble3D(ne,B,pa_data,ea_data,dofs1D,quad1D);
}
}
MFEM_ABORT("Unknown kernel.");
+60 -59
View File
@@ -23,7 +23,7 @@ namespace mfem
// PA Mass Assemble kernel
void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
void MassIntegrator::SetupPA(const FiniteElementSpace &fes, const bool force)
{
// Assuming the same element type
fespace = &fes;
@@ -33,7 +33,7 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
ElementTransformation *T = mesh->GetElementTransformation(0);
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, *T);
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
if (DeviceCanUseCeed() && !force)
{
if (ceedDataPtr) { delete ceedDataPtr; }
CeedData* ptr = new CeedData();
@@ -41,6 +41,8 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
InitCeedCoeff(Q, ptr);
return CeedPAMassAssemble(fes, *ir, *ptr);
}
#else
MFEM_CONTRACT_VAR(force);
#endif
dim = mesh->Dimension();
ne = fes.GetMesh()->GetNE();
@@ -92,64 +94,49 @@ void MassIntegrator::SetupPA(const FiniteElementSpace &fes)
if (dim==2)
{
const int NE = ne;
const int Q1D = quad1D;
const int NQ = nq;
const bool const_c = coeff.Size() == 1;
const auto W = Reshape(ir->GetWeights().Read(), Q1D,Q1D);
const auto J = Reshape(geom->J.Read(), Q1D,Q1D,2,2,NE);
const auto C = const_c ? Reshape(coeff.Read(), 1,1,1) :
Reshape(coeff.Read(), Q1D,Q1D,NE);
auto v = Reshape(pa_data.Write(), Q1D,Q1D, NE);
MFEM_FORALL_2D(e, NE, Q1D,Q1D,1,
auto w = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
auto C =
const_c ? Reshape(coeff.Read(), 1,1) : Reshape(coeff.Read(), NQ,NE);
auto v = Reshape(pa_data.Write(), NQ, NE);
MFEM_FORALL(e, NE,
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
const double J11 = J(qx,qy,0,0,e);
const double J12 = J(qx,qy,1,0,e);
const double J21 = J(qx,qy,0,1,e);
const double J22 = J(qx,qy,1,1,e);
const double detJ = (J11*J22)-(J21*J12);
const double coeff = const_c ? C(0,0,0) : C(qx,qy,e);
v(qx,qy,e) = W(qx,qy) * coeff * detJ;
}
const double J11 = J(q,0,0,e);
const double J12 = J(q,1,0,e);
const double J21 = J(q,0,1,e);
const double J22 = J(q,1,1,e);
const double detJ = (J11*J22)-(J21*J12);
const double coeff = const_c ? C(0,0) : C(q,e);
v(q,e) = w[q] * coeff * detJ;
}
});
}
if (dim==3)
{
const int NE = ne;
const int Q1D = quad1D;
const int NQ = nq;
const bool const_c = coeff.Size() == 1;
const auto W = Reshape(ir->GetWeights().Read(), Q1D,Q1D,Q1D);
const auto J = Reshape(geom->J.Read(), Q1D,Q1D,Q1D,3,3,NE);
const auto C = const_c ? Reshape(coeff.Read(), 1,1,1,1) :
Reshape(coeff.Read(), Q1D,Q1D,Q1D,NE);
auto v = Reshape(pa_data.Write(), Q1D,Q1D,Q1D,NE);
MFEM_FORALL_3D(e, NE, Q1D, Q1D, Q1D,
auto W = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
auto C =
const_c ? Reshape(coeff.Read(), 1,1) : Reshape(coeff.Read(), NQ,NE);
auto v = Reshape(pa_data.Write(), NQ,NE);
MFEM_FORALL(e, NE,
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
const double J11 = J(qx,qy,qz,0,0,e);
const double J21 = J(qx,qy,qz,1,0,e);
const double J31 = J(qx,qy,qz,2,0,e);
const double J12 = J(qx,qy,qz,0,1,e);
const double J22 = J(qx,qy,qz,1,1,e);
const double J32 = J(qx,qy,qz,2,1,e);
const double J13 = J(qx,qy,qz,0,2,e);
const double J23 = J(qx,qy,qz,1,2,e);
const double J33 = J(qx,qy,qz,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double coeff = const_c ? C(0,0,0,0) : C(qx,qy,qz,e);
v(qx,qy,qz,e) = W(qx,qy,qz) * coeff * detJ;
}
}
const double J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
const double J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
const double J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
const double detJ = J11 * (J22 * J33 - J32 * J23) -
/* */ J21 * (J12 * J33 - J32 * J13) +
/* */ J31 * (J12 * J23 - J22 * J13);
const double coeff = const_c ? C(0,0) : C(q,e);
v(q,e) = W[q] * coeff * detJ;
}
});
}
@@ -468,16 +455,8 @@ static void PAMassAssembleDiagonal(const int dim, const int D1D,
void MassIntegrator::AssembleDiagonalPA(Vector &diag)
{
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
CeedAssembleDiagonalPA(ceedDataPtr, diag);
}
else
#endif
{
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
}
if (pa_data.Size()==0) { SetupPA(*fespace, true); }
PAMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
}
@@ -1230,7 +1209,29 @@ void MassIntegrator::AddMultPA(const Vector &x, Vector &y) const
#ifdef MFEM_USE_CEED
if (DeviceCanUseCeed())
{
CeedAddMultPA(ceedDataPtr, x, y);
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorSyncArray(ceedDataPtr->v, mem);
}
else
#endif
+56 -139
View File
@@ -16,171 +16,88 @@ namespace mfem
{
void TransposeIntegrator::AssembleEA(const FiniteElementSpace &fes,
Vector &ea_data, const bool add)
Vector &ea_data)
{
if (add)
Vector ea_data_tmp(ea_data.Size());
ea_data_tmp = 0.0;
bfi->AssembleEA(fes, ea_data_tmp);
const int ne = fes.GetNE();
if (ne == 0) { return; }
const int dofs = fes.GetFE(0)->GetDof();
auto A = Reshape(ea_data_tmp.Write(), dofs, dofs, ne);
auto AT = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
MFEM_FORALL(e, ne,
{
Vector ea_data_tmp(ea_data.Size());
bfi->AssembleEA(fes, ea_data_tmp, false);
const int ne = fes.GetNE();
if (ne == 0) { return; }
const int dofs = fes.GetFE(0)->GetDof();
auto A = Reshape(ea_data_tmp.Read(), dofs, dofs, ne);
auto AT = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
MFEM_FORALL(e, ne,
for (int i = 0; i < dofs; i++)
{
for (int i = 0; i < dofs; i++)
for (int j = 0; j < dofs; j++)
{
for (int j = 0; j < dofs; j++)
{
const double a = A(i, j, e);
AT(j, i, e) += a;
}
const double a = A(i, j, e);
AT(j, i, e) += a;
}
});
}
else
{
bfi->AssembleEA(fes, ea_data, false);
const int ne = fes.GetNE();
if (ne == 0) { return; }
const int dofs = fes.GetFE(0)->GetDof();
auto A = Reshape(ea_data.ReadWrite(), dofs, dofs, ne);
MFEM_FORALL(e, ne,
{
for (int i = 0; i < dofs; i++)
{
for (int j = i+1; j < dofs; j++)
{
const double aij = A(i, j, e);
const double aji = A(j, i, e);
A(j, i, e) = aij;
A(i, j, e) = aji;
}
}
});
}
}
});
}
void TransposeIntegrator::AssembleEAInteriorFaces(const FiniteElementSpace& fes,
Vector &ea_data_int,
Vector &ea_data_ext,
const bool add)
Vector &ea_data_ext)
{
const int nf = fes.GetNFbyType(FaceType::Interior);
if (nf == 0) { return; }
if (add)
Vector ea_data_int_tmp(ea_data_int.Size());
Vector ea_data_ext_tmp(ea_data_ext.Size());
ea_data_int_tmp = 0.0;
ea_data_ext_tmp = 0.0;
bfi->AssembleEAInteriorFaces(fes, ea_data_int_tmp, ea_data_ext_tmp);
const int faceDofs = fes.GetTraceElement(0,
fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof();
auto A_int = Reshape(ea_data_int_tmp.Read(), faceDofs, faceDofs, 2, nf);
auto A_ext = Reshape(ea_data_ext_tmp.Read(), faceDofs, faceDofs, 2, nf);
auto AT_int = Reshape(ea_data_int.ReadWrite(), faceDofs, faceDofs, 2, nf);
auto AT_ext = Reshape(ea_data_ext.ReadWrite(), faceDofs, faceDofs, 2, nf);
MFEM_FORALL(f, nf,
{
Vector ea_data_int_tmp(ea_data_int.Size());
Vector ea_data_ext_tmp(ea_data_ext.Size());
bfi->AssembleEAInteriorFaces(fes, ea_data_int_tmp, ea_data_ext_tmp, false);
const int faceDofs = fes.GetTraceElement(0,
fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof();
auto A_int = Reshape(ea_data_int_tmp.Read(), faceDofs, faceDofs, 2, nf);
auto A_ext = Reshape(ea_data_ext_tmp.Read(), faceDofs, faceDofs, 2, nf);
auto AT_int = Reshape(ea_data_int.ReadWrite(), faceDofs, faceDofs, 2, nf);
auto AT_ext = Reshape(ea_data_ext.ReadWrite(), faceDofs, faceDofs, 2, nf);
MFEM_FORALL(f, nf,
for (int i = 0; i < faceDofs; i++)
{
for (int i = 0; i < faceDofs; i++)
for (int j = 0; j < faceDofs; j++)
{
for (int j = 0; j < faceDofs; j++)
{
const double a_int0 = A_int(i, j, 0, f);
const double a_int1 = A_int(i, j, 1, f);
const double a_ext0 = A_ext(i, j, 0, f);
const double a_ext1 = A_ext(i, j, 1, f);
AT_int(j, i, 0, f) += a_int0;
AT_int(j, i, 1, f) += a_int1;
AT_ext(j, i, 0, f) += a_ext1;
AT_ext(j, i, 1, f) += a_ext0;
}
const double a_int0 = A_int(i, j, 0, f);
const double a_int1 = A_int(i, j, 1, f);
const double a_ext0 = A_ext(i, j, 0, f);
const double a_ext1 = A_ext(i, j, 1, f);
AT_int(j, i, 0, f) += a_int0;
AT_int(j, i, 1, f) += a_int1;
AT_ext(j, i, 0, f) += a_ext1;
AT_ext(j, i, 1, f) += a_ext0;
}
});
}
else
{
bfi->AssembleEAInteriorFaces(fes, ea_data_int, ea_data_ext, false);
const int faceDofs = fes.GetTraceElement(0,
fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof();
auto A_int = Reshape(ea_data_int.ReadWrite(), faceDofs, faceDofs, 2, nf);
auto A_ext = Reshape(ea_data_ext.ReadWrite(), faceDofs, faceDofs, 2, nf);
MFEM_FORALL(f, nf,
{
for (int i = 0; i < faceDofs; i++)
{
for (int j = i+1; j < faceDofs; j++)
{
const double aij_int0 = A_int(i, j, 0, f);
const double aij_int1 = A_int(i, j, 1, f);
const double aji_int0 = A_int(j, i, 0, f);
const double aji_int1 = A_int(j, i, 1, f);
A_int(j, i, 0, f) = aij_int0;
A_int(j, i, 1, f) = aij_int1;
A_int(i, j, 0, f) = aji_int0;
A_int(i, j, 1, f) = aji_int1;
}
}
for (int i = 0; i < faceDofs; i++)
{
for (int j = 0; j < faceDofs; j++)
{
const double aij_ext0 = A_ext(i, j, 0, f);
const double aji_ext1 = A_ext(j, i, 1, f);
A_ext(j, i, 1, f) = aij_ext0;
A_ext(i, j, 0, f) = aji_ext1;
}
}
});
}
}
});
}
void TransposeIntegrator::AssembleEABoundaryFaces(const FiniteElementSpace& fes,
Vector &ea_data_bdr,
const bool add)
Vector &ea_data_bdr)
{
const int nf = fes.GetNFbyType(FaceType::Boundary);
if (nf == 0) { return; }
if (add)
Vector ea_data_bdr_tmp(ea_data_bdr.Size());
ea_data_bdr_tmp = 0.0;
bfi->AssembleEABoundaryFaces(fes, ea_data_bdr_tmp);
const int faceDofs = fes.GetTraceElement(0,
fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof();
auto A_bdr = Reshape(ea_data_bdr_tmp.Read(), faceDofs, faceDofs, nf);
auto AT_bdr = Reshape(ea_data_bdr.ReadWrite(), faceDofs, faceDofs, nf);
MFEM_FORALL(f, nf,
{
Vector ea_data_bdr_tmp(ea_data_bdr.Size());
bfi->AssembleEABoundaryFaces(fes, ea_data_bdr_tmp, false);
const int faceDofs = fes.GetTraceElement(0,
fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof();
auto A_bdr = Reshape(ea_data_bdr_tmp.Read(), faceDofs, faceDofs, nf);
auto AT_bdr = Reshape(ea_data_bdr.ReadWrite(), faceDofs, faceDofs, nf);
MFEM_FORALL(f, nf,
for (int i = 0; i < faceDofs; i++)
{
for (int i = 0; i < faceDofs; i++)
for (int j = 0; j < faceDofs; j++)
{
for (int j = 0; j < faceDofs; j++)
{
const double a_bdr = A_bdr(i, j, f);
AT_bdr(j, i, f) += a_bdr;
}
const double a_bdr = A_bdr(i, j, f);
AT_bdr(j, i, f) += a_bdr;
}
});
}
else
{
bfi->AssembleEABoundaryFaces(fes, ea_data_bdr, false);
const int faceDofs = fes.GetTraceElement(0,
fes.GetMesh()->GetFaceBaseGeometry(0))->GetDof();
auto A_bdr = Reshape(ea_data_bdr.ReadWrite(), faceDofs, faceDofs, nf);
MFEM_FORALL(f, nf,
{
for (int i = 0; i < faceDofs; i++)
{
for (int j = i+1; j < faceDofs; j++)
{
const double aij_bdr = A_bdr(i, j, f);
const double aji_bdr = A_bdr(j, i, f);
A_bdr(j, i, f) = aij_bdr;
A_bdr(i, j, f) = aji_bdr;
}
}
});
}
}
});
}
}
+76 -139
View File
@@ -20,7 +20,7 @@ void PAHcurlSetup2D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
Vector &coeff,
Vector &_coeff,
Vector &op);
void PAHcurlSetup3D(const int Q1D,
@@ -28,73 +28,50 @@ void PAHcurlSetup3D(const int Q1D,
const int NE,
const Array<double> &w,
const Vector &j,
Vector &coeff,
Vector &_coeff,
Vector &op);
void PAHcurlMassAssembleDiagonal2D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &bo,
const Array<double> &bc,
const Vector &pa_data,
Vector &diag);
const Array<double> &_Bo,
const Array<double> &_Bc,
const Vector &_op,
Vector &_diag);
void PAHcurlMassAssembleDiagonal3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &bo,
const Array<double> &bc,
const Vector &pa_data,
Vector &diag);
template<int T_D1D = 0, int T_Q1D = 0>
void SmemPAHcurlMassAssembleDiagonal3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &bo,
const Array<double> &bc,
const Vector &pa_data,
Vector &diag);
const Array<double> &_Bo,
const Array<double> &_Bc,
const Vector &_op,
Vector &_diag);
void PAHcurlMassApply2D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &bo,
const Array<double> &bc,
const Array<double> &bot,
const Array<double> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y);
const Array<double> &_Bo,
const Array<double> &_Bc,
const Array<double> &_Bot,
const Array<double> &_Bct,
const Vector &_op,
const Vector &_x,
Vector &_y);
void PAHcurlMassApply3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &bo,
const Array<double> &bc,
const Array<double> &bot,
const Array<double> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y);
template<int T_D1D = 0, int T_Q1D = 0>
void SmemPAHcurlMassApply3D(const int D1D,
const int Q1D,
const int NE,
const bool symmetric,
const Array<double> &bo,
const Array<double> &bc,
const Array<double> &bot,
const Array<double> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y);
const Array<double> &_Bo,
const Array<double> &_Bc,
const Array<double> &_Bot,
const Array<double> &_Bct,
const Vector &_op,
const Vector &_x,
Vector &_y);
void PAHdivSetup2D(const int Q1D,
const int NE,
@@ -113,24 +90,24 @@ void PAHdivSetup3D(const int Q1D,
void PAHcurlH1Apply2D(const int D1D,
const int Q1D,
const int NE,
const Array<double> &bc,
const Array<double> &gc,
const Array<double> &bot,
const Array<double> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y);
const Array<double> &_Bc,
const Array<double> &_Gc,
const Array<double> &_Bot,
const Array<double> &_Bct,
const Vector &_op,
const Vector &_x,
Vector &_y);
void PAHcurlH1Apply3D(const int D1D,
const int Q1D,
const int NE,
const Array<double> &bc,
const Array<double> &gc,
const Array<double> &bot,
const Array<double> &bct,
const Vector &pa_data,
const Vector &x,
Vector &y);
const Array<double> &_Bc,
const Array<double> &_Gc,
const Array<double> &_Bot,
const Array<double> &_Bct,
const Vector &_op,
const Vector &_x,
Vector &_y);
void PAHdivMassAssembleDiagonal2D(const int D1D,
const int Q1D,
@@ -769,12 +746,10 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &trial_fes,
symmetric = MQ ? MQ->IsSymmetric() : true;
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
const bool test_curl = (test_fetype == mfem::FiniteElement::CURL);
const bool test_div = (test_fetype == mfem::FiniteElement::DIV);
if ((trial_curl && test_div) || (trial_div && test_curl))
if ((trial_fetype == mfem::FiniteElement::CURL &&
test_fetype == mfem::FiniteElement::DIV) ||
(trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL))
pa_data.SetSize((coeffDim == 1 ? 1 : dim*dim) * nq * ne,
Device::GetMemoryType());
else
@@ -854,27 +829,34 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &trial_fes,
}
}
if (trial_curl && test_curl && dim == 3)
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype
&& dim == 3)
{
PAHcurlSetup3D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (trial_curl && test_curl && dim == 2)
else if (trial_fetype == mfem::FiniteElement::CURL
&& test_fetype == trial_fetype && dim == 2)
{
PAHcurlSetup2D(quad1D, coeffDim, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (trial_div && test_div && dim == 3)
else if (trial_fetype == mfem::FiniteElement::DIV
&& test_fetype == trial_fetype && dim == 3)
{
PAHdivSetup3D(quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (trial_div && test_div && dim == 2)
else if (trial_fetype == mfem::FiniteElement::DIV
&& test_fetype == trial_fetype && dim == 2)
{
PAHdivSetup2D(quad1D, ne, ir->GetWeights(), geom->J,
coeff, pa_data);
}
else if (((trial_curl && test_div) || (trial_div && test_curl)) &&
else if (((trial_fetype == mfem::FiniteElement::CURL &&
test_fetype == mfem::FiniteElement::DIV) ||
(trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL)) &&
test_fel->GetOrder() == trial_fel->GetOrder())
{
if (coeffDim == 1)
@@ -883,7 +865,8 @@ void VectorFEMassIntegrator::AssemblePA(const FiniteElementSpace &trial_fes,
}
else
{
const bool tr = (trial_div && test_curl);
const bool tr = (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL);
if (dim == 3)
PAHcurlHdivSetup3D(quad1D, coeffDim, ne, tr, ir->GetWeights(),
geom->J, coeff, pa_data);
@@ -904,30 +887,8 @@ void VectorFEMassIntegrator::AssembleDiagonalPA(Vector& diag)
{
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
const int ID = (dofs1D << 4) | quad1D;
switch (ID)
{
case 0x23: return SmemPAHcurlMassAssembleDiagonal3D<2,3>(dofs1D, quad1D, ne,
symmetric,
mapsO->B, mapsC->B, pa_data, diag);
case 0x34: return SmemPAHcurlMassAssembleDiagonal3D<3,4>(dofs1D, quad1D, ne,
symmetric,
mapsO->B, mapsC->B, pa_data, diag);
case 0x45: return SmemPAHcurlMassAssembleDiagonal3D<4,5>(dofs1D, quad1D, ne,
symmetric,
mapsO->B, mapsC->B, pa_data, diag);
case 0x56: return SmemPAHcurlMassAssembleDiagonal3D<5,6>(dofs1D, quad1D, ne,
symmetric,
mapsO->B, mapsC->B, pa_data, diag);
default: return SmemPAHcurlMassAssembleDiagonal3D(dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, pa_data, diag);
}
}
else
PAHcurlMassAssembleDiagonal3D(dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, pa_data, diag);
PAHcurlMassAssembleDiagonal3D(dofs1D, quad1D, ne, symmetric,
mapsO->B, mapsC->B, pa_data, diag);
}
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == trial_fetype)
@@ -962,58 +923,29 @@ void VectorFEMassIntegrator::AssembleDiagonalPA(Vector& diag)
void VectorFEMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
const bool trial_curl = (trial_fetype == mfem::FiniteElement::CURL);
const bool trial_div = (trial_fetype == mfem::FiniteElement::DIV);
const bool test_curl = (test_fetype == mfem::FiniteElement::CURL);
const bool test_div = (test_fetype == mfem::FiniteElement::DIV);
if (dim == 3)
{
if (trial_curl && test_curl)
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype)
{
if (Device::Allows(Backend::DEVICE_MASK))
{
const int ID = (dofs1D << 4) | quad1D;
switch (ID)
{
case 0x23: return SmemPAHcurlMassApply3D<2,3>(dofs1D, quad1D, ne, symmetric,
mapsO->B,
mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
case 0x34: return SmemPAHcurlMassApply3D<3,4>(dofs1D, quad1D, ne, symmetric,
mapsO->B,
mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
case 0x45: return SmemPAHcurlMassApply3D<4,5>(dofs1D, quad1D, ne, symmetric,
mapsO->B,
mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
case 0x56: return SmemPAHcurlMassApply3D<5,6>(dofs1D, quad1D, ne, symmetric,
mapsO->B,
mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
default: return SmemPAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric, mapsO->B,
mapsC->B,
mapsO->Bt, mapsC->Bt, pa_data, x, y);
}
}
else
PAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
PAHcurlMassApply3D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
mapsO->Bt, mapsC->Bt, pa_data, x, y);
}
else if (trial_div && test_div)
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == trial_fetype)
{
PAHdivMassApply3D(dofs1D, quad1D, ne, mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
}
else if (trial_curl && test_div)
else if (trial_fetype == mfem::FiniteElement::CURL &&
test_fetype == mfem::FiniteElement::DIV)
{
const bool scalarCoeff = !(VQ || MQ);
PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
true, mapsO->B, mapsC->B, mapsOtest->Bt,
mapsCtest->Bt, pa_data, x, y);
}
else if (trial_div && test_curl)
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL)
{
const bool scalarCoeff = !(VQ || MQ);
PAHcurlHdivMassApply3D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
@@ -1027,21 +959,26 @@ void VectorFEMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
}
else
{
if (trial_curl && test_curl)
if (trial_fetype == mfem::FiniteElement::CURL && test_fetype == trial_fetype)
{
PAHcurlMassApply2D(dofs1D, quad1D, ne, symmetric, mapsO->B, mapsC->B,
mapsO->Bt, mapsC->Bt, pa_data, x, y);
}
else if (trial_div && test_div)
else if (trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == trial_fetype)
{
PAHdivMassApply2D(dofs1D, quad1D, ne, mapsO->B, mapsC->B, mapsO->Bt,
mapsC->Bt, pa_data, x, y);
}
else if ((trial_curl && test_div) || (trial_div && test_curl))
else if ((trial_fetype == mfem::FiniteElement::CURL &&
test_fetype == mfem::FiniteElement::DIV) ||
(trial_fetype == mfem::FiniteElement::DIV &&
test_fetype == mfem::FiniteElement::CURL))
{
const bool scalarCoeff = !(VQ || MQ);
const bool trialHcurl = (trial_fetype == mfem::FiniteElement::CURL);
PAHcurlHdivMassApply2D(dofs1D, dofs1Dtest, quad1D, ne, scalarCoeff,
trial_curl, mapsO->B, mapsC->B, mapsOtest->Bt,
trialHcurl, mapsO->B, mapsC->B, mapsOtest->Bt,
mapsCtest->Bt, pa_data, x, y);
}
else
+4 -37
View File
@@ -38,8 +38,8 @@ protected:
void Destroy() { delete gfr; delete gfi; }
public:
/** @brief Construct a ComplexGridFunction associated with the
FiniteElementSpace @a *f. */
/* @brief Construct a ComplexGridFunction associated with the
FiniteElementSpace @a *f. */
ComplexGridFunction(FiniteElementSpace *f);
void Update();
@@ -71,14 +71,6 @@ public:
const GridFunction & real() const { return *gfr; }
const GridFunction & imag() const { return *gfi; }
/// Update the memory location of the real and imaginary GridFunction @a gfr
/// and @a gfi to match the ComplexGridFunction.
void Sync() { gfr->SyncMemory(*this); gfi->SyncMemory(*this); }
/// Update the alias memory location of the real and imaginary GridFunction
/// @a gfr and @a gfi to match the ComplexGridFunction.
void SyncAlias() { gfr->SyncAliasMemory(*this); gfi->SyncAliasMemory(*this); }
/// Destroys the grid function.
virtual ~ComplexGridFunction() { Destroy(); }
@@ -165,14 +157,6 @@ public:
const LinearForm & real() const { return *lfr; }
const LinearForm & imag() const { return *lfi; }
/// Update the memory location of the real and imaginary LinearForm @a lfr
/// and @a lfi to match the ComplexLinearForm.
void Sync() { lfr->SyncMemory(*this); lfi->SyncMemory(*this); }
/// Update the alias memory location of the real and imaginary LinearForm @a
/// lfr and @a lfi to match the ComplexLinearForm.
void SyncAlias() { lfr->SyncAliasMemory(*this); lfi->SyncAliasMemory(*this); }
void Update();
void Update(FiniteElementSpace *f);
@@ -339,8 +323,8 @@ protected:
public:
/** @brief Construct a ParComplexGridFunction associated with the
ParFiniteElementSpace @a *pf. */
/* @brief Construct a ParComplexGridFunction associated with the
ParFiniteElementSpace @a *pf. */
ParComplexGridFunction(ParFiniteElementSpace *pf);
void Update();
@@ -381,15 +365,6 @@ public:
const ParGridFunction & real() const { return *pgfr; }
const ParGridFunction & imag() const { return *pgfi; }
/// Update the memory location of the real and imaginary ParGridFunction @a
/// pgfr and @a pgfi to match the ParComplexGridFunction.
void Sync() { pgfr->SyncMemory(*this); pgfi->SyncMemory(*this); }
/// Update the alias memory location of the real and imaginary
/// ParGridFunction @a pgfr and @a pgfi to match the ParComplexGridFunction.
void SyncAlias() { pgfr->SyncAliasMemory(*this); pgfi->SyncAliasMemory(*this); }
virtual double ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
const IntegrationRule *irs[] = NULL) const
{
@@ -500,14 +475,6 @@ public:
const ParLinearForm & real() const { return *plfr; }
const ParLinearForm & imag() const { return *plfi; }
/// Update the memory location of the real and imaginary ParLinearForm @a lfr
/// and @a lfi to match the ParComplexLinearForm.
void Sync() { plfr->SyncMemory(*this); plfi->SyncMemory(*this); }
/// Update the alias memory location of the real and imaginary ParLinearForm
/// @a plfr and @a plfi to match the ParComplexLinearForm.
void SyncAlias() { plfr->SyncAliasMemory(*this); plfi->SyncAliasMemory(*this); }
void Update(ParFiniteElementSpace *pf = NULL);
/// Assembles the linear form i.e. sums over all domain/bdr integrators.
-297
View File
@@ -1,297 +0,0 @@
#include "convergence.hpp"
using namespace std;
namespace mfem
{
void ConvergenceStudy::Reset()
{
counter=0;
dcounter=0;
fcounter=0;
cont_type=-1;
print_flag=1;
L2Errors.SetSize(0);
L2Rates.SetSize(0);
DErrors.SetSize(0);
DRates.SetSize(0);
EnErrors.SetSize(0);
EnRates.SetSize(0);
DGFaceErrors.SetSize(0);
DGFaceRates.SetSize(0);
ndofs.SetSize(0);
}
double ConvergenceStudy::GetNorm(GridFunction *gf, Coefficient *scalar_u,
VectorCoefficient *vector_u)
{
bool norm_set = false;
double norm=0.0;
int order = gf->FESpace()->GetOrder(0);
int order_quad = std::max(2, 2*order+1);
const IntegrationRule *irs[Geometry::NumGeom];
for (int i=0; i < Geometry::NumGeom; ++i)
{
irs[i] = &(IntRules.Get(i, order_quad));
}
#ifdef MFEM_USE_MPI
ParGridFunction *pgf = dynamic_cast<ParGridFunction *>(gf);
if (pgf)
{
ParMesh *pmesh = pgf->ParFESpace()->GetParMesh();
if (scalar_u)
{
norm = ComputeGlobalLpNorm(2.0,*scalar_u,*pmesh,irs);
}
else if (vector_u)
{
norm = ComputeGlobalLpNorm(2.0,*vector_u,*pmesh,irs);
}
norm_set = true;
}
#endif
if (!norm_set)
{
Mesh *mesh = gf->FESpace()->GetMesh();
if (scalar_u)
{
norm = ComputeLpNorm(2.0,*scalar_u,*mesh,irs);
}
else if (vector_u)
{
norm = ComputeLpNorm(2.0,*vector_u,*mesh,irs);
}
}
return norm;
}
void ConvergenceStudy::AddL2Error(GridFunction *gf,
Coefficient *scalar_u, VectorCoefficient *vector_u)
{
int tdofs=0;
#ifdef MFEM_USE_MPI
ParGridFunction *pgf = dynamic_cast<ParGridFunction *>(gf);
if (pgf)
{
MPI_Comm comm = pgf->ParFESpace()->GetComm();
int rank;
MPI_Comm_rank(comm, &rank);
print_flag = 0;
if (rank==0) { print_flag = 1; }
tdofs = pgf->ParFESpace()->GlobalTrueVSize();
}
#endif
if (!tdofs) { tdofs = gf->FESpace()->GetTrueVSize(); }
ndofs.Append(tdofs);
double L2Err;
if (scalar_u)
{
L2Err = gf->ComputeL2Error(*scalar_u);
CoeffNorm = GetNorm(gf,scalar_u,nullptr);
}
else if (vector_u)
{
L2Err = gf->ComputeL2Error(*vector_u);
CoeffNorm = GetNorm(gf,nullptr,vector_u);
}
else
{
MFEM_ABORT("Exact Solution Coefficient pointer is NULL");
}
L2Errors.Append(L2Err);
// Compute the rate of convergence by:
// rate = log (||u - u_h|| / ||u - u_{h/2}||)/log(2)
double val = (counter) ? log(L2Errors[counter-1]/L2Err)/log(2.0) : 0.0;
L2Rates.Append(val);
counter++;
}
void ConvergenceStudy::AddGf(GridFunction *gf, Coefficient *scalar_u,
VectorCoefficient *grad,
Coefficient *ell_coeff, double Nu)
{
cont_type = gf->FESpace()->FEColl()->GetContType();
MFEM_VERIFY((cont_type == mfem::FiniteElementCollection::CONTINUOUS) ||
(cont_type == mfem::FiniteElementCollection::DISCONTINUOUS),
"This constructor is intended for H1 or L2 Elements")
AddL2Error(gf,scalar_u, nullptr);
if (grad)
{
double GradErr = gf->ComputeGradError(grad);
DErrors.Append(GradErr);
double err = sqrt(L2Errors[counter-1]*L2Errors[counter-1]+GradErr*GradErr);
EnErrors.Append(err);
// Compute the rate of convergence by:
// rate = log (||u - u_h|| / ||u - u_{h/2}||)/log(2)
double val = (dcounter) ? log(DErrors[dcounter-1]/GradErr)/log(2.0) : 0.0;
double eval = (dcounter) ? log(EnErrors[dcounter-1]/err)/log(2.0) : 0.0;
DRates.Append(val);
EnRates.Append(eval);
CoeffDNorm = GetNorm(gf,nullptr,grad);
dcounter++;
MFEM_VERIFY(counter == dcounter,
"Number of added solutions and derivatives do not match")
}
if (cont_type == mfem::FiniteElementCollection::DISCONTINUOUS && ell_coeff)
{
double DGErr = gf->ComputeDGFaceJumpError(scalar_u,ell_coeff,Nu);
DGFaceErrors.Append(DGErr);
// Compute the rate of convergence by:
// rate = log (||u - u_h|| / ||u - u_{h/2}||)/log(2)
double val=(fcounter) ? log(DGFaceErrors[fcounter-1]/DGErr)/log(2.0):0.;
DGFaceRates.Append(val);
fcounter++;
MFEM_VERIFY(fcounter == counter, "Number of added solutions mismatch");
}
}
void ConvergenceStudy::AddGf(GridFunction *gf, VectorCoefficient *vector_u,
VectorCoefficient *curl, Coefficient *div)
{
cont_type = gf->FESpace()->FEColl()->GetContType();
AddL2Error(gf,nullptr,vector_u);
double DErr = 0.0;
bool derivative = false;
if (curl)
{
DErr = gf->ComputeCurlError(curl);
CoeffDNorm = GetNorm(gf,nullptr,curl);
derivative = true;
}
else if (div)
{
DErr = gf->ComputeDivError(div);
// update coefficient norm
CoeffDNorm = GetNorm(gf,div,nullptr);
derivative = true;
}
if (derivative)
{
double err = sqrt(L2Errors[counter-1]*L2Errors[counter-1] + DErr*DErr);
DErrors.Append(DErr);
EnErrors.Append(err);
// Compute the rate of convergence by:
// rate = log (||u - u_h|| / ||u - u_{h/2}||)/log(2)
double val = (dcounter) ? log(DErrors[dcounter-1]/DErr)/log(2.0) : 0.0;
double eval = (dcounter) ? log(EnErrors[dcounter-1]/err)/log(2.0) : 0.0;
DRates.Append(val);
EnRates.Append(eval);
dcounter++;
MFEM_VERIFY(counter == dcounter,
"Number of added solutions and derivatives do not match")
}
}
void ConvergenceStudy::Print(bool relative, std::ostream &out)
{
if (print_flag)
{
std::string title = (relative) ? "Relative " : "Absolute ";
out << "\n";
out << " -------------------------------------------" << "\n";
out << std::setw(21) << title << "L2 Error " << "\n";
out << " -------------------------------------------"
<< "\n";
out << std::right<< std::setw(11)<< "DOFs "<< std::setw(13) << "Error ";
out << std::setw(15) << "Rate " << "\n";
out << " -------------------------------------------"
<< "\n";
out << std::setprecision(4);
double d = (relative) ? CoeffNorm : 1.0;
for (int i =0; i<counter; i++)
{
out << std::right << std::setw(10)<< ndofs[i] << std::setw(16)
<< std::scientific << L2Errors[i]/d << std::setw(13)
<< std::fixed << L2Rates[i] << "\n";
}
out << "\n";
if (dcounter == counter)
{
std::string dname;
switch (cont_type)
{
case 0: dname = "Grad"; break;
case 1: dname = "Curl"; break;
case 2: dname = "Div"; break;
case 3: dname = "DG Grad"; break;
default: break;
}
out << " -------------------------------------------" << "\n";
out << std::setw(21) << title << dname << " Error " << "\n";
out << " -------------------------------------------" << "\n";
out << std::right<<std::setw(11)<< "DOFs "<< std::setw(13) << "Error";
out << std::setw(15) << "Rate " << "\n";
out << " -------------------------------------------"
<< "\n";
out << std::setprecision(4);
d = (relative) ? CoeffDNorm : 1.0;
for (int i =0; i<dcounter; i++)
{
out << std::right << std::setw(10)<< ndofs[i] << std::setw(16)
<< std::scientific << DErrors[i]/d << std::setw(13)
<< std::fixed << DRates[i] << "\n";
}
out << "\n";
switch (cont_type)
{
case 0: dname = "H1"; break;
case 1: dname = "H(Curl)"; break;
case 2: dname = "H(Div)"; break;
case 3: dname = "DG H1"; break;
default: break;
}
if (dcounter)
{
d = (relative) ?
sqrt(CoeffNorm*CoeffNorm + CoeffDNorm*CoeffDNorm):1.0;
out << " -------------------------------------------" << "\n";
out << std::setw(21) << title << dname << " Error " << "\n";
out << " -------------------------------------------" << "\n";
out << std::right<< std::setw(11)<< "DOFs "<< std::setw(13);
out << "Error ";
out << std::setw(15) << "Rate " << "\n";
out << " -------------------------------------------"
<< "\n";
out << std::setprecision(4);
for (int i =0; i<dcounter; i++)
{
out << std::right << std::setw(10)<< ndofs[i] << std::setw(16)
<< std::scientific << EnErrors[i]/d << std::setw(13)
<< std::fixed << EnRates[i] << "\n";
}
out << "\n";
}
if (cont_type == 3 && fcounter)
{
out << " -------------------------------------------" << "\n";
out << " DG Face Jump Error " << "\n";
out << " -------------------------------------------"
<< "\n";
out << std::right<< std::setw(11)<< "DOFs "<< std::setw(13);
out << "Error ";
out << std::setw(15) << "Rate " << "\n";
out << " -------------------------------------------"
<< "\n";
out << std::setprecision(4);
for (int i =0; i<fcounter; i++)
{
out << std::right << std::setw(10)<< ndofs[i] << std::setw(16)
<< std::scientific << DGFaceErrors[i] << std::setw(13)
<< std::fixed << DGFaceRates[i] << "\n";
}
out << "\n";
}
}
}
}
} // namespace mfem
-149
View File
@@ -1,149 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_CONVERGENCE
#define MFEM_CONVERGENCE
#include "../linalg/linalg.hpp"
#include "gridfunc.hpp"
#ifdef MFEM_USE_MPI
#include "pgridfunc.hpp"
#endif
namespace mfem
{
/** @brief Class to compute error and convergence rates.
It supports H1, H(curl) (ND elements), H(div) (RT elements) and L2 (DG).
For "smooth enough" solutions the Galerkin error measured in the appropriate
norm satisfies || u - u_h || ~ h^k
Here, k is called the asymptotic rate of convergence
For successive uniform h-refinements the rate can be estimated by
k = log(||u - u_h|| / ||u - u_{h/2}||)/log(2)
*/
class ConvergenceStudy
{
private:
// counters for solutions/derivatives
int counter=0;
int dcounter=0;
int fcounter=0;
// space continuity type
int cont_type=-1;
// printing flag for helpful for MPI calls
int print_flag=1;
// exact solution and derivatives
double CoeffNorm;
double CoeffDNorm;
// Arrays to store error/rates
Array<double> L2Errors, DGFaceErrors, DErrors, EnErrors;
Array<double> L2Rates, DGFaceRates, DRates, EnRates;
Array<int> ndofs;
void AddL2Error(GridFunction *gf, Coefficient *scalar_u,
VectorCoefficient *vector_u);
void AddGf(GridFunction *gf, Coefficient *scalar_u,
VectorCoefficient *grad=nullptr,
Coefficient *ell_coeff=nullptr, double Nu=1.0);
void AddGf(GridFunction *gf, VectorCoefficient *vector_u,
VectorCoefficient *curl, Coefficient *div);
// returns the L2-norm of scalar_u or vector_u
double GetNorm(GridFunction *gf, Coefficient *scalar_u,
VectorCoefficient *vector_u);
public:
/// Clear any internal data
void Reset();
/// Add L2 GridFunction, the exact solution and possibly its gradient and/or
/// DG face jumps parameters
void AddL2GridFunction(GridFunction *gf, Coefficient *scalar_u,
VectorCoefficient *grad=nullptr,
Coefficient *ell_coeff=nullptr, double Nu=1.0)
{
AddGf(gf, scalar_u, grad, ell_coeff, Nu);
}
/// Add H1 GridFunction, the exact solution and possibly its gradient
void AddH1GridFunction(GridFunction *gf, Coefficient *scalar_u,
VectorCoefficient *grad=nullptr)
{
AddGf(gf, scalar_u, grad);
}
/// Add H(curl) GridFunction, the exact solution and possibly its curl
void AddHcurlGridFunction(GridFunction *gf, VectorCoefficient *vector_u,
VectorCoefficient *curl=nullptr)
{
AddGf(gf, vector_u, curl, nullptr);
}
/// Add H(div) GridFunction, the exact solution and possibly its div
void AddHdivGridFunction(GridFunction *gf, VectorCoefficient *vector_u,
Coefficient *div=nullptr)
{
AddGf(gf,vector_u, nullptr, div);
}
/// Get the L2 error at step n
double GetL2Error(int n)
{
MFEM_VERIFY( n <= counter,"Step out of bounds")
return L2Errors[n];
}
/// Get all L2 errors
void GetL2Errors(Array<double> & L2Errors_)
{
L2Errors_ = L2Errors;
}
/// Get the Grad/Curl/Div error at step n
double GetDError(int n)
{
MFEM_VERIFY(n <= dcounter,"Step out of bounds")
return DErrors[n];
}
/// Get all Grad/Curl/Div errors
void GetDErrors(Array<double> & DErrors_)
{
DErrors_ = DErrors;
}
/// Get the DGFaceJumps error at step n
double GetDGFaceJumpsError(int n)
{
MFEM_VERIFY(n<= fcounter,"Step out of bounds")
return DGFaceErrors[n];
}
/// Get all DGFaceJumps errors
void GetDGFaceJumpsErrors(Array<double> & DGFaceErrors_)
{
DGFaceErrors_ = DGFaceErrors;
}
/// Print rates and errors
void Print(bool relative = false, std::ostream &out = mfem::out);
};
} // namespace mfem
#endif // MFEM_CONVERGENCE
-2
View File
@@ -563,8 +563,6 @@ void VisItDataCollection::LoadVisItRootFile(const std::string& root_name)
void VisItDataCollection::LoadMesh()
{
// GetMeshFileName() uses 'serial', so we need to set it in advance.
serial = (format == SERIAL_FORMAT);
std::string mesh_fname = GetMeshFileName();
named_ifgzstream file(mesh_fname);
// TODO: in parallel, check for errors on all processors
+3 -17
View File
@@ -77,9 +77,6 @@ public:
ElementTransformation();
/** @brief Force the reevaluation of the Jacobian in the next call. */
void Reset() { EvalState = 0; }
/** @brief Set the integration point @a ip that weights and Jacobians will
be evaluated at. */
void SetIntPoint(const IntegrationPoint *ip)
@@ -360,17 +357,9 @@ private:
// Evaluate the Hessian of the transformation at the IntPoint and store it
// in d2Fdx2.
virtual const DenseMatrix &EvalHessian();
public:
IsoparametricTransformation() : FElem(NULL) {}
/// Set the element that will be used to compute the transformations
void SetFE(const FiniteElement *FE)
{
MFEM_ASSERT(FE != NULL, "Must provide a valid FiniteElement object!");
EvalState = (FE != FElem) ? 0 : EvalState;
FElem = FE; geom = FE->GetGeomType();
}
void SetFE(const FiniteElement *FE) { FElem = FE; geom = FE->GetGeomType(); }
/// Get the current element used to compute the transformations
const FiniteElement* GetFE() const { return FElem; }
@@ -385,15 +374,12 @@ public:
the column-vector of all basis functions evaluated at \f$ \hat x \f$ .
The columns of @a P represent the control points in physical space
defining the transformation. */
void SetPointMat(const DenseMatrix &pm) { PointMat = pm; EvalState = 0; }
void SetPointMat(const DenseMatrix &pm) { PointMat = pm; }
/// Return the stored point matrix.
const DenseMatrix &GetPointMat() const { return PointMat; }
/// @brief Write access to the stored point matrix. Use with caution.
/** If the point matrix is altered using this member function the Reset
function should also be called to force the reevaluation of the
Jacobian, etc.. */
/// Write access to the stored point matrix. Use with caution.
DenseMatrix &GetPointMat() { return PointMat; }
/// Set the FiniteElement Geometry for the reference elements being used.
-203
View File
@@ -139,12 +139,6 @@ void FiniteElement::Project (
mfem_error ("FiniteElement::Project (...) (vector) is not overloaded !");
}
void FiniteElement::ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{
mfem_error ("FiniteElement::ProjectFromNodes() (vector) is not overloaded!");
}
void FiniteElement::ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{
@@ -931,23 +925,6 @@ void VectorFiniteElement::Project_RT(
}
}
void VectorFiniteElement::Project_RT(
const double *nk, const Array<int> &d2n,
Vector &vc, ElementTransformation &Trans, Vector &dofs) const
{
const int sdim = Trans.GetSpaceDim();
const bool square_J = (dim == sdim);
for (int k = 0; k < dof; k++)
{
Trans.SetIntPoint(&Nodes.IntPoint(k));
// dof_k = nk^t adj(J) xk
Vector vk(vc.GetData()+k*sdim, sdim);
dofs(k) = Trans.AdjugateJacobian().InnerProduct(vk, nk + d2n[k]*dim);
if (!square_J) { dofs(k) /= Trans.Weight(); }
}
}
void VectorFiniteElement::ProjectMatrixCoefficient_RT(
const double *nk, const Array<int> &d2n,
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
@@ -1124,19 +1101,6 @@ void VectorFiniteElement::Project_ND(
}
}
void VectorFiniteElement::Project_ND(
const double *tk, const Array<int> &d2t,
Vector &vc, ElementTransformation &Trans, Vector &dofs) const
{
for (int k = 0; k < dof; k++)
{
Trans.SetIntPoint(&Nodes.IntPoint(k));
Vector vk(vc.GetData()+k*dim, dim);
// dof_k = xk^t J tk
dofs(k) = Trans.Jacobian().InnerProduct(tk + d2t[k]*dim, vk);
}
}
void VectorFiniteElement::ProjectMatrixCoefficient_ND(
const double *tk, const Array<int> &d2t,
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
@@ -7070,95 +7034,6 @@ void Poly_1D::Basis::Eval(const double y, Vector &u, Vector &d) const
}
}
void Poly_1D::Basis::Eval(const double y, Vector &u, Vector &d,
Vector &d2) const
{
MFEM_VERIFY(etype == Barycentric,
"Basis::Eval with second order derivatives not implemented for"
" etype = " << etype);
switch (etype)
{
case ChangeOfBasis:
{
CalcBasis(Ai.Width() - 1, y, x, w);
Ai.Mult(x, u);
Ai.Mult(w, d);
// set d2 (not implemented yet)
break;
}
case Barycentric:
{
int i, k, p = x.Size() - 1;
double l, lp, lp2, lk, sk, si, sk2;
if (p == 0)
{
u(0) = 1.0;
d(0) = 0.0;
d2(0) = 0.0;
return;
}
lk = 1.0;
for (k = 0; k < p; k++)
{
if (y >= (x(k) + x(k+1))/2)
{
lk *= y - x(k);
}
else
{
for (i = k+1; i <= p; i++)
{
lk *= y - x(i);
}
break;
}
}
l = lk * (y - x(k));
sk = 0.0;
sk2 = 0.0;
for (i = 0; i < k; i++)
{
si = 1.0/(y - x(i));
sk += si;
sk2 -= si * si;
u(i) = l * si * w(i);
}
u(k) = lk * w(k);
for (i++; i <= p; i++)
{
si = 1.0/(y - x(i));
sk += si;
sk2 -= si * si;
u(i) = l * si * w(i);
}
lp = l * sk + lk;
lp2 = lp * sk + l * sk2 + sk * lk;
for (i = 0; i < k; i++)
{
d(i) = (lp * w(i) - u(i))/(y - x(i));
d2(i) = (lp2 * w(i) - 2 * d(i))/(y - x(i));
}
d(k) = sk * u(k);
d2(k) = sk2 * u(k) + sk * d(k);
for (i++; i <= p; i++)
{
d(i) = (lp * w(i) - u(i))/(y - x(i));
d2(i) = (lp2 * w(i) - 2 * d(i))/(y - x(i));
}
break;
}
case Positive:
CalcBernstein(x.Size() - 1, y, u, d);
break;
default: break;
}
}
const int *Poly_1D::Binom(const int p)
{
if (binom.NumCols() <= p)
@@ -7714,7 +7589,6 @@ H1_SegmentElement::H1_SegmentElement(const int p, const int btype)
#ifndef MFEM_THREAD_SAFE
shape_x.SetSize(p+1);
dshape_x.SetSize(p+1);
d2shape_x.SetSize(p+1);
#endif
Nodes.IntPoint(0).x = cp[0];
@@ -7763,25 +7637,6 @@ void H1_SegmentElement::CalcDShape(const IntegrationPoint &ip,
}
}
void H1_SegmentElement::CalcHessian(const IntegrationPoint &ip,
DenseMatrix &Hessian) const
{
const int p = order;
#ifdef MFEM_THREAD_SAFE
Vector shape_x(p+1), dshape_x(p+1), d2shape_x(p+1);
#endif
basis1d.Eval(ip.x, shape_x, dshape_x, d2shape_x);
Hessian(0,0) = d2shape_x(0);
Hessian(1,0) = d2shape_x(p);
for (int i = 1; i < p; i++)
{
Hessian(i+1,0) = d2shape_x(i);
}
}
void H1_SegmentElement::ProjectDelta(int vertex, Vector &dofs) const
{
const int p = order;
@@ -7822,8 +7677,6 @@ H1_QuadrilateralElement::H1_QuadrilateralElement(const int p, const int btype)
shape_y.SetSize(p1);
dshape_x.SetSize(p1);
dshape_y.SetSize(p1);
d2shape_x.SetSize(p1);
d2shape_y.SetSize(p1);
#endif
int o = 0;
@@ -7877,30 +7730,6 @@ void H1_QuadrilateralElement::CalcDShape(const IntegrationPoint &ip,
}
}
void H1_QuadrilateralElement::CalcHessian(const IntegrationPoint &ip,
DenseMatrix &Hessian) const
{
const int p = order;
#ifdef MFEM_THREAD_SAFE
Vector shape_x(p+1), shape_y(p+1), dshape_x(p+1), dshape_y(p+1),
d2shape_x(p+1), d2shape_y(p+1);
#endif
basis1d.Eval(ip.x, shape_x, dshape_x, d2shape_x);
basis1d.Eval(ip.y, shape_y, dshape_y, d2shape_y);
for (int o = 0, j = 0; j <= p; j++)
{
for (int i = 0; i <= p; i++)
{
Hessian(dof_map[o],0) = d2shape_x(i)* shape_y(j);
Hessian(dof_map[o],1) = dshape_x(i)* dshape_y(j);
Hessian(dof_map[o],2) = shape_x(i)*d2shape_y(j); o++;
}
}
}
void H1_QuadrilateralElement::ProjectDelta(int vertex, Vector &dofs) const
{
const int p = order;
@@ -7964,9 +7793,6 @@ H1_HexahedronElement::H1_HexahedronElement(const int p, const int btype)
dshape_x.SetSize(p1);
dshape_y.SetSize(p1);
dshape_z.SetSize(p1);
d2shape_x.SetSize(p1);
d2shape_y.SetSize(p1);
d2shape_z.SetSize(p1);
#endif
int o = 0;
@@ -8023,35 +7849,6 @@ void H1_HexahedronElement::CalcDShape(const IntegrationPoint &ip,
}
}
void H1_HexahedronElement::CalcHessian(const IntegrationPoint &ip,
DenseMatrix &Hessian) const
{
const int p = order;
#ifdef MFEM_THREAD_SAFE
Vector shape_x(p+1), shape_y(p+1), shape_z(p+1);
Vector dshape_x(p+1), dshape_y(p+1), dshape_z(p+1);
Vector d2shape_x(p+1), d2shape_y(p+1), ds2hape_z(p+1);
#endif
basis1d.Eval(ip.x, shape_x, dshape_x, d2shape_x);
basis1d.Eval(ip.y, shape_y, dshape_y, d2shape_y);
basis1d.Eval(ip.z, shape_z, dshape_z, d2shape_z);
for (int o = 0, k = 0; k <= p; k++)
for (int j = 0; j <= p; j++)
for (int i = 0; i <= p; i++)
{
Hessian(dof_map[o],0) = d2shape_x(i)* shape_y(j)* shape_z(k);
Hessian(dof_map[o],1) = dshape_x(i)* dshape_y(j)* shape_z(k);
Hessian(dof_map[o],2) = dshape_x(i)* shape_y(j)* dshape_z(k);
Hessian(dof_map[o],3) = shape_x(i)*d2shape_y(j)* shape_z(k);
Hessian(dof_map[o],4) = shape_x(i)* dshape_y(j)* dshape_z(k);
Hessian(dof_map[o],5) = shape_x(i)* shape_y(j)*d2shape_z(k);
o++;
}
}
void H1_HexahedronElement::ProjectDelta(int vertex, Vector &dofs) const
{
const int p = order;
+10 -60
View File
@@ -446,7 +446,7 @@ public:
/** Each row of the result DenseMatrix @a Hessian contains upper triangular
part of the Hessian of one shape function.
The order in 2D is {u_xx, u_xy, u_yy}.
The size (#dof x (#dim (#dim+1)/2) of @a Hessian must be set in advance.*/
The size (#dof x (#dim (#dim-1)/2) of @a Hessian must be set in advance.*/
virtual void CalcHessian (const IntegrationPoint &ip,
DenseMatrix &Hessian) const;
@@ -504,21 +504,14 @@ public:
/** @brief Given a coefficient and a transformation, compute its projection
(approximation) in the local finite dimensional space in terms
of the degrees of freedom. */
virtual void Project(Coefficient &coeff,
ElementTransformation &Trans, Vector &dofs) const;
virtual void Project (Coefficient &coeff,
ElementTransformation &Trans, Vector &dofs) const;
/** @brief Given a vector coefficient and a transformation, compute its
projection (approximation) in the local finite dimensional space
in terms of the degrees of freedom. (VectorFiniteElements) */
virtual void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const;
/** @brief Given a vector of values at the finite element nodes and a
transformation, compute its projection (approximation) in the local
finite dimensional space in terms of the degrees of freedom. Valid for
VectorFiniteElements. */
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const;
virtual void Project (VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const;
/** @brief Given a matrix coefficient and a transformation, compute an
approximation ("projection") in the local finite dimensional space in
@@ -804,12 +797,7 @@ protected:
VectorCoefficient &vc, ElementTransformation &Trans,
Vector &dofs) const;
/// Projects the vector of values given at FE nodes to RT space
void Project_RT(const double *nk, const Array<int> &d2n,
Vector &vc, ElementTransformation &Trans,
Vector &dofs) const;
/// Project the rows of the matrix coefficient in an RT space
// project the rows of the matrix coefficient in an RT space
void ProjectMatrixCoefficient_RT(
const double *nk, const Array<int> &d2n,
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const;
@@ -837,12 +825,7 @@ protected:
VectorCoefficient &vc, ElementTransformation &Trans,
Vector &dofs) const;
/// Projects the vector of values given at FE nodes to ND space
void Project_ND(const double *tk, const Array<int> &d2t,
Vector &vc, ElementTransformation &Trans,
Vector &dofs) const;
/// Project the rows of the matrix coefficient in an ND space
/// project the rows of the matrix coefficient in an ND space
void ProjectMatrixCoefficient_ND(
const double *tk, const Array<int> &d2t,
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const;
@@ -1867,7 +1850,6 @@ public:
Basis(const int p, const double *nodes, EvalType etype = Barycentric);
void Eval(const double x, Vector &u) const;
void Eval(const double x, Vector &u, Vector &d) const;
void Eval(const double x, Vector &u, Vector &d, Vector &d2) const;
};
private:
@@ -2118,7 +2100,7 @@ class H1_SegmentElement : public NodalTensorFiniteElement
{
private:
#ifndef MFEM_THREAD_SAFE
mutable Vector shape_x, dshape_x, d2shape_x;
mutable Vector shape_x, dshape_x;
#endif
public:
@@ -2127,8 +2109,6 @@ public:
virtual void CalcShape(const IntegrationPoint &ip, Vector &shape) const;
virtual void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const;
virtual void CalcHessian(const IntegrationPoint &ip,
DenseMatrix &Hessian) const;
virtual void ProjectDelta(int vertex, Vector &dofs) const;
};
@@ -2138,7 +2118,7 @@ class H1_QuadrilateralElement : public NodalTensorFiniteElement
{
private:
#ifndef MFEM_THREAD_SAFE
mutable Vector shape_x, shape_y, dshape_x, dshape_y, d2shape_x, d2shape_y;
mutable Vector shape_x, shape_y, dshape_x, dshape_y;
#endif
public:
@@ -2148,8 +2128,6 @@ public:
virtual void CalcShape(const IntegrationPoint &ip, Vector &shape) const;
virtual void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const;
virtual void CalcHessian(const IntegrationPoint &ip,
DenseMatrix &Hessian) const;
virtual void ProjectDelta(int vertex, Vector &dofs) const;
};
@@ -2159,8 +2137,7 @@ class H1_HexahedronElement : public NodalTensorFiniteElement
{
private:
#ifndef MFEM_THREAD_SAFE
mutable Vector shape_x, shape_y, shape_z, dshape_x, dshape_y, dshape_z,
d2shape_x, d2shape_y, d2shape_z;
mutable Vector shape_x, shape_y, shape_z, dshape_x, dshape_y, dshape_z;
#endif
public:
@@ -2169,8 +2146,6 @@ public:
virtual void CalcShape(const IntegrationPoint &ip, Vector &shape) const;
virtual void CalcDShape(const IntegrationPoint &ip,
DenseMatrix &dshape) const;
virtual void CalcHessian(const IntegrationPoint &ip,
DenseMatrix &Hessian) const;
virtual void ProjectDelta(int vertex, Vector &dofs) const;
};
@@ -2706,9 +2681,6 @@ public:
virtual void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const
{ Project_RT(nk, dof2nk, vc, Trans, dofs); }
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{ Project_RT(nk, dof2nk, vc, Trans, dofs); }
virtual void ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{ ProjectMatrixCoefficient_RT(nk, dof2nk, mc, T, dofs); }
@@ -2767,9 +2739,6 @@ public:
virtual void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const
{ Project_RT(nk, dof2nk, vc, Trans, dofs); }
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{ Project_RT(nk, dof2nk, vc, Trans, dofs); }
virtual void ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{ ProjectMatrixCoefficient_RT(nk, dof2nk, mc, T, dofs); }
@@ -2821,9 +2790,6 @@ public:
virtual void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const
{ Project_RT(nk, dof2nk, vc, Trans, dofs); }
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{ Project_RT(nk, dof2nk, vc, Trans, dofs); }
virtual void ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{ ProjectMatrixCoefficient_RT(nk, dof2nk, mc, T, dofs); }
@@ -2881,9 +2847,6 @@ public:
virtual void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const
{ Project_RT(nk, dof2nk, vc, Trans, dofs); }
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{ Project_RT(nk, dof2nk, vc, Trans, dofs); }
virtual void ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{ ProjectMatrixCoefficient_RT(nk, dof2nk, mc, T, dofs); }
@@ -2943,10 +2906,6 @@ public:
ElementTransformation &Trans, Vector &dofs) const
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
virtual void ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{ ProjectMatrixCoefficient_ND(tk, dof2tk, mc, T, dofs); }
@@ -3006,9 +2965,6 @@ public:
virtual void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
virtual void ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{ ProjectMatrixCoefficient_ND(tk, dof2tk, mc, T, dofs); }
@@ -3060,9 +3016,6 @@ public:
virtual void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
virtual void ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{ ProjectMatrixCoefficient_ND(tk, dof2tk, mc, T, dofs); }
@@ -3119,9 +3072,6 @@ public:
virtual void Project(VectorCoefficient &vc,
ElementTransformation &Trans, Vector &dofs) const
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
virtual void ProjectFromNodes(Vector &vc, ElementTransformation &Trans,
Vector &dofs) const
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
virtual void ProjectMatrixCoefficient(
MatrixCoefficient &mc, ElementTransformation &T, Vector &dofs) const
{ ProjectMatrixCoefficient_ND(tk, dof2tk, mc, T, dofs); }
-1
View File
@@ -19,7 +19,6 @@
#include "eltrans.hpp"
#include "coefficient.hpp"
#include "complex_fem.hpp"
#include "convergence.hpp"
#include "lininteg.hpp"
#include "nonlininteg.hpp"
#include "bilininteg.hpp"
-3
View File
@@ -440,7 +440,6 @@ void FiniteElementSpace::MarkerToList(const Array<int> &marker,
if (marker[i]) { num_marked++; }
}
list.SetSize(0);
list.HostWrite();
list.Reserve(num_marked);
for (int i = 0; i < marker.Size(); i++)
{
@@ -452,9 +451,7 @@ void FiniteElementSpace::MarkerToList(const Array<int> &marker,
void FiniteElementSpace::ListToMarker(const Array<int> &list, int marker_size,
Array<int> &marker, int mark_val)
{
list.HostRead(); // make sure we can read the array on host
marker.SetSize(marker_size);
marker.HostWrite();
marker = 0;
for (int i = 0; i < list.Size(); i++)
{
+124 -246
View File
@@ -1833,19 +1833,6 @@ void GridFunction::ImposeBounds(int i, const Vector &weights,
ImposeBounds(i, weights, minv, maxv);
}
void GridFunction::RestrictConforming()
{
const SparseMatrix *R = fes->GetRestrictionMatrix();
const Operator *P = fes->GetProlongationMatrix();
if (P && R)
{
Vector tmp(R->Height());
R->Mult(*this, tmp);
P->Mult(tmp, *this);
}
}
void GridFunction::GetNodalValues(Vector &nval, int vdim) const
{
int i, j;
@@ -2614,7 +2601,11 @@ double GridFunction::ComputeL2Error(
}
}
return (error < 0.0) ? -sqrt(-error) : sqrt(error);
if (error < 0.0)
{
return -sqrt(-error);
}
return sqrt(error);
}
double GridFunction::ComputeL2Error(
@@ -2655,199 +2646,94 @@ double GridFunction::ComputeL2Error(
}
}
return (error < 0.0) ? -sqrt(-error) : sqrt(error);
}
double GridFunction::ComputeGradError(VectorCoefficient *exgrad,
const IntegrationRule *irs[]) const
{
double error = 0.0;
const FiniteElement *fe;
ElementTransformation *Tr;
Array<int> dofs;
Vector grad;
int intorder;
int dim = fes->GetMesh()->SpaceDimension();
Vector vec(dim);
for (int i = 0; i < fes->GetNE(); i++)
if (error < 0.0)
{
fe = fes->GetFE(i);
Tr = fes->GetElementTransformation(i);
intorder = 2*fe->GetOrder() + 3; // <--------
const IntegrationRule *ir;
if (irs)
{
ir = irs[fe->GetGeomType()];
}
else
{
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
}
fes->GetElementDofs(i, dofs);
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
Tr->SetIntPoint(&ip);
GetGradient(*Tr,grad);
exgrad->Eval(vec,*Tr,ip);
vec-=grad;
error += ip.weight * Tr->Weight() * (vec * vec);
}
return -sqrt(-error);
}
return (error < 0.0) ? -sqrt(-error) : sqrt(error);
return sqrt(error);
}
double GridFunction::ComputeCurlError(VectorCoefficient *excurl,
const IntegrationRule *irs[]) const
double GridFunction::ComputeH1Error(
Coefficient *exsol, VectorCoefficient *exgrad,
Coefficient *ell_coeff, double Nu, int norm_type) const
{
double error = 0.0;
const FiniteElement *fe;
ElementTransformation *Tr;
Array<int> dofs;
Vector curl;
int intorder;
int dim = fes->GetMesh()->SpaceDimension();
int n = (dim == 3) ? dim : 1;
Vector vec(n);
for (int i = 0; i < fes->GetNE(); i++)
{
fe = fes->GetFE(i);
Tr = fes->GetElementTransformation(i);
intorder = 2*fe->GetOrder() + 3;
const IntegrationRule *ir;
if (irs)
{
ir = irs[fe->GetGeomType()];
}
else
{
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
}
fes->GetElementDofs(i, dofs);
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
Tr->SetIntPoint(&ip);
GetCurl(*Tr,curl);
excurl->Eval(vec,*Tr,ip);
vec-=curl;
error += ip.weight * Tr->Weight() * ( vec * vec );
}
}
return (error < 0.0) ? -sqrt(-error) : sqrt(error);
}
double GridFunction::ComputeDivError(
Coefficient *exdiv, const IntegrationRule *irs[]) const
{
double error = 0.0, a;
const FiniteElement *fe;
ElementTransformation *Tr;
Array<int> dofs;
int intorder;
for (int i = 0; i < fes->GetNE(); i++)
{
fe = fes->GetFE(i);
Tr = fes->GetElementTransformation(i);
intorder = 2*fe->GetOrder() + 3;
const IntegrationRule *ir;
if (irs)
{
ir = irs[fe->GetGeomType()];
}
else
{
ir = &(IntRules.Get(fe->GetGeomType(), intorder));
}
fes->GetElementDofs(i, dofs);
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
Tr->SetIntPoint (&ip);
a = GetDivergence(*Tr) - exdiv->Eval(*Tr, ip);
error += ip.weight * Tr->Weight() * a * a;
}
}
return (error < 0.0) ? -sqrt(-error) : sqrt(error);
}
double GridFunction::ComputeDGFaceJumpError(Coefficient *exsol,
Coefficient *ell_coeff, double Nu,
const IntegrationRule *irs[]) const
{
int fdof, dim, intorder, k;
// assuming vdim is 1
int i, fdof, dim, intorder, j, k;
Mesh *mesh;
const FiniteElement *fe;
ElementTransformation *transf;
FaceElementTransformations *face_elem_transf;
Vector shape, el_dofs, err_val, ell_coeff_val;
Vector e_grad, a_grad, shape, el_dofs, err_val, ell_coeff_val;
DenseMatrix dshape, dshapet, Jinv;
Array<int> vdofs;
IntegrationPoint eip;
double error = 0.0;
mesh = fes->GetMesh();
dim = mesh->Dimension();
e_grad.SetSize(dim);
a_grad.SetSize(dim);
Jinv.SetSize(dim);
for (int i = 0; i < mesh->GetNumFaces(); i++)
{
face_elem_transf = mesh->GetFaceElementTransformations(i, 5);
int i1 = face_elem_transf->Elem1No;
int i2 = face_elem_transf->Elem2No;
intorder = fes->GetFE(i1)->GetOrder();
if (i2 >= 0)
if ( (k = fes->GetFE(i2)->GetOrder()) > intorder )
{
intorder = k;
}
intorder = 2 * intorder; // <-------------
const IntegrationRule *ir;
if (irs)
if (norm_type & 1)
for (i = 0; i < mesh->GetNE(); i++)
{
ir = irs[face_elem_transf->GetGeometryType()];
}
else
{
ir = &(IntRules.Get(face_elem_transf->GetGeometryType(), intorder));
}
err_val.SetSize(ir->GetNPoints());
ell_coeff_val.SetSize(ir->GetNPoints());
// side 1
transf = face_elem_transf->Elem1;
fe = fes->GetFE(i1);
fdof = fe->GetDof();
fes->GetElementVDofs(i1, vdofs);
shape.SetSize(fdof);
el_dofs.SetSize(fdof);
for (k = 0; k < fdof; k++)
if (vdofs[k] >= 0)
{
el_dofs(k) = (*this)(vdofs[k]);
}
else
{
el_dofs(k) = - (*this)(-1-vdofs[k]);
}
for (int j = 0; j < ir->GetNPoints(); j++)
{
face_elem_transf->Loc1.Transform(ir->IntPoint(j), eip);
fe->CalcShape(eip, shape);
transf->SetIntPoint(&eip);
ell_coeff_val(j) = ell_coeff->Eval(*transf, eip);
err_val(j) = exsol->Eval(*transf, eip) - (shape * el_dofs);
}
if (i2 >= 0)
{
// side 2
face_elem_transf = mesh->GetFaceElementTransformations(i, 10);
transf = face_elem_transf->Elem2;
fe = fes->GetFE(i2);
fe = fes->GetFE(i);
fdof = fe->GetDof();
fes->GetElementVDofs(i2, vdofs);
transf = mesh->GetElementTransformation(i);
el_dofs.SetSize(fdof);
dshape.SetSize(fdof, dim);
dshapet.SetSize(fdof, dim);
intorder = 2 * fe->GetOrder(); // <----------
const IntegrationRule &ir = IntRules.Get(fe->GetGeomType(), intorder);
fes->GetElementVDofs(i, vdofs);
for (k = 0; k < fdof; k++)
if (vdofs[k] >= 0)
{
el_dofs(k) = (*this)(vdofs[k]);
}
else
{
el_dofs(k) = - (*this)(-1-vdofs[k]);
}
for (j = 0; j < ir.GetNPoints(); j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
fe->CalcDShape(ip, dshape);
transf->SetIntPoint(&ip);
exgrad->Eval(e_grad, *transf, ip);
CalcInverse(transf->Jacobian(), Jinv);
Mult(dshape, Jinv, dshapet);
dshapet.MultTranspose(el_dofs, a_grad);
e_grad -= a_grad;
error += (ip.weight * transf->Weight() *
ell_coeff->Eval(*transf, ip) *
(e_grad * e_grad));
}
}
if (norm_type & 2)
for (i = 0; i < mesh->GetNFaces(); i++)
{
face_elem_transf = mesh->GetFaceElementTransformations(i, 5);
int i1 = face_elem_transf->Elem1No;
int i2 = face_elem_transf->Elem2No;
intorder = fes->GetFE(i1)->GetOrder();
if (i2 >= 0)
if ( (k = fes->GetFE(i2)->GetOrder()) > intorder )
{
intorder = k;
}
intorder = 2 * intorder; // <-------------
const IntegrationRule &ir =
IntRules.Get(face_elem_transf->GetGeometryType(), intorder);
err_val.SetSize(ir.GetNPoints());
ell_coeff_val.SetSize(ir.GetNPoints());
// side 1
transf = face_elem_transf->Elem1;
fe = fes->GetFE(i1);
fdof = fe->GetDof();
fes->GetElementVDofs(i1, vdofs);
shape.SetSize(fdof);
el_dofs.SetSize(fdof);
for (k = 0; k < fdof; k++)
@@ -2859,69 +2745,60 @@ double GridFunction::ComputeDGFaceJumpError(Coefficient *exsol,
{
el_dofs(k) = - (*this)(-1-vdofs[k]);
}
for (int j = 0; j < ir->GetNPoints(); j++)
for (j = 0; j < ir.GetNPoints(); j++)
{
face_elem_transf->Loc2.Transform(ir->IntPoint(j), eip);
face_elem_transf->Loc1.Transform(ir.IntPoint(j), eip);
fe->CalcShape(eip, shape);
transf->SetIntPoint(&eip);
ell_coeff_val(j) += ell_coeff->Eval(*transf, eip);
ell_coeff_val(j) *= 0.5;
err_val(j) -= (exsol->Eval(*transf, eip) - (shape * el_dofs));
ell_coeff_val(j) = ell_coeff->Eval(*transf, eip);
err_val(j) = exsol->Eval(*transf, eip) - (shape * el_dofs);
}
if (i2 >= 0)
{
// side 2
face_elem_transf = mesh->GetFaceElementTransformations(i, 10);
transf = face_elem_transf->Elem2;
fe = fes->GetFE(i2);
fdof = fe->GetDof();
fes->GetElementVDofs(i2, vdofs);
shape.SetSize(fdof);
el_dofs.SetSize(fdof);
for (k = 0; k < fdof; k++)
if (vdofs[k] >= 0)
{
el_dofs(k) = (*this)(vdofs[k]);
}
else
{
el_dofs(k) = - (*this)(-1-vdofs[k]);
}
for (j = 0; j < ir.GetNPoints(); j++)
{
face_elem_transf->Loc2.Transform(ir.IntPoint(j), eip);
fe->CalcShape(eip, shape);
transf->SetIntPoint(&eip);
ell_coeff_val(j) += ell_coeff->Eval(*transf, eip);
ell_coeff_val(j) *= 0.5;
err_val(j) -= (exsol->Eval(*transf, eip) - (shape * el_dofs));
}
}
face_elem_transf = mesh->GetFaceElementTransformations(i, 16);
transf = face_elem_transf;
for (j = 0; j < ir.GetNPoints(); j++)
{
const IntegrationPoint &ip = ir.IntPoint(j);
transf->SetIntPoint(&ip);
error += (ip.weight * Nu * ell_coeff_val(j) *
pow(transf->Weight(), 1.0-1.0/(dim-1)) *
err_val(j) * err_val(j));
}
}
face_elem_transf = mesh->GetFaceElementTransformations(i, 16);
transf = face_elem_transf;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
transf->SetIntPoint(&ip);
error += (ip.weight * Nu * ell_coeff_val(j) *
pow(transf->Weight(), 1.0-1.0/(dim-1)) *
err_val(j) * err_val(j));
}
if (error < 0.0)
{
return -sqrt(-error);
}
return (error < 0.0) ? -sqrt(-error) : sqrt(error);
}
double GridFunction::ComputeH1Error(Coefficient *exsol,
VectorCoefficient *exgrad,
Coefficient *ell_coef, double Nu,
int norm_type) const
{
double error1 = 0.0;
double error2 = 0.0;
if (norm_type & 1) { error1 = GridFunction::ComputeGradError(exgrad); }
if (norm_type & 2) { error2 = GridFunction::ComputeDGFaceJumpError(exsol,ell_coef,Nu); }
return sqrt(error1 * error1 + error2 * error2);
}
double GridFunction::ComputeH1Error(Coefficient *exsol,
VectorCoefficient *exgrad,
const IntegrationRule *irs[]) const
{
double L2error = GridFunction::ComputeLpError(2.0,*exsol,NULL,irs);
double GradError = ComputeGradError(exgrad,irs);
return sqrt(L2error*L2error + GradError*GradError);
}
double GridFunction::ComputeHDivError(VectorCoefficient *exsol,
Coefficient *exdiv,
const IntegrationRule *irs[]) const
{
double L2error = GridFunction::ComputeLpError(2.0,*exsol,NULL,NULL,irs);
double DivError = ComputeDivError(exdiv,irs);
return sqrt(L2error*L2error + DivError*DivError);
}
double GridFunction::ComputeHCurlError(VectorCoefficient *exsol,
VectorCoefficient *excurl,
const IntegrationRule *irs[]) const
{
double L2error = GridFunction::ComputeLpError(2.0,*exsol,NULL,NULL,irs);
double CurlError = ComputeCurlError(excurl,irs);
return sqrt(L2error*L2error + CurlError*CurlError);
return sqrt(error);
}
double GridFunction::ComputeMaxError(
@@ -2977,6 +2854,7 @@ double GridFunction::ComputeMaxError(
}
}
}
return error;
}
-46
View File
@@ -334,11 +334,6 @@ public:
void ImposeBounds(int i, const Vector &weights,
double _min = 0.0, double _max = infinity());
/** On a non-conforming mesh, make sure the function lies in the conforming
space by multiplying with R and then with P, the conforming restriction
and prolongation matrices of the space, respectively. */
void RestrictConforming();
/** @brief Project the @a src GridFunction to @a this GridFunction, both of
which must be on the same mesh. */
/** The current implementation assumes that all elements use the same
@@ -427,7 +422,6 @@ public:
virtual void ProjectBdrCoefficientTangent(VectorCoefficient &vcoeff,
Array<int> &bdr_attr);
virtual double ComputeL2Error(Coefficient &exsol,
const IntegrationRule *irs[] = NULL) const
{ return ComputeLpError(2.0, exsol, NULL, irs); }
@@ -439,50 +433,10 @@ public:
const IntegrationRule *irs[] = NULL,
Array<int> *elems = NULL) const;
/// Returns ||grad u_ex - grad u_h||_L2 for H1 or L2 elements
virtual double ComputeGradError(VectorCoefficient *exgrad,
const IntegrationRule *irs[] = NULL) const;
/// Returns ||curl u_ex - curl u_h||_L2 for ND elements
virtual double ComputeCurlError(VectorCoefficient *excurl,
const IntegrationRule *irs[] = NULL) const;
/// Returns ||div u_ex - div u_h||_L2 for RT elements
virtual double ComputeDivError(Coefficient *exdiv,
const IntegrationRule *irs[] = NULL) const;
/// Returns the Face Jumps error for L2 elements
virtual double ComputeDGFaceJumpError(Coefficient *exsol,
Coefficient *ell_coeff,
double Nu,
const IntegrationRule *irs[] = NULL)
const;
/** This method is kept for backward compatibility.
Returns either the H1-seminorm, or the DG face jumps error, or both
depending on norm_type = 1, 2, 3. Additional arguments for the DG face
jumps norm: ell_coeff: mesh-depended coefficient (weight) Nu: scalar
constant weight */
virtual double ComputeH1Error(Coefficient *exsol, VectorCoefficient *exgrad,
Coefficient *ell_coef, double Nu,
int norm_type) const;
/// Returns the error measured in H1-norm for H1 elements or in "broken"
/// H1-norm for L2 elements
virtual double ComputeH1Error(Coefficient *exsol, VectorCoefficient *exgrad,
const IntegrationRule *irs[] = NULL) const;
/// Returns the error measured in H(div)-norm for RT elements
virtual double ComputeHDivError(VectorCoefficient *exsol,
Coefficient *exdiv,
const IntegrationRule *irs[] = NULL) const;
/// Returns the error measured in H(curl)-norm for ND elements
virtual double ComputeHCurlError(VectorCoefficient *exsol,
VectorCoefficient *excurl,
const IntegrationRule *irs[] = NULL) const;
virtual double ComputeMaxError(Coefficient &exsol,
const IntegrationRule *irs[] = NULL) const
{
+86 -403
View File
@@ -29,13 +29,10 @@ namespace mfem
{
FindPointsGSLIB::FindPointsGSLIB()
: mesh(NULL), meshsplit(NULL), ir_simplex(NULL),
fdata2D(NULL), fdata3D(NULL), cr(NULL), gsl_comm(NULL),
dim(-1), points_cnt(0), setupflag(false), default_interp_value(0),
avgtype(AvgType::ARITHMETIC)
: mesh(NULL), ir_simplex(NULL), fdata2D(NULL), fdata3D(NULL),
dim(-1), gsl_mesh(), gsl_ref(), gsl_dist(), setupflag(false)
{
gsl_comm = new comm;
cr = new crystal;
#ifdef MFEM_USE_MPI
int initialized;
MPI_Initialized(&initialized);
@@ -50,20 +47,15 @@ FindPointsGSLIB::FindPointsGSLIB()
FindPointsGSLIB::~FindPointsGSLIB()
{
delete gsl_comm;
delete cr;
delete ir_simplex;
delete meshsplit;
}
#ifdef MFEM_USE_MPI
FindPointsGSLIB::FindPointsGSLIB(MPI_Comm _comm)
: mesh(NULL), meshsplit(NULL), ir_simplex(NULL),
fdata2D(NULL), fdata3D(NULL), cr(NULL), gsl_comm(NULL),
dim(-1), points_cnt(0), setupflag(false), default_interp_value(0),
avgtype(AvgType::ARITHMETIC)
: mesh(NULL), ir_simplex(NULL), fdata2D(NULL), fdata3D(NULL),
dim(-1), gsl_mesh(), gsl_ref(), gsl_dist(), setupflag(false)
{
gsl_comm = new comm;
cr = new crystal;
comm_init(gsl_comm, _comm);
}
#endif
@@ -78,7 +70,6 @@ void FindPointsGSLIB::Setup(Mesh &m, const double bb_t, const double newt_tol,
// call FreeData if FindPointsGSLIB::Setup has been called already
if (setupflag) { FreeData(); }
crystal_init(cr, gsl_comm);
mesh = &m;
dim = mesh->Dimension();
const FiniteElement *fe = mesh->GetNodalFESpace()->GetFE(0);
@@ -122,16 +113,14 @@ void FindPointsGSLIB::Setup(Mesh &m, const double bb_t, const double newt_tol,
setupflag = true;
}
void FindPointsGSLIB::FindPoints(const Vector &point_pos)
void FindPointsGSLIB::FindPoints(const Vector &point_pos,
Array<unsigned int> &codes,
Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids,
Vector &ref_pos, Vector &dist)
{
MFEM_VERIFY(setupflag, "Use FindPointsGSLIB::Setup before finding points.");
points_cnt = point_pos.Size() / dim;
gsl_code.SetSize(points_cnt);
gsl_proc.SetSize(points_cnt);
gsl_elem.SetSize(points_cnt);
gsl_ref.SetSize(points_cnt * dim);
gsl_dist.SetSize(points_cnt);
const int points_cnt = point_pos.Size() / dim;
if (dim == 2)
{
const double *xv_base[2];
@@ -140,11 +129,11 @@ void FindPointsGSLIB::FindPoints(const Vector &point_pos)
unsigned xv_stride[2];
xv_stride[0] = sizeof(double);
xv_stride[1] = sizeof(double);
findpts_2(gsl_code.GetData(), sizeof(unsigned int),
gsl_proc.GetData(), sizeof(unsigned int),
gsl_elem.GetData(), sizeof(unsigned int),
gsl_ref.GetData(), sizeof(double) * dim,
gsl_dist.GetData(), sizeof(double),
findpts_2(codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
dist.GetData(), sizeof(double),
xv_base, xv_stride, points_cnt, fdata2D);
}
else
@@ -157,27 +146,25 @@ void FindPointsGSLIB::FindPoints(const Vector &point_pos)
xv_stride[0] = sizeof(double);
xv_stride[1] = sizeof(double);
xv_stride[2] = sizeof(double);
findpts_3(gsl_code.GetData(), sizeof(unsigned int),
gsl_proc.GetData(), sizeof(unsigned int),
gsl_elem.GetData(), sizeof(unsigned int),
gsl_ref.GetData(), sizeof(double) * dim,
gsl_dist.GetData(), sizeof(double),
findpts_3(codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
dist.GetData(), sizeof(double),
xv_base, xv_stride, points_cnt, fdata3D);
}
}
// Set the element number and reference position to 0 for points not found
for (int i = 0; i < points_cnt; i++)
{
if (gsl_code[i] == 2)
{
gsl_elem[i] = 0;
for (int d = 0; d < dim; d++) { gsl_ref(i*dim + d) = -1.; }
}
}
void FindPointsGSLIB::FindPoints(const Vector &point_pos)
{
const int points_cnt = point_pos.Size() / dim;
gsl_code.SetSize(points_cnt);
gsl_proc.SetSize(points_cnt);
gsl_elem.SetSize(points_cnt);
gsl_ref.SetSize(points_cnt * dim);
gsl_dist.SetSize(points_cnt);
// Map element number for simplices, and ref_pos from [-1,1] to [0,1] for
// both simplices and quads.
MapRefPosAndElemIndices();
FindPoints(point_pos, gsl_code, gsl_proc, gsl_elem, gsl_ref, gsl_dist);
}
void FindPointsGSLIB::FindPoints(Mesh &m, const Vector &point_pos,
@@ -191,24 +178,72 @@ void FindPointsGSLIB::FindPoints(Mesh &m, const Vector &point_pos,
FindPoints(point_pos);
}
void FindPointsGSLIB::Interpolate(Array<unsigned int> &codes,
Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids,
Vector &ref_pos, const GridFunction &field_in,
Vector &field_out)
{
FiniteElementSpace ind_fes(mesh, field_in.FESpace()->FEColl());
GridFunction field_in_scalar(&ind_fes);
Vector node_vals;
const int ncomp = field_in.FESpace()->GetVDim(),
points_fld = field_in.Size() / ncomp,
points_cnt = codes.Size();
field_out.SetSize(points_cnt*ncomp);
for (int i = 0; i < ncomp; i++)
{
const int dataptrin = i*points_fld,
dataptrout = i*points_cnt;
field_in_scalar.NewDataAndSize(field_in.GetData()+dataptrin, points_fld);
GetNodeValues(field_in_scalar, node_vals);
if (dim==2)
{
findpts_eval_2(field_out.GetData()+dataptrout, sizeof(double),
codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
points_cnt, node_vals.GetData(), fdata2D);
}
else
{
findpts_eval_3(field_out.GetData()+dataptrout, sizeof(double),
codes.GetData(), sizeof(unsigned int),
proc_ids.GetData(), sizeof(unsigned int),
elem_ids.GetData(), sizeof(unsigned int),
ref_pos.GetData(), sizeof(double) * dim,
points_cnt, node_vals.GetData(), fdata3D);
}
}
}
void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
Vector &field_out)
{
Interpolate(gsl_code, gsl_proc, gsl_elem, gsl_ref, field_in, field_out);
}
void FindPointsGSLIB::Interpolate(const Vector &point_pos,
const GridFunction &field_in, Vector &field_out)
{
FindPoints(point_pos);
Interpolate(field_in, field_out);
Interpolate(gsl_code, gsl_proc, gsl_elem, gsl_ref, field_in, field_out);
}
void FindPointsGSLIB::Interpolate(Mesh &m, const Vector &point_pos,
const GridFunction &field_in, Vector &field_out)
{
FindPoints(m, point_pos);
Interpolate(field_in, field_out);
Interpolate(gsl_code, gsl_proc, gsl_elem, gsl_ref, field_in, field_out);
}
void FindPointsGSLIB::FreeData()
{
if (!setupflag) { return; }
crystal_free(cr);
if (dim == 2)
{
findpts_free_2(fdata2D);
@@ -217,13 +252,13 @@ void FindPointsGSLIB::FreeData()
{
findpts_free_3(fdata3D);
}
setupflag = false;
gsl_code.DeleteAll();
gsl_proc.DeleteAll();
gsl_elem.DeleteAll();
gsl_mesh.Destroy();
gsl_ref.Destroy();
gsl_dist.Destroy();
setupflag = false;
}
void FindPointsGSLIB::GetNodeValues(const GridFunction &gf_in,
@@ -323,8 +358,9 @@ void FindPointsGSLIB::GetSimplexNodalCoordinates()
const FiniteElement *fe = mesh->GetNodalFESpace()->GetFE(0);
const Geometry::Type gt = fe->GetGeomType();
const GridFunction *nodes = mesh->GetNodes();
Mesh *meshsplit = NULL;
const int NE = mesh->GetNE();
int NEsplit = 0;
int NEsplit = -1;
// Split the reference element into a reference submesh of quads or hexes.
if (gt == Geometry::TRIANGLE)
@@ -480,361 +516,8 @@ void FindPointsGSLIB::GetSimplexNodalCoordinates()
pt_id++;
}
}
}
void FindPointsGSLIB::MapRefPosAndElemIndices()
{
gsl_mfem_ref = gsl_ref;
gsl_mfem_elem = gsl_elem;
const FiniteElement *fe = mesh->GetNodalFESpace()->GetFE(0);
const Geometry::Type gt = fe->GetGeomType();
int NEsplit = 0;
gsl_mfem_ref -= -1.; // map [-1, 1] to
gsl_mfem_ref *= 0.5; // [0, 1]
if (gt == Geometry::SQUARE || gt == Geometry::CUBE) { return; }
H1_FECollection feclin(1, dim);
FiniteElementSpace nodal_fes_lin(meshsplit, &feclin, dim);
GridFunction gf_lin(&nodal_fes_lin);
if (gt == Geometry::TRIANGLE)
{
const double quad_v[7][2] =
{
{0, 0}, {0.5, 0}, {1, 0}, {0, 0.5},
{1./3., 1./3.}, {0.5, 0.5}, {0, 1}
};
for (int k = 0; k < dim; k++)
{
for (int j = 0; j < gf_lin.Size()/dim; j++)
{
gf_lin(j+k*gf_lin.Size()/dim) = quad_v[j][k];
}
}
NEsplit = 3;
}
else if (gt == Geometry::TETRAHEDRON)
{
const double hex_v[15][3] =
{
{0, 0, 0.}, {1, 0., 0.}, {0., 1., 0.}, {0, 0., 1.},
{0.5, 0., 0.}, {0.5, 0.5, 0.}, {0., 0.5, 0.},
{0., 0., 0.5}, {0.5, 0., 0.5}, {0., 0.5, 0.5},
{1./3., 0., 1./3.}, {1./3., 1./3., 1./3.}, {0, 1./3., 1./3.},
{1./3., 1./3., 0}, {0.25, 0.25, 0.25}
};
for (int k = 0; k < dim; k++)
{
for (int j = 0; j < gf_lin.Size()/dim; j++)
{
gf_lin(j+k*gf_lin.Size()/dim) = hex_v[j][k];
}
}
NEsplit = 4;
}
else if (gt == Geometry::PRISM)
{
const double hex_v[14][3] =
{
{0, 0, 0}, {0.5, 0, 0}, {1, 0, 0}, {0, 0.5, 0},
{1./3., 1./3., 0}, {0.5, 0.5, 0}, {0, 1, 0},
{0, 0, 1}, {0.5, 0, 1}, {1, 0, 1}, {0, 0.5, 1},
{1./3., 1./3., 1}, {0.5, 0.5, 1}, {0, 1, 1}
};
for (int k = 0; k < dim; k++)
{
for (int j = 0; j < gf_lin.Size()/dim; j++)
{
gf_lin(j+k*gf_lin.Size()/dim) = hex_v[j][k];
}
}
NEsplit = 3;
}
else
{
MFEM_ABORT("Element type not currently supported.");
}
// Simplices are split into quads/hexes for GSLIB. For MFEM, we need to find
// the original element number and map the rst from micro to macro element.
for (int i = 0; i < points_cnt; i++)
{
if (gsl_code[i] == 2) { continue; }
int local_elem = gsl_elem[i]%NEsplit;
gsl_mfem_elem[i] = (gsl_elem[i] - local_elem)/NEsplit; // macro element number
IntegrationPoint ip;
Vector mfem_ref(gsl_mfem_ref.GetData()+i*dim, dim);
ip.Set2(mfem_ref.GetData());
if (dim == 3) { ip.z = mfem_ref(2); }
gf_lin.GetVectorValue(local_elem, ip, mfem_ref); // map to rst of macro element
}
}
void FindPointsGSLIB::Interpolate(const GridFunction &field_in,
Vector &field_out)
{
const int gf_order = field_in.FESpace()->GetFE(0)->GetOrder(),
mesh_order = mesh->GetNodalFESpace()->GetFE(0)->GetOrder();
const FiniteElementCollection *fec_in = field_in.FESpace()->FEColl();
const H1_FECollection *fec_h1 = dynamic_cast<const H1_FECollection *>(fec_in);
const L2_FECollection *fec_l2 = dynamic_cast<const L2_FECollection *>(fec_in);
if (fec_h1 && gf_order == mesh_order &&
fec_h1->GetBasisType() == BasisType::GaussLobatto)
{
InterpolateH1(field_in, field_out);
return;
}
else
{
InterpolateGeneral(field_in, field_out);
if (!fec_l2 || avgtype == AvgType::NONE) { return; }
}
// For points on element borders, project the L2 GridFunction to H1 and
// re-interpolate.
if (fec_l2)
{
Array<int> indl2;
for (int i = 0; i < points_cnt; i++)
{
if (gsl_code[i] == 1) { indl2.Append(i); }
}
if (indl2.Size() == 0) { return; } // no points on element borders
Vector field_out_l2(field_out.Size());
VectorGridFunctionCoefficient field_in_dg(&field_in);
int gf_order_h1 = std::max(gf_order, 1); // H1 should be at least order 1
H1_FECollection fec(gf_order_h1, dim);
const int ncomp = field_in.FESpace()->GetVDim();
FiniteElementSpace fes(mesh, &fec, ncomp);
GridFunction field_in_h1(&fes);
if (avgtype == AvgType::ARITHMETIC)
{
field_in_h1.ProjectDiscCoefficient(field_in_dg, GridFunction::ARITHMETIC);
}
else if (avgtype == AvgType::HARMONIC)
{
field_in_h1.ProjectDiscCoefficient(field_in_dg, GridFunction::HARMONIC);
}
else
{
MFEM_ABORT("Invalid averaging type.");
}
if (gf_order_h1 == mesh_order) // basis is GaussLobatto by default
{
InterpolateH1(field_in_h1, field_out_l2);
}
else
{
InterpolateGeneral(field_in_h1, field_out_l2);
}
// Copy interpolated values for the points on element border
for (int j = 0; j < ncomp; j++)
{
for (int i = 0; i < indl2.Size(); i++)
{
int idx = indl2[i] + j*points_cnt;
field_out(idx) = field_out_l2(idx);
}
}
}
}
void FindPointsGSLIB::InterpolateH1(const GridFunction &field_in,
Vector &field_out)
{
FiniteElementSpace ind_fes(mesh, field_in.FESpace()->FEColl());
GridFunction field_in_scalar(&ind_fes);
Vector node_vals;
const int ncomp = field_in.FESpace()->GetVDim(),
points_fld = field_in.Size() / ncomp,
points_cnt = gsl_code.Size();
field_out.SetSize(points_cnt*ncomp);
field_out = default_interp_value;
for (int i = 0; i < ncomp; i++)
{
const int dataptrin = i*points_fld,
dataptrout = i*points_cnt;
field_in_scalar.NewDataAndSize(field_in.GetData()+dataptrin, points_fld);
GetNodeValues(field_in_scalar, node_vals);
if (dim==2)
{
findpts_eval_2(field_out.GetData()+dataptrout, sizeof(double),
gsl_code.GetData(), sizeof(unsigned int),
gsl_proc.GetData(), sizeof(unsigned int),
gsl_elem.GetData(), sizeof(unsigned int),
gsl_ref.GetData(), sizeof(double) * dim,
points_cnt, node_vals.GetData(), fdata2D);
}
else
{
findpts_eval_3(field_out.GetData()+dataptrout, sizeof(double),
gsl_code.GetData(), sizeof(unsigned int),
gsl_proc.GetData(), sizeof(unsigned int),
gsl_elem.GetData(), sizeof(unsigned int),
gsl_ref.GetData(), sizeof(double) * dim,
points_cnt, node_vals.GetData(), fdata3D);
}
}
}
void FindPointsGSLIB::InterpolateGeneral(const GridFunction &field_in,
Vector &field_out)
{
int ncomp = field_in.VectorDim(),
nptorig = points_cnt,
npt = points_cnt;
field_out.SetSize(points_cnt*ncomp);
field_out = default_interp_value;
if (gsl_comm->np == 1) // serial
{
for (int index = 0; index < npt; index++)
{
if (gsl_code[index] == 2) { continue; }
IntegrationPoint ip;
ip.Set2(gsl_mfem_ref.GetData()+index*dim);
if (dim == 3) { ip.z = gsl_mfem_ref(index*dim + 2); }
Vector localval(ncomp);
field_in.GetVectorValue(gsl_mfem_elem[index], ip, localval);
for (int i = 0; i < ncomp; i++)
{
field_out(index + i*npt) = localval(i);
}
}
}
else // parallel
{
// Determine number of points to be sent
int nptsend = 0;
for (int index = 0; index < npt; index++)
{
if (gsl_code[index] != 2) { nptsend +=1; }
}
// Pack data to send via crystal router
struct array *outpt = new array;
struct out_pt { double r[3], ival; uint index, el, proc; };
struct out_pt *pt;
array_init(struct out_pt, outpt, nptsend);
outpt->n=nptsend;
pt = (struct out_pt *)outpt->ptr;
for (int index = 0; index < npt; index++)
{
if (gsl_code[index] == 2) { continue; }
for (int d = 0; d < dim; ++d) { pt->r[d]= gsl_mfem_ref(index*dim + d); }
pt->index = index;
pt->proc = gsl_proc[index];
pt->el = gsl_mfem_elem[index];
++pt;
}
// Transfer data to target MPI ranks
sarray_transfer(struct out_pt, outpt, proc, 1, cr);
if (ncomp == 1)
{
// Interpolate the grid function
npt = outpt->n;
pt = (struct out_pt *)outpt->ptr;
for (int index = 0; index < npt; index++)
{
IntegrationPoint ip;
ip.Set3(&pt->r[0]);
pt->ival = field_in.GetValue(pt->el, ip, 1);
++pt;
}
// Transfer data back to source MPI rank
sarray_transfer(struct out_pt, outpt, proc, 1, cr);
npt = outpt->n;
pt = (struct out_pt *)outpt->ptr;
for (int index = 0; index < npt; index++)
{
field_out(pt->index) = pt->ival;
++pt;
}
array_free(outpt);
delete outpt;
}
else // ncomp > 1
{
// Interpolate data and store in a Vector
npt = outpt->n;
pt = (struct out_pt *)outpt->ptr;
Vector vec_int_vals(npt*ncomp);
for (int index = 0; index < npt; index++)
{
IntegrationPoint ip;
ip.Set3(&pt->r[0]);
Vector localval(vec_int_vals.GetData()+index*ncomp, ncomp);
field_in.GetVectorValue(pt->el, ip, localval);
++pt;
}
// Save index and proc data in a struct
struct array *savpt = new array;
struct sav_pt { uint index, proc; };
struct sav_pt *spt;
array_init(struct sav_pt, savpt, npt);
savpt->n=npt;
spt = (struct sav_pt *)savpt->ptr;
pt = (struct out_pt *)outpt->ptr;
for (int index = 0; index < npt; index++)
{
spt->index = pt->index;
spt->proc = pt->proc;
++pt; ++spt;
}
array_free(outpt);
delete outpt;
// Copy data from save struct to send struct and send component wise
struct array *sendpt = new array;
struct send_pt { double ival; uint index, proc; };
struct send_pt *sdpt;
for (int j = 0; j < ncomp; j++)
{
array_init(struct send_pt, sendpt, npt);
sendpt->n=npt;
spt = (struct sav_pt *)savpt->ptr;
sdpt = (struct send_pt *)sendpt->ptr;
for (int index = 0; index < npt; index++)
{
sdpt->index = spt->index;
sdpt->proc = spt->proc;
sdpt->ival = vec_int_vals(j + index*ncomp);
++sdpt; ++spt;
}
sarray_transfer(struct send_pt, sendpt, proc, 1, cr);
sdpt = (struct send_pt *)sendpt->ptr;
for (int index = 0; index < nptorig; index++)
{
int idx = sdpt->index + j*nptorig;
field_out(idx) = sdpt->ival;
++sdpt;
}
array_free(sendpt);
}
array_free(savpt);
delete sendpt;
delete savpt;
} // ncomp > 1
} // parallel
delete meshsplit;
}
} // namespace mfem
+45 -93
View File
@@ -20,66 +20,28 @@
struct comm;
struct findpts_data_2;
struct findpts_data_3;
struct array;
struct crystal;
namespace mfem
{
/** \brief FindPointsGSLIB can robustly evaluate a GridFunction on an arbitrary
* collection of points. There are three key functions in FindPointsGSLIB:
*
* 1. Setup - constructs the internal data structures of gslib.
*
* 2. FindPoints - for any given arbitrary set of points in physical space,
* gslib finds the element number, MPI rank, and the reference space
* coordinates inside the element that each point is located in. gslib also
* returns a code that indicates whether the point was found inside an
* element, on element border, or not found in the domain.
*
* 3. Interpolate - Interpolates any grid function at the points found using 2.
*
* FindPointsGSLIB provides interface to use these functions individually or
* using a single call.
*/
class FindPointsGSLIB
{
public:
enum AvgType {NONE, ARITHMETIC, HARMONIC}; // Average type for L2 functions
protected:
Mesh *mesh, *meshsplit;
IntegrationRule *ir_simplex; // IntegrationRule to split quads/hex -> simplex
struct findpts_data_2 *fdata2D; // gslib's internal data
struct findpts_data_3 *fdata3D; // gslib's internal data
struct crystal *cr; // gslib's internal data
struct comm *gsl_comm; // gslib's internal data
int dim, points_cnt;
Array<unsigned int> gsl_code, gsl_proc, gsl_elem, gsl_mfem_elem;
Vector gsl_mesh, gsl_ref, gsl_dist, gsl_mfem_ref;
bool setupflag; // flag to indicate whether gslib data has been setup
double default_interp_value; // used for points that are not found in the mesh
AvgType avgtype; // average type used for L2 functions
Mesh *mesh;
IntegrationRule *ir_simplex;
struct findpts_data_2 *fdata2D;
struct findpts_data_3 *fdata3D;
int dim;
Array<unsigned int> gsl_code, gsl_proc, gsl_elem;
Vector gsl_mesh, gsl_ref, gsl_dist;
bool setupflag;
struct comm *gsl_comm;
/// Get GridFunction from MFEM format to GSLIB format
void GetNodeValues(const GridFunction &gf_in, Vector &node_vals);
/// Get nodal coordinates from mesh to the format expected by GSLIB for quads
/// and hexes
void GetQuadHexNodalCoordinates();
/// Convert simplices to quad/hexes and then get nodal coordinates for each
/// split element into format expected by GSLIB
void GetSimplexNodalCoordinates();
/// Use GSLIB for communication and interpolation
void InterpolateH1(const GridFunction &field_in, Vector &field_out);
/// Uses GSLIB Crystal Router for communication followed by MFEM's
/// interpolation functions
void InterpolateGeneral(const GridFunction &field_in, Vector &field_out);
/// Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For simplices mesh
/// find the original element number (that was split into micro quads/hexes
/// by GetSimplexNodalCoordinates())
void MapRefPosAndElemIndices();
public:
FindPointsGSLIB();
@@ -102,37 +64,45 @@ public:
void Setup(Mesh &m, const double bb_t = 0.1, const double newt_tol = 1.0e-12,
const int npt_max = 256);
/** Searches positions given in physical space by @a point_pos. These positions
must by ordered by nodes: (XXX...,YYY...,ZZZ).
This function populates the following member variables:
#gsl_code Return codes for each point: inside element (0),
element boundary (1), not found (2).
#gsl_proc MPI proc ids where the points were found.
#gsl_elem Element ids where the points were found.
Defaults to 0 for points that were not found.
#gsl_mfem_elem Element ids corresponding to MFEM-mesh where the points
were found. #gsl_mfem_elem != #gsl_elem for simplices
Defaults to 0 for points that were not found.
#gsl_ref Reference coordinates of the found point.
Ordered by vdim (XYZ,XYZ,XYZ...). Defaults to -1 for
points that were not found. Note: the gslib reference
frame is [-1,1].
#gsl_mfem_ref Reference coordinates #gsl_ref mapped to [0,1].
Defaults to 0 for points that were not found.
#gsl_dist Distance between the sought and the found point
in physical space. */
/** Searches positions given in physical space by @a point_pos. All output
Arrays and Vectors are expected to have the correct size.
@param[in] point_pos Positions to be found. Must by ordered by nodes
(XXX...,YYY...,ZZZ).
@param[out] codes Return codes for each point: inside element (0),
element boundary (1), not found (2).
@param[out] proc_ids MPI proc ids where the points were found.
@param[out] elem_ids Element ids where the points were found.
@param[out] ref_pos Reference coordinates of the found point. Ordered
by vdim (XYZ,XYZ,XYZ...).
Note: the gslib reference frame is [-1,1].
@param[out] dist Distance between the sought and the found point
in physical space. */
void FindPoints(const Vector &point_pos, Array<unsigned int> &codes,
Array<unsigned int> &proc_ids, Array<unsigned int> &elem_ids,
Vector &ref_pos, Vector &dist);
void FindPoints(const Vector &point_pos);
/// Setup FindPoints and search positions
void FindPoints(Mesh &m, const Vector &point_pos, const double bb_t = 0.1,
const double newt_tol = 1.0e-12, const int npt_max = 256);
/** Interpolation of field values at prescribed reference space positions.
@param[in] codes Return codes for each point: inside element (0),
element boundary (1), not found (2).
@param[in] proc_ids MPI proc ids where the points were found.
@param[in] elem_ids Element ids where the points were found.
@param[in] ref_pos Reference coordinates of the found point. Ordered
by vdim (XYZ,XYZ,XYZ...).
Note: the gslib reference frame is [-1,1].
@param[in] field_in Function values that will be interpolated on the
reference positions. Note: it is assumed that
@a field_in is in H1 and in the same space as the
mesh that was given to Setup().
@param[out] field_out Interpolated values. For points that are not found
the value is set to #default_interp_value. */
@param[out] field_out Interpolated values. */
void Interpolate(Array<unsigned int> &codes, Array<unsigned int> &proc_ids,
Array<unsigned int> &elem_ids, Vector &ref_pos,
const GridFunction &field_in, Vector &field_out);
void Interpolate(const GridFunction &field_in, Vector &field_out);
/** Search positions and interpolate */
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
@@ -141,45 +111,27 @@ public:
void Interpolate(Mesh &m, const Vector &point_pos,
const GridFunction &field_in, Vector &field_out);
/// Average type to be used for L2 functions in-case a point is located at
/// an element boundary where the function might be multi-valued.
void SetL2AvgType(AvgType avgtype_) { avgtype = avgtype_; }
/// Set the default interpolation value for points that are not found in the
/// mesh.
void SetDefaultInterpolationValue(double interp_value_)
{
default_interp_value = interp_value_;
}
/** Cleans up memory allocated internally by gslib.
Note that in parallel, this must be called before MPI_Finalize(), as it
calls MPI_Comm_free() for internal gslib communicators. */
Note that in parallel, this must be called before MPI_Finalize(), as
it calls MPI_Comm_free() for internal gslib communicators. */
void FreeData();
/// Return code for each point searched by FindPoints: inside element (0), on
/// element boundary (1), or not found (2).
const Array<unsigned int> &GetCode() const { return gsl_code; }
/// Return element number for each point found by FindPoints.
const Array<unsigned int> &GetElem() const { return gsl_mfem_elem; }
const Array<unsigned int> &GetElem() const { return gsl_elem; }
/// Return MPI rank on which each point was found by FindPoints.
const Array<unsigned int> &GetProc() const { return gsl_proc; }
/// Return reference coordinates for each point found by FindPoints.
const Vector &GetReferencePosition() const { return gsl_mfem_ref; }
const Vector &GetReferencePosition() const { return gsl_ref; }
/// Return distance Distance between the sought and the found point
/// in physical space, for each point found by FindPoints.
const Vector &GetDist() const { return gsl_dist; }
/// Return element number for each point found by FindPoints corresponding to
/// GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with simplices.
const Array<unsigned int> &GetGSLIBElem() const { return gsl_elem; }
/// Return reference coordinates in [-1,1] (internal range in GSLIB) for each
/// point found by FindPoints.
const Vector &GetGSLIBReferencePosition() const { return gsl_ref; }
};
} // namespace mfem
#endif // MFEM_USE_GSLIB
#endif //MFEM_USE_GSLIB
#endif // MFEM_GSLIB
#endif //MFEM_GSLIB guard
+25 -202
View File
@@ -35,9 +35,6 @@ extern Ceed ceed;
std::string ceed_path;
extern CeedBasisMap ceed_basis_map;
extern CeedRestrMap ceed_restr_map;
}
void InitCeedCoeff(Coefficient* Q, CeedData* ptr)
@@ -84,9 +81,10 @@ static CeedElemTopology GetCeedTopology(Geometry::Type geom)
}
}
static void InitCeedNonTensorBasis(const FiniteElementSpace &fes,
const IntegrationRule &ir,
Ceed ceed, CeedBasis *basis)
static void InitCeedNonTensorBasisAndRestriction(const FiniteElementSpace &fes,
const IntegrationRule &ir,
Ceed ceed, CeedBasis *basis,
CeedElemRestriction *restr)
{
Mesh *mesh = fes.GetMesh();
const FiniteElement *fe = fes.GetFE(0);
@@ -99,73 +97,7 @@ static void InitCeedNonTensorBasis(const FiniteElementSpace &fes,
Vector qweight(Q);
Vector shape_i(P);
DenseMatrix grad_i(P, dim);
const Table &el_dof = fes.GetElementToDofTable();
Array<int> tp_el_dof(el_dof.Size_of_connections());
const TensorBasisElement * tfe =
dynamic_cast<const TensorBasisElement *>(fe);
if (tfe) // Lexicographic ordering using dof_map
{
const Array<int>& dof_map = tfe->GetDofMap();
for (int i = 0; i < Q; i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
qref(0,i) = ip.x;
if (dim>1) { qref(1,i) = ip.y; }
if (dim>2) { qref(2,i) = ip.z; }
qweight(i) = ip.weight;
fe->CalcShape(ip, shape_i);
fe->CalcDShape(ip, grad_i);
for (int j = 0; j < P; j++)
{
shape(j, i) = shape_i(dof_map[j]);
for (int d = 0; d < dim; ++d)
{
grad(j+i*P+d*Q*P) = grad_i(dof_map[j], d);
}
}
}
}
else // Native ordering
{
for (int i = 0; i < Q; i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
qref(0,i) = ip.x;
if (dim>1) { qref(1,i) = ip.y; }
if (dim>2) { qref(2,i) = ip.z; }
qweight(i) = ip.weight;
fe->CalcShape(ip, shape_i);
fe->CalcDShape(ip, grad_i);
for (int j = 0; j < P; j++)
{
shape(j, i) = shape_i(j);
for (int d = 0; d < dim; ++d)
{
grad(j+i*P+d*Q*P) = grad_i(j, d);
}
}
}
}
CeedBasisCreateH1(ceed, GetCeedTopology(fe->GetGeomType()), fes.GetVDim(),
fe->GetDof(), ir.GetNPoints(), shape.GetData(),
grad.GetData(), qref.GetData(), qweight.GetData(), basis);
}
static void InitCeedNonTensorRestriction(const FiniteElementSpace &fes,
const IntegrationRule &ir,
Ceed ceed, CeedElemRestriction *restr)
{
Mesh *mesh = fes.GetMesh();
const FiniteElement *fe = fes.GetFE(0);
const int dim = mesh->Dimension();
const int P = fe->GetDof();
const int Q = ir.GetNPoints();
DenseMatrix shape(P, Q);
Vector grad(P*dim*Q);
DenseMatrix qref(dim, Q);
Vector qweight(Q);
Vector shape_i(P);
DenseMatrix grad_i(P, dim);
CeedInt compstride = fes.GetOrdering()==Ordering::byVDIM ? 1 : fes.GetNDofs();
const Table &el_dof = fes.GetElementToDofTable();
Array<int> tp_el_dof(el_dof.Size_of_connections());
@@ -192,6 +124,7 @@ static void InitCeedNonTensorRestriction(const FiniteElementSpace &fes,
}
}
}
for (int i = 0; i < mesh->GetNE(); i++)
{
const int el_offset = fe->GetDof() * i;
@@ -229,6 +162,7 @@ static void InitCeedNonTensorRestriction(const FiniteElementSpace &fes,
}
}
}
for (int e = 0; e < mesh->GetNE(); e++)
{
for (int i = 0; i < P; i++)
@@ -244,15 +178,19 @@ static void InitCeedNonTensorRestriction(const FiniteElementSpace &fes,
}
}
}
CeedBasisCreateH1(ceed, GetCeedTopology(fe->GetGeomType()), fes.GetVDim(),
fe->GetDof(), ir.GetNPoints(), shape.GetData(),
grad.GetData(), qref.GetData(), qweight.GetData(), basis);
CeedElemRestrictionCreate(ceed, mesh->GetNE(), fe->GetDof(), fes.GetVDim(),
compstride, (fes.GetVDim())*(fes.GetNDofs()),
CEED_MEM_HOST, CEED_COPY_VALUES,
tp_el_dof.GetData(), restr);
}
static void InitCeedTensorBasis(const FiniteElementSpace &fes,
const IntegrationRule &ir,
Ceed ceed, CeedBasis *basis)
static void InitCeedTensorBasisAndRestriction(const FiniteElementSpace &fes,
const IntegrationRule &ir,
Ceed ceed, CeedBasis *basis,
CeedElemRestriction *restr)
{
Mesh *mesh = fes.GetMesh();
const FiniteElement *fe = fes.GetFE(0);
@@ -260,6 +198,7 @@ static void InitCeedTensorBasis(const FiniteElementSpace &fes,
const TensorBasisElement * tfe =
dynamic_cast<const TensorBasisElement *>(fe);
MFEM_VERIFY(tfe, "invalid FE");
const Array<int>& dof_map = tfe->GetDofMap();
const FiniteElement *fe1d =
fes.FEColl()->FiniteElementForGeometry(Geometry::SEGMENT);
DenseMatrix shape1d(fe1d->GetDof(), ir.GetNPoints());
@@ -288,28 +227,6 @@ static void InitCeedTensorBasis(const FiniteElementSpace &fes,
ir.GetNPoints(), shape1d.GetData(),
grad1d.GetData(), qref1d.GetData(),
qweight1d.GetData(), basis);
}
static void InitCeedTensorRestriction(const FiniteElementSpace &fes,
const IntegrationRule &ir,
Ceed ceed, CeedElemRestriction *restr)
{
Mesh *mesh = fes.GetMesh();
const FiniteElement *fe = fes.GetFE(0);
const TensorBasisElement * tfe =
dynamic_cast<const TensorBasisElement *>(fe);
MFEM_VERIFY(tfe, "invalid FE");
const Array<int>& dof_map = tfe->GetDofMap();
const FiniteElement *fe1d =
fes.FEColl()->FiniteElementForGeometry(Geometry::SEGMENT);
DenseMatrix shape1d(fe1d->GetDof(), ir.GetNPoints());
DenseMatrix grad1d(fe1d->GetDof(), ir.GetNPoints());
Vector qref1d(ir.GetNPoints()), qweight1d(ir.GetNPoints());
Vector shape_i(shape1d.Height());
DenseMatrix grad_i(grad1d.Height(), 1);
const H1_SegmentElement *h1_fe1d =
dynamic_cast<const H1_SegmentElement *>(fe1d);
MFEM_VERIFY(h1_fe1d, "invalid FE");
CeedInt compstride = fes.GetOrdering()==Ordering::byVDIM ? 1 : fes.GetNDofs();
const Table &el_dof = fes.GetElementToDofTable();
@@ -341,52 +258,14 @@ void InitCeedBasisAndRestriction(const FiniteElementSpace &fes,
Ceed ceed, CeedBasis *basis,
CeedElemRestriction *restr)
{
// Check for FES -> basis, restriction in hash tables
const Mesh *mesh = fes.GetMesh();
const FiniteElement *fe = fes.GetFE(0);
const int P = fe->GetDof();
const int Q = irm.GetNPoints();
const int nelem = mesh->GetNE();
const int ncomp = fes.GetVDim();
CeedBasisKey basis_key(&fes, &irm, ncomp, P, Q);
auto basis_itr = internal::ceed_basis_map.find(basis_key);
CeedRestrKey restr_key(&fes, nelem, P, ncomp);
auto restr_itr = internal::ceed_restr_map.find(restr_key);
// Init or retreive key values
if (basis_itr == internal::ceed_basis_map.end())
if (UsesTensorBasis(fes))
{
if (UsesTensorBasis(fes))
{
const IntegrationRule &ir = IntRules.Get(Geometry::SEGMENT, irm.GetOrder());
InitCeedTensorBasis(fes, ir, ceed, basis);
}
else
{
InitCeedNonTensorBasis(fes, irm, ceed, basis);
}
internal::ceed_basis_map[basis_key] = *basis;
const IntegrationRule &ir = IntRules.Get(Geometry::SEGMENT, irm.GetOrder());
InitCeedTensorBasisAndRestriction(fes, ir, ceed, basis, restr);
}
else
{
*basis = basis_itr->second;
}
if (restr_itr == internal::ceed_restr_map.end())
{
if (UsesTensorBasis(fes))
{
const IntegrationRule &ir = IntRules.Get(Geometry::SEGMENT, irm.GetOrder());
InitCeedTensorRestriction(fes, ir, ceed, restr);
}
else
{
InitCeedNonTensorRestriction(fes, irm, ceed, restr);
}
internal::ceed_restr_map[restr_key] = *restr;
}
else
{
*restr = restr_itr->second;
InitCeedNonTensorBasisAndRestriction(fes, irm, ceed, basis, restr);
}
}
@@ -448,8 +327,8 @@ void CeedPAAssemble(const CeedPAOperator& op,
CeedVectorCreate(ceed, nelem * nqpts * qdatasize, &ceedData.rho);
// Context data to be passed to the 'f_build_diff' Q-function.
ceedData.build_ctx_data.dim = mesh->Dimension();
ceedData.build_ctx_data.space_dim = mesh->SpaceDimension();
ceedData.build_ctx.dim = mesh->Dimension();
ceedData.build_ctx.space_dim = mesh->SpaceDimension();
std::string qf_file = GetCeedPath() + op.header;
std::string qf;
@@ -463,7 +342,7 @@ void CeedPAAssemble(const CeedPAOperator& op,
CeedQFunctionCreateInterior(ceed, 1, op.const_qf,
qf.c_str(),
&ceedData.build_qfunc);
ceedData.build_ctx_data.coeff = ((CeedConstCoeff*)ceedData.coeff)->val;
ceedData.build_ctx.coeff = ((CeedConstCoeff*)ceedData.coeff)->val;
break;
case CeedCoeff::Grid:
qf = qf_file + op.grid_func;
@@ -479,12 +358,8 @@ void CeedPAAssemble(const CeedPAOperator& op,
CeedQFunctionAddInput(ceedData.build_qfunc, "weights", 1, CEED_EVAL_WEIGHT);
CeedQFunctionAddOutput(ceedData.build_qfunc, "qdata", qdatasize,
CEED_EVAL_NONE);
CeedQFunctionContextCreate(ceed, &ceedData.build_ctx);
CeedQFunctionContextSetData(ceedData.build_ctx, CEED_MEM_HOST, CEED_USE_POINTER,
sizeof(ceedData.build_ctx_data),
&ceedData.build_ctx_data);
CeedQFunctionSetContext(ceedData.build_qfunc, ceedData.build_ctx);
CeedQFunctionSetContext(ceedData.build_qfunc, &ceedData.build_ctx,
sizeof(ceedData.build_ctx));
// Create the operator that builds the quadrature data for the operator.
CeedOperatorCreate(ceed, ceedData.build_qfunc, NULL, NULL,
@@ -524,7 +399,8 @@ void CeedPAAssemble(const CeedPAOperator& op,
CeedQFunctionAddInput(ceedData.apply_qfunc, "qdata", qdatasize,
CEED_EVAL_NONE);
CeedQFunctionAddOutput(ceedData.apply_qfunc, "v", dimV, op.test_op);
CeedQFunctionSetContext(ceedData.apply_qfunc, ceedData.build_ctx);
CeedQFunctionSetContext(ceedData.apply_qfunc, &ceedData.build_ctx,
sizeof(ceedData.build_ctx));
// Create the diff operator.
CeedOperatorCreate(ceed, ceedData.apply_qfunc, NULL, NULL, &ceedData.oper);
@@ -539,59 +415,6 @@ void CeedPAAssemble(const CeedPAOperator& op,
CeedVectorCreate(ceed, fes.GetNDofs(), &ceedData.v);
}
void CeedAddMultPA(const CeedData *ceedDataPtr,
const Vector &x,
Vector &y)
{
const CeedScalar *x_ptr;
CeedScalar *y_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
x_ptr = x.Read();
y_ptr = y.ReadWrite();
}
else
{
x_ptr = x.HostRead();
y_ptr = y.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->u, mem, CEED_USE_POINTER,
const_cast<CeedScalar*>(x_ptr));
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, y_ptr);
CeedOperatorApplyAdd(ceedDataPtr->oper, ceedDataPtr->u, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorTakeArray(ceedDataPtr->u, mem, const_cast<CeedScalar**>(&x_ptr));
CeedVectorTakeArray(ceedDataPtr->v, mem, &y_ptr);
}
void CeedAssembleDiagonalPA(const CeedData *ceedDataPtr,
Vector &diag)
{
CeedScalar *d_ptr;
CeedMemType mem;
CeedGetPreferredMemType(internal::ceed, &mem);
if ( Device::Allows(Backend::CUDA) && mem==CEED_MEM_DEVICE )
{
d_ptr = diag.ReadWrite();
}
else
{
d_ptr = diag.HostReadWrite();
mem = CEED_MEM_HOST;
}
CeedVectorSetArray(ceedDataPtr->v, mem, CEED_USE_POINTER, d_ptr);
CeedOperatorLinearAssembleAddDiagonal(ceedDataPtr->oper, ceedDataPtr->v,
CEED_REQUEST_IMMEDIATE);
CeedVectorTakeArray(ceedDataPtr->v, mem, &d_ptr);
}
} // namespace mfem
#endif // MFEM_USE_CEED
+8 -56
View File
@@ -16,11 +16,7 @@
#ifdef MFEM_USE_CEED
#include "../../general/device.hpp"
#include "../../linalg/vector.hpp"
#include <ceed.h>
#include <ceed-hash.h>
#include <tuple>
#include <unordered_map>
namespace mfem
{
@@ -30,47 +26,7 @@ class GridFunction;
class IntegrationRule;
class Coefficient;
// Hash table for CeedBasis
using CeedBasisKey =
std::tuple<const FiniteElementSpace*, const IntegrationRule*, int, int, int>;
struct CeedBasisHash
{
std::size_t operator()(const CeedBasisKey& k) const
{
return CeedHashCombine(CeedHashCombine(CeedHashInt(
reinterpret_cast<CeedHash64_t>(std::get<0>(k))),
CeedHashInt(
reinterpret_cast<CeedHash64_t>(std::get<1>(k)))),
CeedHashCombine(CeedHashCombine(CeedHashInt(std::get<2>(k)),
CeedHashInt(std::get<3>(k))),
CeedHashInt(std::get<4>(k))));
}
};
using CeedBasisMap =
std::unordered_map<const CeedBasisKey, CeedBasis, CeedBasisHash>;
// Hash table for CeedElemRestriction
using CeedRestrKey = std::tuple<const FiniteElementSpace*, int, int, int>;
struct CeedRestrHash
{
std::size_t operator()(const CeedRestrKey& k) const
{
return CeedHashCombine(CeedHashCombine(CeedHashInt(
reinterpret_cast<CeedHash64_t>(std::get<0>(k))),
CeedHashInt(std::get<1>(k))),
CeedHashCombine(CeedHashInt(std::get<2>(k)),
CeedHashInt(std::get<3>(k))));
}
};
using CeedRestrMap =
std::unordered_map<const CeedRestrKey, CeedElemRestriction, CeedRestrHash>;
namespace internal
{
extern Ceed ceed; // defined in device.cpp
extern CeedBasisMap basis_map;
extern CeedRestrMap restr_map;
}
namespace internal { extern Ceed ceed; } // defined in device.cpp
/// A structure used to pass additional data to f_build_diff and f_apply_diff
struct BuildContext { CeedInt dim, space_dim; CeedScalar coeff; };
@@ -99,8 +55,7 @@ struct CeedData
CeedVector node_coords, rho;
CeedCoeff coeff_type;
void* coeff;
CeedQFunctionContext build_ctx;
BuildContext build_ctx_data;
BuildContext build_ctx;
CeedVector u, v;
@@ -108,6 +63,10 @@ struct CeedData
{
CeedOperatorDestroy(&build_oper);
CeedOperatorDestroy(&oper);
CeedBasisDestroy(&basis);
CeedBasisDestroy(&mesh_basis);
CeedElemRestrictionDestroy(&restr);
CeedElemRestrictionDestroy(&mesh_restr);
CeedElemRestrictionDestroy(&restr_i);
CeedElemRestrictionDestroy(&mesh_restr_i);
CeedQFunctionDestroy(&apply_qfunc);
@@ -117,6 +76,8 @@ struct CeedData
if (coeff_type==CeedCoeff::Grid)
{
CeedGridCoeff* c = (CeedGridCoeff*)coeff;
CeedBasisDestroy(&c->basis);
CeedElemRestrictionDestroy(&c->restr);
CeedVectorDestroy(&c->coeffVector);
delete c;
}
@@ -183,15 +144,6 @@ const std::string &GetCeedPath();
void CeedPAAssemble(const CeedPAOperator& op,
CeedData& ceedData);
/** @brief Function that applies a libCEED PA operator. */
void CeedAddMultPA(const CeedData *ceedDataPtr,
const Vector &x,
Vector &y);
/** @brief Function that assembles a libCEED PA operator diagonal. */
void CeedAssembleDiagonalPA(const CeedData *ceedDataPtr,
Vector &diag);
/** @brief Function that determines if a CEED kernel should be used, based on
the current mfem::Device configuration. */
inline bool DeviceCanUseCeed()
+4 -5
View File
@@ -178,11 +178,10 @@ public:
/** @brief Make the LinearForm reference external data on a new
FiniteElementSpace. */
/** This method changes the FiniteElementSpace associated with the LinearForm
@a *f and sets the data of the Vector @a v (plus the @a v_offset) as
external data in the LinearForm.
@note This version of the method will also perform bounds checks when the
build option MFEM_DEBUG is enabled. */
@a *f and sets the data of the Vector @a v (plus the @a v_offset)
as external data in the LinearForm.
@note This version of the method will also perform bounds checks when
the build option MFEM_DEBUG is enabled. */
virtual void MakeRef(FiniteElementSpace *f, Vector &v, int v_offset);
/// Return the action of the LinearForm as a linear mapping.
+39 -6
View File
@@ -457,8 +457,20 @@ void VectorFEDomainLFCurlIntegrator::AssembleRHSElementVect(
Tr.SetIntPoint (&ip);
el.CalcPhysCurlShape(Tr, curlshape);
QF->Eval(vec, Tr, ip);
switch (spaceDim)
{
case 3:
MFEM_VERIFY(QF, "VectorFunctionCoefficient not provided");
QF->Eval(vec, Tr, ip);
break;
case 2:
MFEM_VERIFY(Q, "FunctionCoefficient (Scalar) not provided");
vec[0] = Q->Eval(Tr, ip);
break;
default:
break; // This should be unreachable
}
vec *= ip.weight * Tr.Weight();
curlshape.AddMult (vec, elvect);
}
@@ -468,17 +480,38 @@ void VectorFEDomainLFCurlIntegrator::AssembleDeltaElementVect(
const FiniteElement &fe, ElementTransformation &Trans, Vector &elvect)
{
int spaceDim = Trans.GetSpaceDim();
MFEM_ASSERT(vec_delta != NULL,
"coefficient must be VectorDeltaCoefficient");
switch (spaceDim)
{
case 3:
MFEM_ASSERT(vec_delta != NULL,
"coefficient must be VectorDeltaCoefficient");
break;
case 2:
MFEM_ASSERT(delta != NULL,
"coefficient must be DeltaCoefficient");
break;
default:
break; // This should be unreachable
}
int dof = fe.GetDof();
int n=(spaceDim == 3)? spaceDim : 1;
vec.SetSize(n);
curlshape.SetSize(dof, n);
elvect.SetSize(dof);
fe.CalcPhysCurlShape(Trans, curlshape);
vec_delta->EvalDelta(vec, Trans, Trans.GetIntPoint());
curlshape.Mult(vec, elvect);
switch (spaceDim)
{
case 3:
vec_delta->EvalDelta(vec, Trans, Trans.GetIntPoint());
curlshape.Mult(vec, elvect);
break;
case 2:
curlshape.GetColumn(0,elvect);
elvect *= delta->EvalDelta(Trans, Trans.GetIntPoint());
break;
default:
break; // This should be unreachable
}
}
void VectorFEDomainLFDivIntegrator::AssembleRHSElementVect(
+3
View File
@@ -284,6 +284,7 @@ class VectorFEDomainLFCurlIntegrator : public DeltaLFIntegrator
{
private:
VectorCoefficient *QF=nullptr;
Coefficient *Q=nullptr;
DenseMatrix curlshape;
Vector vec;
@@ -291,6 +292,8 @@ public:
/// Constructs the domain integrator (Q, curl v)
VectorFEDomainLFCurlIntegrator(VectorCoefficient &F)
: DeltaLFIntegrator(F), QF(&F) { }
VectorFEDomainLFCurlIntegrator(Coefficient &F)
: DeltaLFIntegrator(F), Q(&F) { }
virtual void AssembleRHSElementVect(const FiniteElement &el,
ElementTransformation &Tr,
+2 -162
View File
@@ -655,167 +655,6 @@ void ParGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient &vcoeff,
#endif
}
double ParGridFunction::ComputeDGFaceJumpError(Coefficient *exsol,
Coefficient *ell_coeff,
double Nu,
const IntegrationRule *irs[]) const
{
const_cast<ParGridFunction *>(this)->ExchangeFaceNbrData();
int fdof, dim, intorder, k;
ElementTransformation *transf;
Vector shape, el_dofs, err_val, ell_coeff_val;
Array<int> vdofs;
IntegrationPoint eip;
double error = 0.0;
ParMesh *mesh = pfes->GetParMesh();
dim = mesh->Dimension();
std::map<int,int> local_to_shared;
for (int i = 0; i < mesh->GetNSharedFaces(); ++i)
{
int i_local = mesh->GetSharedFace(i);
local_to_shared[i_local] = i;
}
for (int i = 0; i < mesh->GetNumFaces(); i++)
{
double shared_face_factor = 1.0;
bool shared_face = false;
int iel1, iel2, info1, info2;
mesh->GetFaceElements(i, &iel1, &iel2);
mesh->GetFaceInfos(i, &info1, &info2);
intorder = fes->GetFE(iel1)->GetOrder();
FaceElementTransformations *face_elem_transf;
const FiniteElement *fe1, *fe2;
if (info2 >= 0 && iel2 < 0)
{
int ishared = local_to_shared[i];
face_elem_transf = mesh->GetSharedFaceTransformations(ishared);
iel2 = face_elem_transf->Elem2No - mesh->GetNE();
fe2 = pfes->GetFaceNbrFE(iel2);
if ( (k = fe2->GetOrder()) > intorder )
{
intorder = k;
}
shared_face = true;
shared_face_factor = 0.5;
}
else
{
face_elem_transf = mesh->GetFaceElementTransformations(i);
if (iel2 >= 0)
{
fe2 = pfes->GetFE(iel2);
if ( (k = fe2->GetOrder()) > intorder )
{
intorder = k;
}
}
else
{
fe2 = NULL;
}
}
intorder = 2 * intorder; // <-------------
const IntegrationRule *ir;
if (irs)
{
ir = irs[face_elem_transf->GetGeometryType()];
}
else
{
ir = &(IntRules.Get(face_elem_transf->GetGeometryType(), intorder));
}
err_val.SetSize(ir->GetNPoints());
ell_coeff_val.SetSize(ir->GetNPoints());
// side 1
transf = face_elem_transf->Elem1;
fe1 = fes->GetFE(iel1);
fdof = fe1->GetDof();
fes->GetElementVDofs(iel1, vdofs);
shape.SetSize(fdof);
el_dofs.SetSize(fdof);
for (k = 0; k < fdof; k++)
if (vdofs[k] >= 0)
{
el_dofs(k) = (*this)(vdofs[k]);
}
else
{
el_dofs(k) = - (*this)(-1-vdofs[k]);
}
for (int j = 0; j < ir->GetNPoints(); j++)
{
face_elem_transf->Loc1.Transform(ir->IntPoint(j), eip);
fe1->CalcShape(eip, shape);
transf->SetIntPoint(&eip);
ell_coeff_val(j) = ell_coeff->Eval(*transf, eip);
err_val(j) = exsol->Eval(*transf, eip) - (shape * el_dofs);
}
if (fe2 != NULL)
{
// side 2
transf = face_elem_transf->Elem2;
fdof = fe2->GetDof();
shape.SetSize(fdof);
el_dofs.SetSize(fdof);
if (shared_face)
{
pfes->GetFaceNbrElementVDofs(iel2, vdofs);
for (k = 0; k < fdof; k++)
if (vdofs[k] >= 0)
{
el_dofs(k) = face_nbr_data[vdofs[k]];
}
else
{
el_dofs(k) = - face_nbr_data[-1-vdofs[k]];
}
}
else
{
pfes->GetElementVDofs(iel2, vdofs);
for (k = 0; k < fdof; k++)
if (vdofs[k] >= 0)
{
el_dofs(k) = (*this)(vdofs[k]);
}
else
{
el_dofs(k) = - (*this)(-1 - vdofs[k]);
}
}
for (int j = 0; j < ir->GetNPoints(); j++)
{
face_elem_transf->Loc2.Transform(ir->IntPoint(j), eip);
fe2->CalcShape(eip, shape);
transf->SetIntPoint(&eip);
ell_coeff_val(j) += ell_coeff->Eval(*transf, eip);
ell_coeff_val(j) *= 0.5;
err_val(j) -= (exsol->Eval(*transf, eip) - (shape * el_dofs));
}
}
transf = face_elem_transf;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
transf->SetIntPoint(&ip);
error += shared_face_factor*(ip.weight * Nu * ell_coeff_val(j) *
pow(transf->Weight(), 1.0-1.0/(dim-1)) *
err_val(j) * err_val(j));
}
}
error = (error < 0.0) ? -sqrt(-error) : sqrt(error);
return GlobalLpNorm(2.0, error, pfes->GetComm());
}
void ParGridFunction::Save(std::ostream &out) const
{
double *data_ = const_cast<double*>(HostRead());
@@ -1021,6 +860,7 @@ double GlobalLpNorm(const double p, double loc_norm, MPI_Comm comm)
return glob_norm;
}
void ParGridFunction::ComputeFlux(
BilinearFormIntegrator &blfi,
GridFunction &flux, bool wcoef, int subdomain)
@@ -1161,6 +1001,6 @@ double L2ZZErrorEstimator(BilinearFormIntegrator &flux_integrator,
return pow(glob_error, 1.0/norm_p);
}
} // namespace mfem
}
#endif // MFEM_USE_MPI
-71
View File
@@ -283,77 +283,6 @@ public:
pfes->GetComm());
}
/// Returns ||grad u_ex - grad u_h||_L2 for H1 or L2 elements
virtual double ComputeGradError(VectorCoefficient *exgrad,
const IntegrationRule *irs[] = NULL) const
{
return GlobalLpNorm(2.0, GridFunction::ComputeGradError(exgrad,irs),
pfes->GetComm());
}
/// Returns ||curl u_ex - curl u_h||_L2 for ND elements
virtual double ComputeCurlError(VectorCoefficient *excurl,
const IntegrationRule *irs[] = NULL) const
{
return GlobalLpNorm(2.0, GridFunction::ComputeCurlError(excurl,irs),
pfes->GetComm());
}
/// Returns ||div u_ex - div u_h||_L2 for RT elements
virtual double ComputeDivError(Coefficient *exdiv,
const IntegrationRule *irs[] = NULL) const
{
return GlobalLpNorm(2.0, GridFunction::ComputeDivError(exdiv,irs),
pfes->GetComm());
}
/// Returns the Face Jumps error for L2 elements
virtual double ComputeDGFaceJumpError(Coefficient *exsol,
Coefficient *ell_coeff,
double Nu,
const IntegrationRule *irs[]=NULL)
const;
/// Returns either the H1-seminorm or the DG Face Jumps error or both
/// depending on norm_type = 1, 2, 3
virtual double ComputeH1Error(Coefficient *exsol, VectorCoefficient *exgrad,
Coefficient *ell_coef, double Nu,
int norm_type) const
{
return GlobalLpNorm(2.0,
GridFunction::ComputeH1Error(exsol,exgrad,ell_coef,
Nu, norm_type),
pfes->GetComm());
}
/// Returns the error measured in H1-norm for H1 elements or in "broken"
/// H1-norm for L2 elements
virtual double ComputeH1Error(Coefficient *exsol, VectorCoefficient *exgrad,
const IntegrationRule *irs[] = NULL) const
{
return GlobalLpNorm(2.0, GridFunction::ComputeH1Error(exsol,exgrad,irs),
pfes->GetComm());
}
/// Returns the error measured H(div)-norm for RT elements
virtual double ComputeHDivError(VectorCoefficient *exsol,
Coefficient *exdiv,
const IntegrationRule *irs[] = NULL) const
{
return GlobalLpNorm(2.0, GridFunction::ComputeHDivError(exsol,exdiv,irs),
pfes->GetComm());
}
/// Returns the error measured H(curl)-norm for ND elements
virtual double ComputeHCurlError(VectorCoefficient *exsol,
VectorCoefficient *excurl,
const IntegrationRule *irs[] = NULL) const
{
return GlobalLpNorm(2.0,
GridFunction::ComputeHCurlError(exsol,excurl,irs),
pfes->GetComm());
}
virtual double ComputeMaxError(Coefficient *exsol[],
const IntegrationRule *irs[] = NULL) const
{
-1
View File
@@ -34,7 +34,6 @@ void ParLinearForm::MakeRef(FiniteElementSpace *f, Vector &v, int v_offset)
{
LinearForm::MakeRef(f, v, v_offset);
pfes = dynamic_cast<ParFiniteElementSpace*>(f);
MFEM_ASSERT(pfes != NULL, "not a ParFiniteElementSpace");
}
void ParLinearForm::MakeRef(ParFiniteElementSpace *pf, Vector &v, int v_offset)
+14 -16
View File
@@ -95,22 +95,20 @@ public:
/** @brief Make the ParLinearForm reference external data on a new
FiniteElementSpace. */
/** This method changes the FiniteElementSpace associated with the
ParLinearForm to @a *f and sets the data of the Vector @a v (plus the @a
v_offset) as external data in the ParLinearForm.
@note This version of the method will also perform bounds checks when the
build option MFEM_DEBUG is enabled. */
/** This method changes the FiniteElementSpace associated with the ParLinearForm
to @a *f and sets the data of the Vector @a v (plus the @a v_offset) as external
data in the ParLinearForm.
@note This version of the method will also perform bounds checks when
the build option MFEM_DEBUG is enabled. */
virtual void MakeRef(FiniteElementSpace *f, Vector &v, int v_offset);
/** @brief Make the ParLinearForm reference external data on a new
ParFiniteElementSpace. */
/** This method changes the ParFiniteElementSpace associated with the
ParLinearForm to @a *pf and sets the data of the Vector @a v (plus the @a
v_offset) as external data in the ParLinearForm.
@note This version of the method will also perform bounds checks when the
build option MFEM_DEBUG is enabled. */
/** This method changes the ParFiniteElementSpace associated with the ParLinearForm
to @a *pf and sets the data of the Vector @a v (plus the @a v_offset) as external
data in the ParLinearForm.
@note This version of the method will also perform bounds checks when
the build option MFEM_DEBUG is enabled. */
void MakeRef(ParFiniteElementSpace *pf, Vector &v, int v_offset);
/// Assemble the vector on the true dofs, i.e. P^t v.
@@ -120,10 +118,10 @@ public:
HypreParVector *ParallelAssemble();
/// Return the action of the ParLinearForm as a linear mapping.
/** Linear forms are linear functionals which map ParGridFunction%s to the
real numbers. This method performs this mapping which in this case is
equivalent as an inner product of the ParLinearForm and
ParGridFunction. */
/** Linear forms are linear functionals which map ParGridFunction%s to
the real numbers. This method performs this mapping which in
this case is equivalent as an inner product of the ParLinearForm
and ParGridFunction. */
double operator()(const ParGridFunction &gf) const
{
return InnerProduct(pfes->GetComm(), *this, gf);
+8 -14
View File
@@ -62,7 +62,6 @@ void QuadratureInterpolator::Eval2D(
const int nq = maps.nqpt;
const int ND = T_ND ? T_ND : nd;
const int NQ = T_NQ ? T_NQ : nq;
const int NMAX = NQ > ND ? NQ : ND;
const int VDIM = T_VDIM ? T_VDIM : vdim;
MFEM_VERIFY(ND <= MAX_ND2D, "");
MFEM_VERIFY(NQ <= MAX_NQ2D, "");
@@ -73,24 +72,22 @@ void QuadratureInterpolator::Eval2D(
auto val = Reshape(q_val.Write(), NQ, VDIM, NE);
auto der = Reshape(q_der.Write(), NQ, VDIM, 2, NE);
auto det = Reshape(q_det.Write(), NQ, NE);
MFEM_FORALL_2D(e, NE, NMAX, 1, 1,
MFEM_FORALL(e, NE,
{
const int ND = T_ND ? T_ND : nd;
const int NQ = T_NQ ? T_NQ : nq;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int max_ND = T_ND ? T_ND : MAX_ND2D;
constexpr int max_VDIM = T_VDIM ? T_VDIM : MAX_VDIM2D;
MFEM_SHARED double s_E[max_VDIM*max_ND];
MFEM_FOREACH_THREAD(d, x, ND)
double s_E[max_VDIM*max_ND];
for (int d = 0; d < ND; d++)
{
for (int c = 0; c < VDIM; c++)
{
s_E[c+d*VDIM] = E(d,c,e);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(q, x, NQ)
for (int q = 0; q < NQ; ++q)
{
if (eval_flags & VALUES)
{
@@ -153,7 +150,6 @@ void QuadratureInterpolator::Eval3D(
const int nq = maps.nqpt;
const int ND = T_ND ? T_ND : nd;
const int NQ = T_NQ ? T_NQ : nq;
const int NMAX = NQ > ND ? NQ : ND;
const int VDIM = T_VDIM ? T_VDIM : vdim;
MFEM_VERIFY(ND <= MAX_ND3D, "");
MFEM_VERIFY(NQ <= MAX_NQ3D, "");
@@ -164,24 +160,22 @@ void QuadratureInterpolator::Eval3D(
auto val = Reshape(q_val.Write(), NQ, VDIM, NE);
auto der = Reshape(q_der.Write(), NQ, VDIM, 3, NE);
auto det = Reshape(q_det.Write(), NQ, NE);
MFEM_FORALL_2D(e, NE, NMAX, 1, 1,
MFEM_FORALL(e, NE,
{
const int ND = T_ND ? T_ND : nd;
const int NQ = T_NQ ? T_NQ : nq;
const int VDIM = T_VDIM ? T_VDIM : vdim;
constexpr int max_ND = T_ND ? T_ND : MAX_ND3D;
constexpr int max_VDIM = T_VDIM ? T_VDIM : MAX_VDIM3D;
MFEM_SHARED double s_E[max_VDIM*max_ND];
MFEM_FOREACH_THREAD(d, x, ND)
double s_E[max_VDIM*max_ND];
for (int d = 0; d < ND; d++)
{
for (int c = 0; c < VDIM; c++)
{
s_E[c+d*VDIM] = E(d,c,e);
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD(q, x, NQ)
for (int q = 0; q < NQ; ++q)
{
if (eval_flags & VALUES)
{
+48 -47
View File
@@ -1968,11 +1968,11 @@ double TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
Jpt.SetSize(dim);
PMatI.UseExternalData(elfun.GetData(), dof, dim);
const IntegrationRule &ir = EnergyIntegrationRule(el);
const IntegrationRule *ir = EnergyIntegrationRule(el);
energy = 0.0;
DenseTensor Jtr(dim, dim, ir.GetNPoints());
targetC->ComputeElementTargets(T.ElementNo, el, ir, elfun, Jtr);
DenseTensor Jtr(dim, dim, ir->GetNPoints());
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
// Limited case.
Vector shape, p, p0, d_vals;
@@ -1989,11 +1989,11 @@ double TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
nodes0->GetSubVector(pos_dofs, pos0V);
if (lim_dist)
{
lim_dist->GetValues(T.ElementNo, ir, d_vals);
lim_dist->GetValues(T.ElementNo, *ir, d_vals);
}
else
{
d_vals.SetSize(ir.GetNPoints()); d_vals = 1.0;
d_vals.SetSize(ir->GetNPoints()); d_vals = 1.0;
}
}
@@ -2019,13 +2019,13 @@ double TMOP_Integrator::GetElementEnergy(const FiniteElement &el,
Vector zeta_q, zeta0_q;
if (adaptive_limiting)
{
zeta->GetValues(T.ElementNo, ir, zeta_q);
zeta_0->GetValues(T.ElementNo, ir, zeta0_q);
zeta->GetValues(T.ElementNo, *ir, zeta_q);
zeta_0->GetValues(T.ElementNo, *ir, zeta0_q);
}
for (int i = 0; i < ir.GetNPoints(); i++)
for (int i = 0; i < ir->GetNPoints(); i++)
{
const IntegrationPoint &ip = ir.IntPoint(i);
const IntegrationPoint &ip = ir->IntPoint(i);
const DenseMatrix &Jtr_i = Jtr(i);
metric->SetTargetJacobian(Jtr_i);
CalcInverse(Jtr_i, Jrt);
@@ -2105,14 +2105,14 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
elvect.SetSize(dof*dim);
PMatO.UseExternalData(elvect.GetData(), dof, dim);
const IntegrationRule &ir = ActionIntegrationRule(el);
const int nqp = ir.GetNPoints();
const IntegrationRule *ir = ActionIntegrationRule(el);
const int nqp = ir->GetNPoints();
elvect = 0.0;
Vector weights(nqp);
DenseTensor Jtr(dim, dim, nqp);
DenseTensor dJtr(dim, dim, dim*nqp);
targetC->ComputeElementTargets(T.ElementNo, el, ir, elfun, Jtr);
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
// Limited case.
DenseMatrix pos0;
@@ -2129,7 +2129,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
nodes0->GetSubVector(pos_dofs, pos0V);
if (lim_dist)
{
lim_dist->GetValues(T.ElementNo, ir, d_vals);
lim_dist->GetValues(T.ElementNo, *ir, d_vals);
}
else
{
@@ -2144,12 +2144,11 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
Tpr = new IsoparametricTransformation;
Tpr->SetFE(&el);
Tpr->ElementNo = T.ElementNo;
Tpr->ElementType = ElementTransformation::ELEMENT;
Tpr->Attribute = T.Attribute;
Tpr->GetPointMat().Transpose(PMatI); // PointMat = PMatI^T
if (exact_action)
{
targetC->ComputeElementTargetsGradient(ir, elfun, *Tpr, dJtr);
targetC->ComputeElementTargetsGradient(*ir, elfun, *Tpr, dJtr);
}
}
@@ -2159,7 +2158,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
const IntegrationPoint &ip = ir->IntPoint(q);
const DenseMatrix &Jtr_q = Jtr(q);
metric->SetTargetJacobian(Jtr_q);
CalcInverse(Jtr_q, Jrt);
@@ -2186,7 +2185,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
DenseMatrix dwdx(dim);
for (int d = 0; d < dim; d++)
{
const DenseMatrix &dJtr_q = dJtr(q + d * nqp);
const DenseMatrix &dJtr_q = dJtr(q + d*ir->GetNPoints());
Mult(Jrt, dJtr_q, dwdx );
d_detW_dx(d) = dwdx.Trace();
}
@@ -2221,7 +2220,7 @@ void TMOP_Integrator::AssembleElementVectorExact(const FiniteElement &el,
}
}
if (zeta) { AssembleElemVecAdaptLim(el, weights, *Tpr, ir, PMatO); }
if (zeta) { AssembleElemVecAdaptLim(el, weights, *Tpr, *ir, PMatO); }
delete Tpr;
}
@@ -2240,13 +2239,13 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
PMatI.UseExternalData(elfun.GetData(), dof, dim);
elmat.SetSize(dof*dim);
const IntegrationRule &ir = GradientIntegrationRule(el);
const int nqp = ir.GetNPoints();
const IntegrationRule *ir = GradientIntegrationRule(el);
const int nqp = ir->GetNPoints();
elmat = 0.0;
Vector weights(nqp);
DenseTensor Jtr(dim, dim, nqp);
targetC->ComputeElementTargets(T.ElementNo, el, ir, elfun, Jtr);
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
// Limited case.
DenseMatrix pos0, grad_grad;
@@ -2263,7 +2262,7 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
nodes0->GetSubVector(pos_dofs, pos0V);
if (lim_dist)
{
lim_dist->GetValues(T.ElementNo, ir, d_vals);
lim_dist->GetValues(T.ElementNo, *ir, d_vals);
}
else
{
@@ -2285,7 +2284,7 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
const IntegrationPoint &ip = ir->IntPoint(q);
const DenseMatrix &Jtr_q = Jtr(q);
metric->SetTargetJacobian(Jtr_q);
CalcInverse(Jtr_q, Jrt);
@@ -2302,6 +2301,7 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
// TODO: derivatives of adaptivity-based targets.
// TODO optimize by symmetry.
if (coeff0)
{
el.CalcShape(ip, shape);
@@ -2327,7 +2327,7 @@ void TMOP_Integrator::AssembleElementGradExact(const FiniteElement &el,
}
}
if (zeta) { AssembleElemGradAdaptLim(el, weights, *Tpr, ir, elmat); }
if (zeta) { AssembleElemGradAdaptLim(el, weights, *Tpr, *ir, elmat); }
delete Tpr;
}
@@ -2498,10 +2498,10 @@ void TMOP_Integrator::AssembleElementVectorFD(const FiniteElement &el,
// Contributions from adaptive limiting (exact derivatives).
if (zeta)
{
const IntegrationRule &ir = ActionIntegrationRule(el);
const int nqp = ir.GetNPoints();
const IntegrationRule *ir = ActionIntegrationRule(el);
const int nqp = ir->GetNPoints();
DenseTensor Jtr(dim, dim, nqp);
targetC->ComputeElementTargets(T.ElementNo, el, ir, elfun, Jtr);
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
IsoparametricTransformation Tpr;
Tpr.SetFE(&el);
@@ -2513,11 +2513,11 @@ void TMOP_Integrator::AssembleElementVectorFD(const FiniteElement &el,
Vector weights(nqp);
for (int q = 0; q < nqp; q++)
{
weights(q) = ir.IntPoint(q).weight * Jtr(q).Det();
weights(q) = ir->IntPoint(q).weight * Jtr(q).Det();
}
PMatO.UseExternalData(elvect.GetData(), dof, dim);
AssembleElemVecAdaptLim(el, weights, Tpr, ir, PMatO);
AssembleElemVecAdaptLim(el, weights, Tpr, *ir, PMatO);
}
}
@@ -2594,10 +2594,10 @@ void TMOP_Integrator::AssembleElementGradFD(const FiniteElement &el,
// Contributions from adaptive limiting.
if (zeta)
{
const IntegrationRule &ir = GradientIntegrationRule(el);
const int nqp = ir.GetNPoints();
const IntegrationRule *ir = GradientIntegrationRule(el);
const int nqp = ir->GetNPoints();
DenseTensor Jtr(dim, dim, nqp);
targetC->ComputeElementTargets(T.ElementNo, el, ir, elfun, Jtr);
targetC->ComputeElementTargets(T.ElementNo, el, *ir, elfun, Jtr);
IsoparametricTransformation Tpr;
Tpr.SetFE(&el);
@@ -2609,10 +2609,10 @@ void TMOP_Integrator::AssembleElementGradFD(const FiniteElement &el,
Vector weights(nqp);
for (int q = 0; q < nqp; q++)
{
weights(q) = ir.IntPoint(q).weight * Jtr(q).Det();
weights(q) = ir->IntPoint(q).weight * Jtr(q).Det();
}
AssembleElemGradAdaptLim(el, weights, Tpr, ir, elmat);
AssembleElemGradAdaptLim(el, weights, Tpr, *ir, elmat);
}
}
@@ -2642,32 +2642,33 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
Array<int> vdofs;
Vector x_vals;
const FiniteElementSpace* const fes = x.FESpace();
const FiniteElement *fe = fes->GetFE(0);
const int dim = fes->GetMesh()->Dimension();
const int dof = fes->GetFE(0)->GetDof(), dim = fes->GetFE(0)->GetDim();
DSh.SetSize(dof, dim);
Jrt.SetSize(dim);
Jpr.SetSize(dim);
Jpt.SetSize(dim);
const IntegrationRule *ir = EnergyIntegrationRule(*fe);
const int nqp = ir->GetNPoints();
DenseTensor Jtr(dim, dim, nqp);
metric_energy = 0.0;
lim_energy = 0.0;
for (int i = 0; i < fes->GetNE(); i++)
{
const FiniteElement *fe = fes->GetFE(i);
const IntegrationRule &ir = EnergyIntegrationRule(*fe);
const int nqp = ir.GetNPoints();
DenseTensor Jtr(dim, dim, nqp);
const int dof = fe->GetDof();
DSh.SetSize(dof, dim);
fe = fes->GetFE(i);
fes->GetElementVDofs(i, vdofs);
x.GetSubVector(vdofs, x_vals);
PMatI.UseExternalData(x_vals.GetData(), dof, dim);
targetC->ComputeElementTargets(i, *fe, ir, x_vals, Jtr);
targetC->ComputeElementTargets(i, *fe, *ir, x_vals, Jtr);
for (int q = 0; q < nqp; q++)
{
const IntegrationPoint &ip = ir.IntPoint(q);
const IntegrationPoint &ip = ir->IntPoint(q);
metric->SetTargetJacobian(Jtr(q));
CalcInverse(Jtr(q), Jrt);
const double weight = ip.weight * Jtr(q).Det();
@@ -2691,9 +2692,9 @@ void TMOP_Integrator::ComputeMinJac(const Vector &x,
const FiniteElementSpace &fes)
{
const FiniteElement *fe = fes.GetFE(0);
const IntegrationRule &ir = EnergyIntegrationRule(*fe);
const IntegrationRule *ir = EnergyIntegrationRule(*fe);
const int NE = fes.GetMesh()->GetNE(), dim = fe->GetDim(),
dof = fe->GetDof(), nsp = ir.GetNPoints();
dof = fe->GetDof(), nsp = ir->GetNPoints();
Array<int> xdofs(dof * dim);
DenseMatrix Jpr(dim), dshape(dof, dim), pos(dof, dim);
@@ -2710,7 +2711,7 @@ void TMOP_Integrator::ComputeMinJac(const Vector &x,
detv_sum = 0.;
for (int j = 0; j < nsp; j++)
{
fes.GetFE(i)->CalcDShape(ir.IntPoint(j), dshape);
fes.GetFE(i)->CalcDShape(ir->IntPoint(j), dshape);
MultAtB(pos, dshape, Jpr);
detv_sum += std::fabs(Jpr.Det());
}
+6 -22
View File
@@ -890,10 +890,6 @@ protected:
TMOP_QualityMetric *metric; // not owned
const TargetConstructor *targetC; // not owned
// Custom integration rules.
IntegrationRules *IntegRules;
int integ_order;
// Weight Coefficient multiplying the quality metric term.
Coefficient *coeff1; // not owned, if NULL -> coeff1 is 1.
// Normalization factor for the metric term.
@@ -992,21 +988,17 @@ protected:
nodes0 = NULL; coeff0 = NULL; lim_dist = NULL; lim_func = NULL;
}
const IntegrationRule &EnergyIntegrationRule(const FiniteElement &el) const
const IntegrationRule *EnergyIntegrationRule(const FiniteElement &el) const
{
if (IntegRules)
{
return IntegRules->Get(el.GetGeomType(), integ_order);
}
return (IntRule) ? *IntRule
/* */ : IntRules.Get(el.GetGeomType(), 2*el.GetOrder() + 3);
return (IntRule) ? IntRule
/* */ : &(IntRules.Get(el.GetGeomType(), 2*el.GetOrder() + 3));
}
const IntegrationRule &ActionIntegrationRule(const FiniteElement &el) const
const IntegrationRule *ActionIntegrationRule(const FiniteElement &el) const
{
// TODO the energy most likely needs less integration points.
return EnergyIntegrationRule(el);
}
const IntegrationRule &GradientIntegrationRule(const FiniteElement &el) const
const IntegrationRule *GradientIntegrationRule(const FiniteElement &el) const
{
// TODO the action and energy most likely need less integration points.
return EnergyIntegrationRule(el);
@@ -1016,7 +1008,7 @@ public:
/** @param[in] m TMOP_QualityMetric that will be integrated (not owned).
@param[in] tc Target-matrix construction algorithm to use (not owned). */
TMOP_Integrator(TMOP_QualityMetric *m, TargetConstructor *tc)
: metric(m), targetC(tc), IntegRules(NULL), integ_order(-1),
: metric(m), targetC(tc),
coeff1(NULL), metric_normal(1.0),
nodes0(NULL), coeff0(NULL),
lim_dist(NULL), lim_func(NULL), lim_normal(1.0),
@@ -1027,14 +1019,6 @@ public:
~TMOP_Integrator();
/// Prescribe a set of integration rules; relevant for mixed meshes.
/** This function has priority over SetIntRule(), if both are called. */
void SetIntegrationRules(IntegrationRules &irules, int order)
{
IntegRules = &irules;
integ_order = order;
}
/// Sets a scaling Coefficient for the quality metric term of the integrator.
/** With this addition, the integrator becomes
@f$ \int w1 W(Jpt) dx @f$.
+38 -30
View File
@@ -176,8 +176,8 @@ SerialAdvectorCGOper::SerialAdvectorCGOper(const Vector &x_start,
MassIntegrator *Minteg = new MassIntegrator;
M.AddDomainIntegrator(Minteg);
M.Assemble(0);
M.Finalize(0);
M.Assemble();
M.Finalize();
}
void SerialAdvectorCGOper::Mult(const Vector &ind, Vector &di_dt) const
@@ -220,8 +220,8 @@ ParAdvectorCGOper::ParAdvectorCGOper(const Vector &x_start,
MassIntegrator *Minteg = new MassIntegrator;
M.AddDomainIntegrator(Minteg);
M.Assemble(0);
M.Finalize(0);
M.Assemble();
M.Finalize();
}
void ParAdvectorCGOper::Mult(const Vector &ind, Vector &di_dt) const
@@ -298,12 +298,34 @@ void InterpolatorFP::SetInitialField(const Vector &init_nodes,
field0_gf = init_field;
dim = f->GetFE(0)->GetDim();
const int pts_cnt = init_nodes.Size() / dim;
el_id_out.SetSize(pts_cnt);
code_out.SetSize(pts_cnt);
task_id_out.SetSize(pts_cnt);
pos_r_out.SetSize(pts_cnt*dim);
dist_p_out.SetSize(pts_cnt);
}
void InterpolatorFP::ComputeAtNewPosition(const Vector &new_nodes,
Vector &new_field)
{
finder->Interpolate(new_nodes, field0_gf, new_field);
const int pts_cnt = new_nodes.Size() / dim;
// The sizes may change between calls due to AMR.
if (el_id_out.Size() != pts_cnt)
{
el_id_out.SetSize(pts_cnt);
code_out.SetSize(pts_cnt);
task_id_out.SetSize(pts_cnt);
pos_r_out.SetSize(pts_cnt*dim);
dist_p_out(pts_cnt);
}
// Interpolate FE function values on the found points.
finder->FindPoints(new_nodes, code_out, task_id_out,
el_id_out, pos_r_out, dist_p_out);
finder->Interpolate(code_out, task_id_out, el_id_out,
pos_r_out, field0_gf, new_field);
}
#endif
@@ -331,12 +353,13 @@ double TMOPNewtonSolver::ComputeScalingFactor(const Vector &x,
energy_in = nlf->GetEnergy(x);
}
const int NE = fes->GetMesh()->GetNE(), dim = fes->GetMesh()->Dimension();
Array<int> xdofs;
DenseMatrix Jpr(dim);
// Get the local prolongation of the solution vector.
const int NE = fes->GetMesh()->GetNE(), dim = fes->GetFE(0)->GetDim(),
dof = fes->GetFE(0)->GetDof(), nsp = ir.GetNPoints();
Array<int> xdofs(dof * dim);
DenseMatrix Jpr(dim), dshape(dof, dim), pos(dof, dim);
Vector posV(pos.Data(), dof * dim);
Vector x_out_loc(fes->GetVSize());
if (serial)
{
const SparseMatrix *cP = fes->GetConformingProlongation();
@@ -350,23 +373,15 @@ double TMOPNewtonSolver::ComputeScalingFactor(const Vector &x,
}
#endif
// Check if the starting mesh (given by x) is inverted.
// Note that x hasn't been modified by the Newton update yet.
double min_detJ = infinity();
for (int i = 0; i < NE; i++)
{
const int dof = fes->GetFE(i)->GetDof();
DenseMatrix dshape(dof, dim), pos(dof, dim);
Vector posV(pos.Data(), dof * dim);
fes->GetElementVDofs(i, xdofs);
x_out_loc.GetSubVector(xdofs, posV);
const IntegrationRule &irule = GetIntegrationRule(*fes->GetFE(i));
const int nsp = irule.GetNPoints();
for (int j = 0; j < nsp; j++)
{
fes->GetFE(i)->CalcDShape(irule.IntPoint(j), dshape);
fes->GetFE(i)->CalcDShape(ir.IntPoint(j), dshape);
MultAtB(pos, dshape, Jpr);
min_detJ = std::min(min_detJ, Jpr.Det());
}
@@ -379,18 +394,18 @@ double TMOPNewtonSolver::ComputeScalingFactor(const Vector &x,
p_nlf->ParFESpace()->GetComm());
}
#endif
const bool untangling = (min_detJ_all <= 0) ? true : false;
bool untangling = false;
if (min_detJ_all <= 0) { untangling = true; }
const bool have_b = (b.Size() == Height());
Vector x_out(x.Size());
bool x_out_ok = false;
double scale = 1.0, energy_out = 0.0;
const double norm0 = Norm(r);
double norm0 = Norm(r);
const double detJ_factor = (solver_type == 1) ? 0.25 : 0.5;
// Perform the line search.
for (int i = 0; i < 12; i++)
{
add(x, -scale, c, x_out);
@@ -414,18 +429,11 @@ double TMOPNewtonSolver::ComputeScalingFactor(const Vector &x,
int jac_ok = 1;
for (int i = 0; i < NE; i++)
{
const int dof = fes->GetFE(i)->GetDof();
DenseMatrix dshape(dof, dim), pos(dof, dim);
Vector posV(pos.Data(), dof * dim);
fes->GetElementVDofs(i, xdofs);
x_out_loc.GetSubVector(xdofs, posV);
const IntegrationRule &irule = GetIntegrationRule(*fes->GetFE(i));
const int nsp = irule.GetNPoints();
for (int j = 0; j < nsp; j++)
{
fes->GetFE(i)->CalcDShape(irule.IntPoint(j), dshape);
fes->GetFE(i)->CalcDShape(ir.IntPoint(j), dshape);
MultAtB(pos, dshape, Jpr);
if (Jpr.Det() <= 0.0) { jac_ok = 0; goto break2; }
}
+4 -25
View File
@@ -49,6 +49,8 @@ private:
Vector nodes0;
GridFunction field0_gf;
FindPointsGSLIB *finder;
Array<uint> el_id_out, code_out, task_id_out;
Vector pos_r_out, dist_p_out;
int dim;
public:
InterpolatorFP() : finder(NULL) { }
@@ -116,39 +118,16 @@ protected:
// Quadrature points that are checked for negative Jacobians etc.
const IntegrationRule &ir;
// These fields are relevant for mixed meshes.
IntegrationRules *IntegRules;
int integ_order;
const IntegrationRule &GetIntegrationRule(const FiniteElement &el) const
{
if (IntegRules)
{
return IntegRules->Get(el.GetGeomType(), integ_order);
}
return ir;
}
void UpdateDiscreteTC(const TMOP_Integrator &ti, const Vector &x_new) const;
public:
#ifdef MFEM_USE_MPI
TMOPNewtonSolver(MPI_Comm comm, const IntegrationRule &irule, int type = 0)
: LBFGSSolver(comm), solver_type(type), parallel(true),
ir(irule), IntegRules(NULL), integ_order(-1) { }
: LBFGSSolver(comm), solver_type(type), parallel(true), ir(irule) { }
#endif
TMOPNewtonSolver(const IntegrationRule &irule, int type = 0)
: LBFGSSolver(), solver_type(type), parallel(false),
ir(irule), IntegRules(NULL), integ_order(-1) { }
/// Prescribe a set of integration rules; relevant for mixed meshes.
/** If called, this function has priority over the IntegrationRule given to
the constructor of the class. */
void SetIntegrationRules(IntegrationRules &irules, int order)
{
IntegRules = &irules;
integ_order = order;
}
: LBFGSSolver(), solver_type(type), parallel(false), ir(irule) { }
virtual double ComputeScalingFactor(const Vector &x, const Vector &b) const;
+3 -29
View File
@@ -235,8 +235,8 @@ void adios2stream::Print(const Mesh& mesh, const mode print_mode)
}
// format info
SafeDefineAttribute<std::string>(io, "format", "MFEM ADIOS2 BP v0.2" );
SafeDefineAttribute<std::string>(io, "format/version", "0.2" );
SafeDefineAttribute<std::string>(io, "format", "MFEM ADIOS2 BP v0.1" );
SafeDefineAttribute<std::string>(io, "format/version", "0.1" );
std::string mesh_type = "Unknown";
std::vector<std::string> viz_tools;
viz_tools.reserve(2); //for now
@@ -298,7 +298,6 @@ void adios2stream::Print(const Mesh& mesh, const mode print_mode)
element_nvertices = static_cast<size_t>(mesh.elements[0]->GetNVertices());
}
SafeDefineVariable<uint64_t>(io, "connectivity", {}, {}, {nelements, element_nvertices+1});
SafeDefineVariable<int32_t>(io, "material", {}, {}, {nelements});
// vertices
SafeDefineVariable<uint32_t>(io,"NumOfVertices", {adios2::LocalValueDim});
@@ -349,15 +348,8 @@ void adios2stream::Print(const Mesh& mesh, const mode print_mode)
io.InquireVariable<uint64_t>("connectivity");
adios2::Variable<uint64_t>::Span span_connectivity = engine.Put<uint64_t>
(var_connectivity);
adios2::Variable<int32_t> var_element_attribute =
io.InquireVariable<int32_t>("material");
adios2::Variable<int32_t>::Span span_element_attribute = engine.Put<int32_t>
(var_element_attribute);
size_t span_vertices_offset = 0;
size_t span_connectivity_offset = 0;
size_t span_element_attribute_offset = 0;
// use for setting absolute node id for each element
size_t point_id = 0;
DenseMatrix pmatrix;
@@ -378,9 +370,6 @@ void adios2stream::Print(const Mesh& mesh, const mode print_mode)
}
span_vertices_offset += static_cast<size_t>(pmatrix.Width()*pmatrix.Height());
// element attribute
const int element_attribute = mesh.GetAttribute(e);
// connectivity
const int nv = Geometries.GetVertices(type)->GetNPoints();
const Array<int> &element_vertices = refined_geometry->RefGeoms;
@@ -390,10 +379,6 @@ void adios2stream::Print(const Mesh& mesh, const mode print_mode)
span_connectivity[span_connectivity_offset] = static_cast<uint64_t>(nv);
++span_connectivity_offset;
span_element_attribute[span_element_attribute_offset] = static_cast<int32_t>
(element_attribute);
++span_element_attribute_offset;
for (int k =0; k < nv; k++, v++ )
{
span_connectivity[span_connectivity_offset] = static_cast<uint64_t>
@@ -434,17 +419,9 @@ void adios2stream::Print(const Mesh& mesh, const mode print_mode)
adios2::Variable<uint64_t>::Span spanConnectivity =
engine.Put<uint64_t>(varConnectivity);
adios2::Variable<int32_t> varElementAttribute =
io.InquireVariable<int32_t>("material");
// zero-copy access to adios2 buffer to put non-contiguous to contiguous memory
adios2::Variable<int32_t>::Span spanElementAttribute =
engine.Put<int32_t>(varElementAttribute);
size_t elementPosition = 0;
for (int e = 0; e < mesh.GetNE(); ++e)
{
spanElementAttribute[e] = static_cast<int32_t>(mesh.GetAttribute(e));
const int nVertices = mesh.elements[e]->GetNVertices();
spanConnectivity[elementPosition] = nVertices;
for (int v = 0; v < nVertices; ++v)
@@ -711,7 +688,7 @@ std::string adios2stream::VTKSchema() const noexcept
{
std::string vtkSchema = R"(
<?xml version="1.0"?>
<VTKFile type="UnstructuredGrid" version="0.2" byte_order="LittleEndian">
<VTKFile type="UnstructuredGrid" version="0.1" byte_order="LittleEndian">
<UnstructuredGrid>
<Piece NumberOfPoints="NumOfVertices" NumberOfCells="NumOfElements">
<Points>
@@ -719,9 +696,6 @@ std::string adios2stream::VTKSchema() const noexcept
vtkSchema += R"(
</Points>
<CellData>
<DataArray Name="material" />
</CellData>
<Cells>
<DataArray Name="connectivity" />
<DataArray Name="types" />
+4 -18
View File
@@ -12,10 +12,9 @@
#include "forall.hpp"
#include "occa.hpp"
#ifdef MFEM_USE_CEED
#include "../fem/libceed/ceed.hpp"
#include <ceed.h>
#endif
#include <unordered_map>
#include <string>
#include <map>
@@ -34,16 +33,13 @@ occa::device occaDevice;
#ifdef MFEM_USE_CEED
Ceed ceed = NULL;
CeedBasisMap ceed_basis_map;
CeedRestrMap ceed_restr_map;
#endif
// Backends listed by priority, high to low:
static const Backend::Id backend_list[Backend::NUM_BACKENDS] =
{
Backend::CEED_CUDA, Backend::OCCA_CUDA, Backend::RAJA_CUDA, Backend::CUDA,
Backend::HIP, Backend::DEBUG_DEVICE,
Backend::HIP, Backend::DEBUG,
Backend::OCCA_OMP, Backend::RAJA_OMP, Backend::OMP,
Backend::CEED_CPU, Backend::OCCA_CPU, Backend::RAJA_CPU, Backend::CPU
};
@@ -158,16 +154,6 @@ Device::~Device()
{
free(device_option);
#ifdef MFEM_USE_CEED
// Destroy FES -> CeedBasis, CeedElemRestriction hash table contents
for (auto entry : internal::ceed_basis_map)
{
CeedBasisDestroy(&entry.second);
}
for (auto entry : internal::ceed_restr_map)
{
CeedElemRestrictionDestroy(&entry.second);
}
// Destroy Ceed context
CeedDestroy(&internal::ceed);
#endif
mm.Destroy();
@@ -280,7 +266,7 @@ void Device::Print(std::ostream &out)
void Device::UpdateMemoryTypeAndClass()
{
const bool debug = Device::Allows(Backend::DEBUG_DEVICE);
const bool debug = Device::Allows(Backend::DEBUG);
const bool device = Device::Allows(Backend::DEVICE_MASK);
@@ -518,7 +504,7 @@ void Device::Setup(const int device)
CeedDeviceSetup(device_option);
}
}
if (Allows(Backend::DEBUG_DEVICE)) { ngpu = 1; }
if (Allows(Backend::DEBUG)) { ngpu = 1; }
}
} // mfem
+4 -6
View File
@@ -64,9 +64,8 @@ struct Backend
/** @brief [device] Debug backend: host memory is READ/WRITE protected
while a device is in use. It allows to test the "device" code-path
(using separate host/device memory pools and host <-> device
transfers) without any GPU hardware. As 'DEBUG' is sometimes used
as a macro, `_DEVICE` has been added to avoid conflicts. */
DEBUG_DEVICE = 1 << 12
transfers) without any GPU hardware. */
DEBUG = 1 << 12
};
/** @brief Additional useful constants. For example, the *_MASK constants can
@@ -87,7 +86,7 @@ struct Backend
/// Bitwise-OR of all CEED backends
CEED_MASK = CEED_CPU | CEED_CUDA,
/// Biwise-OR of all device backends
DEVICE_MASK = CUDA_MASK | HIP_MASK | DEBUG_DEVICE,
DEVICE_MASK = CUDA_MASK | HIP_MASK | DEBUG,
/// Biwise-OR of all RAJA backends
RAJA_MASK = RAJA_CPU | RAJA_OMP | RAJA_CUDA,
@@ -194,8 +193,7 @@ public:
* The available backends are described by the Backend class.
* The string name of a backend is the lowercase version of the
Backend::Id enumeration constant with '_' replaced by '-', e.g. the
string name of 'RAJA_CPU' is 'raja-cpu'. The string name of the debug
backend (Backend::Id 'DEBUG_DEVICE') is exceptionally set to 'debug'.
string name of 'RAJA_CPU' is 'raja-cpu'.
* The 'cpu' backend is always enabled with lowest priority.
* The current backend priority from highest to lowest is:
'ceed-cuda', 'occa-cuda', 'raja-cuda', 'cuda', 'hip', 'debug',
+1 -1
View File
@@ -343,7 +343,7 @@ inline void ForallWrap(const bool use_dev, const int N,
{ return HipWrap3D(N, d_body, X, Y, Z); }
#endif
if (Device::Allows(Backend::DEBUG_DEVICE)) { goto backend_cpu; }
if (Device::Allows(Backend::DEBUG)) { goto backend_cpu; }
#if defined(MFEM_USE_RAJA) && defined(RAJA_ENABLE_OPENMP)
// Handle all allowed OpenMP backends except Backend::OMP
+12 -18
View File
@@ -136,10 +136,8 @@ struct Memory
void *d_ptr;
const size_t bytes;
const MemoryType h_mt, d_mt;
mutable bool h_rw, d_rw;
Memory(void *p, size_t b, MemoryType h, MemoryType d):
h_ptr(p), d_ptr(nullptr), bytes(b), h_mt(h), d_mt(d),
h_rw(true), d_rw(true) { }
h_ptr(p), d_ptr(nullptr), bytes(b), h_mt(h), d_mt(d) { }
};
/// Alias class that holds the base memory region and the offset
@@ -175,8 +173,8 @@ public:
virtual ~HostMemorySpace() { }
virtual void Alloc(void **ptr, size_t bytes) { *ptr = std::malloc(bytes); }
virtual void Dealloc(void *ptr) { std::free(ptr); }
virtual void Protect(const Memory&, size_t) { }
virtual void Unprotect(const Memory&, size_t) { }
virtual void Protect(const void*, size_t) { }
virtual void Unprotect(const void*, size_t) { }
virtual void AliasProtect(const void*, size_t) { }
virtual void AliasUnprotect(const void*, size_t) { }
};
@@ -354,10 +352,8 @@ public:
MmuHostMemorySpace(): HostMemorySpace() { MmuInit(); }
void Alloc(void **ptr, size_t bytes) { MmuAlloc(ptr, bytes); }
void Dealloc(void *ptr) { MmuDealloc(ptr, maps->memories.at(ptr).bytes); }
void Protect(const Memory& mem, size_t bytes)
{ if (mem.h_rw) { mem.h_rw = false; MmuProtect(mem.h_ptr, bytes); } }
void Unprotect(const Memory &mem, size_t bytes)
{ if (!mem.h_rw) { mem.h_rw = true; MmuAllow(mem.h_ptr, bytes); } }
void Protect(const void *ptr, size_t bytes) { MmuProtect(ptr, bytes); }
void Unprotect(const void *ptr, size_t bytes) { MmuAllow(ptr, bytes); }
/// Aliases need to be restricted during protection
void AliasProtect(const void *ptr, size_t bytes)
{ MmuProtect(MmuAddrR(ptr), MmuLengthR(ptr, bytes)); }
@@ -446,10 +442,8 @@ public:
MmuDeviceMemorySpace(): DeviceMemorySpace() { }
void Alloc(Memory &m) { MmuAlloc(&m.d_ptr, m.bytes); }
void Dealloc(Memory &m) { MmuDealloc(m.d_ptr, m.bytes); }
void Protect(const Memory &m)
{ if (m.d_rw) { m.d_rw = false; MmuProtect(m.d_ptr, m.bytes); } }
void Unprotect(const Memory &m)
{ if (!m.d_rw) { m.d_rw = true; MmuAllow(m.d_ptr, m.bytes); } }
void Protect(const Memory &m) { MmuProtect(m.d_ptr, m.bytes); }
void Unprotect(const Memory &m) { MmuAllow(m.d_ptr, m.bytes); }
/// Aliases need to be restricted during protection
void AliasProtect(const void *ptr, size_t bytes)
{ MmuProtect(MmuAddrR(ptr), MmuLengthR(ptr, bytes)); }
@@ -975,8 +969,11 @@ void MemoryManager::Copy_(void *dst_h_ptr, const void *src_h_ptr,
{
if (dst_h_ptr != src_d_ptr && bytes != 0)
{
internal::Memory &dst_h_base = maps->memories.at(dst_h_ptr);
internal::Memory &src_d_base = maps->memories.at(src_d_ptr);
MemoryType dst_h_mt = dst_h_base.h_mt;
MemoryType src_d_mt = src_d_base.d_mt;
ctrl->Host(dst_h_mt)->Unprotect(dst_h_ptr, bytes);
ctrl->Device(src_d_mt)->DtoH(dst_h_ptr, src_d_ptr, bytes);
}
}
@@ -1177,14 +1174,13 @@ void *MemoryManager::GetDevicePtr(const void *h_ptr, size_t bytes,
const MemoryType &d_mt = mem.d_mt;
MFEM_VERIFY_TYPES(h_mt, d_mt);
if (!mem.d_ptr) { ctrl->Device(d_mt)->Alloc(mem); }
// Aliases might have done some protections
ctrl->Device(d_mt)->Unprotect(mem);
if (copy_data)
{
MFEM_ASSERT(bytes <= mem.bytes, "invalid copy size");
ctrl->Device(d_mt)->HtoD(mem.d_ptr, h_ptr, bytes);
}
ctrl->Host(h_mt)->Protect(mem, bytes);
ctrl->Host(h_mt)->Protect(h_ptr, bytes);
return mem.d_ptr;
}
@@ -1210,7 +1206,6 @@ void *MemoryManager::GetAliasDevicePtr(const void *alias_ptr, size_t bytes,
void *alias_d_ptr = static_cast<char*>(mem.d_ptr) + offset;
MFEM_ASSERT(alias_h_ptr == alias_ptr, "internal error");
MFEM_ASSERT(bytes <= alias.bytes, "internal error");
mem.d_rw = false;
ctrl->Device(d_mt)->AliasUnprotect(alias_d_ptr, bytes);
ctrl->Host(h_mt)->AliasUnprotect(alias_ptr, bytes);
if (copy) { ctrl->Device(d_mt)->HtoD(alias_d_ptr, alias_h_ptr, bytes); }
@@ -1226,8 +1221,8 @@ void *MemoryManager::GetHostPtr(const void *ptr, size_t bytes, bool copy)
const MemoryType &h_mt = mem.h_mt;
const MemoryType &d_mt = mem.d_mt;
MFEM_VERIFY_TYPES(h_mt, d_mt);
ctrl->Host(h_mt)->Unprotect(mem.h_ptr, bytes);
// Aliases might have done some protections
ctrl->Host(h_mt)->Unprotect(mem, bytes);
if (mem.d_ptr) { ctrl->Device(d_mt)->Unprotect(mem); }
if (copy && mem.d_ptr) { ctrl->Device(d_mt)->DtoH(mem.h_ptr, mem.d_ptr, bytes); }
if (mem.d_ptr) { ctrl->Device(d_mt)->Protect(mem); }
@@ -1245,7 +1240,6 @@ void *MemoryManager::GetAliasHostPtr(const void *ptr, size_t bytes,
void *alias_h_ptr = static_cast<char*>(mem->h_ptr) + alias.offset;
void *alias_d_ptr = static_cast<char*>(mem->d_ptr) + alias.offset;
MFEM_ASSERT(alias_h_ptr == ptr, "internal error");
mem->h_rw = false;
ctrl->Host(h_mt)->AliasUnprotect(alias_h_ptr, bytes);
if (mem->d_ptr) { ctrl->Device(d_mt)->AliasUnprotect(alias_d_ptr, bytes); }
if (copy_data && mem->d_ptr)
-12
View File
@@ -76,13 +76,6 @@ if (MFEM_USE_GINKGO)
list(APPEND HDRS ginkgo.hpp)
endif()
if (MFEM_USE_MUMPS)
list(APPEND SRCS mumps.cpp)
# If this list (HDRS -> HEADERS) is used for install, we probably want the
# header added all the time.
list(APPEND HDRS mumps.hpp)
endif()
if (MFEM_USE_SUNDIALS)
list(APPEND SRCS sundials.cpp)
list(APPEND HDRS sundials.hpp)
@@ -105,11 +98,6 @@ if (MFEM_USE_HIOP)
list(APPEND HDRS hiop.hpp)
endif()
if (MFEM_USE_MKL_CPARDISO)
list(APPEND SRCS cpardiso.cpp)
list(APPEND HDRS cpardiso.hpp)
endif()
convert_filenames_to_full_paths(SRCS)
convert_filenames_to_full_paths(HDRS)
+3 -3
View File
@@ -100,7 +100,7 @@ public:
/** @brief Real or imaginary part accessor methods
The following accessor methods should only be called if the requested
part of the operator is known to exist. This can be checked with
part of the opertor is known to exist. This can be checked with
hasRealPart() or hasImagPart().
*/
virtual Operator & real();
@@ -166,7 +166,7 @@ public:
/** Combine the blocks making up this complex operator into a single
SparseMatrix. The resulting matrix can be passed to solvers which require
access to the matrix entries themselves, such as sparse direct solvers,
rather than simply the action of the operator. Note that this combined
rather than simply the action of the opertor. Note that this combined
operator requires roughly twice the memory of the block structured
operator. */
SparseMatrix * GetSystemMatrix() const;
@@ -269,7 +269,7 @@ public:
HypreParMatrix. The resulting matrix can be passed to solvers which
require access to the matrix entries themselves, such as sparse direct
solvers or Hypre preconditioners, rather than simply the action of the
operator. Note that this combined operator requires roughly twice the
opertor. Note that this combined operator requires roughly twice the
memory of the block structured operator. */
HypreParMatrix * GetSystemMatrix() const;
-233
View File
@@ -1,233 +0,0 @@
#include "../config/config.hpp"
#ifdef MFEM_USE_MKL_CPARDISO
#ifdef MFEM_USE_MPI
#include "cpardiso.hpp"
#include "hypre.hpp"
#include <algorithm>
#include <vector>
#include <numeric>
namespace mfem
{
CPardisoSolver::CPardisoSolver(MPI_Comm comm) : comm_(comm)
{
// Solver default parameters overridden with provided by iparm
iparm[0] = 1;
// Use METIS for fill-in reordering
iparm[1] = 2;
// Write solution into x
iparm[5] = 0;
// Max number of iterative refinement steps
iparm[7] = 2;
// Perturb the pivot elements with 1E-13
iparm[9] = 13;
// Use non-symmetric permutation and scaling MPS
iparm[10] = 1;
// Switch on Maximum Weighted Matching algorithm (default for non-symmetric)
iparm[12] = 1;
// Output: Number of non-zeros in the factor LU
iparm[17] = -1;
// Output: Mflops for LU factorization
iparm[18] = -1;
// Check input data for correctness
iparm[26] = 1;
// 0-based indexing
iparm[34] = 1;
// All inputs are distributed between MPI processes
iparm[39] = 2;
// Maximum number of numerical factorizations
maxfct = 1;
// Which factorization to use
mnum = 1;
// Print statistical information in file
msglvl = 0;
// Initialize error flag
error = 0;
// Real unsymmetric matrix
mtype = MatType::REAL_UNSYMMETRIC;
// Number of right hand sides
nrhs = 1;
};
void CPardisoSolver::SetOperator(const Operator &op)
{
auto hypreParMat = dynamic_cast<const HypreParMatrix &>(op);
MFEM_ASSERT(hypreParMat, "Must pass HypreParMatrix as Operator");
auto parcsr_op = static_cast<hypre_ParCSRMatrix *>(
const_cast<HypreParMatrix &>(hypreParMat));
hypre_CSRMatrix *csr_op = hypre_MergeDiagAndOffd(parcsr_op);
#if MFEM_HYPRE_VERSION >= 21600
hypre_CSRMatrixBigJtoJ(csr_op);
#endif
m = parcsr_op->global_num_rows;
first_row = parcsr_op->first_row_index;
nnz_loc = csr_op->num_nonzeros;
m_loc = csr_op->num_rows;
height = m_loc;
width = m_loc;
double *csr_nzval = csr_op->data;
int *csr_colind = csr_op->j;
delete[] csr_rowptr;
delete[] reordered_csr_colind;
delete[] reordered_csr_nzval;
csr_rowptr = new int[m_loc + 1];
reordered_csr_colind = new int[nnz_loc];
reordered_csr_nzval = new double[nnz_loc];
for (int i = 0; i <= m_loc; i++)
{
csr_rowptr[i] = (csr_op->i)[i];
}
// CPardiso expects the column indices to be sorted for each row
std::vector<int> permutation_idx(nnz_loc);
std::iota(permutation_idx.begin(), permutation_idx.end(), 0);
for (int i = 0; i < m_loc; i++)
{
std::sort(permutation_idx.begin() + csr_rowptr[i],
permutation_idx.begin() + csr_rowptr[i + 1],
[csr_colind](int i1, int i2)
{
return csr_colind[i1] < csr_colind[i2];
});
}
for (int i = 0; i < nnz_loc; i++)
{
reordered_csr_colind[i] = csr_colind[permutation_idx[i]];
reordered_csr_nzval[i] = csr_nzval[permutation_idx[i]];
}
hypre_CSRMatrixDestroy(csr_op);
// The number of row in global matrix, rhs element and solution vector that
// begins the input domain belonging to this MPI process
iparm[40] = first_row;
// The number of row in global matrix, rhs element and solution vector that
// ends the input domain belonging to this MPI process
iparm[41] = first_row + m_loc - 1;
// Analyze inputs
phase = 11;
cluster_sparse_solver(pt,
&maxfct,
&mnum,
&mtype,
&phase,
&m,
reordered_csr_nzval,
csr_rowptr,
reordered_csr_colind,
&idum,
&nrhs,
iparm,
&msglvl,
&ddum,
&ddum,
&comm_,
&error);
MFEM_ASSERT(error == 0, "Pardiso analyze input error");
// Numerical factorization
phase = 22;
cluster_sparse_solver(pt,
&maxfct,
&mnum,
&mtype,
&phase,
&m,
reordered_csr_nzval,
csr_rowptr,
reordered_csr_colind,
&idum,
&nrhs,
iparm,
&msglvl,
&ddum,
&ddum,
&comm_,
&error);
MFEM_ASSERT(error == 0, "Pardiso factorization input error");
}
void CPardisoSolver::Mult(const Vector &b, Vector &x) const
{
// Solve
phase = 33;
cluster_sparse_solver(pt,
&maxfct,
&mnum,
&mtype,
&phase,
&m,
reordered_csr_nzval,
csr_rowptr,
reordered_csr_colind,
&idum,
&nrhs,
iparm,
&msglvl,
b.GetData(),
x.GetData(),
&comm_,
&error);
MFEM_ASSERT(error == 0, "Pardiso solve error");
}
void CPardisoSolver::SetPrintLevel(int print_level)
{
msglvl = print_level;
}
void CPardisoSolver::SetMatrixType(MatType mat_type)
{
mtype = mat_type;
}
CPardisoSolver::~CPardisoSolver()
{
// Release all internal memory
phase = -1;
cluster_sparse_solver(pt,
&maxfct,
&mnum,
&mtype,
&phase,
&m,
reordered_csr_nzval,
csr_rowptr,
reordered_csr_colind,
&idum,
&nrhs,
iparm,
&msglvl,
&ddum,
&ddum,
&comm_,
&error);
MFEM_ASSERT(error == 0, "Pardiso free error");
delete[] csr_rowptr;
delete[] reordered_csr_colind;
delete[] reordered_csr_nzval;
}
} // namespace mfem
#endif // MFEM_USE_MKL_CPARDISO
#endif // MFEM_USE_MPI
-125
View File
@@ -1,125 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_CPARDISO
#define MFEM_CPARDISO
#include "../config/config.hpp"
#ifdef MFEM_USE_MKL_CPARDISO
#ifdef MFEM_USE_MPI
#include "operator.hpp"
#include <mpi.h>
#include "mkl_cluster_sparse_solver.h"
namespace mfem
{
/**
* @brief MKL Parallel Direct Sparse Solver for Clusters
*
* Interface to the MPI enabled MKL version of Pardiso
*/
class CPardisoSolver : public Solver
{
public:
enum MatType
{
REAL_STRUCTURE_SYMMETRIC = 1,
REAL_UNSYMMETRIC = 11
};
/**
* @brief Construct a new CPardisoSolver object
*
* @param comm MPI Communicator
*/
CPardisoSolver(MPI_Comm comm);
/**
* @brief Set the Operator object and perform factorization
*
* @a op needs to be of type HypreParMatrix. The contents are copied and
* reordered in an internal CSR structure.
*
* @param op Operator to use in factorization and solve
*/
void SetOperator(const Operator &op) override;
/**
* @brief Solve
*
* @param b RHS vector
* @param x Solution vector
*/
void Mult(const Vector &b, Vector &x) const override;
/**
* @brief Set the print level for Pardiso
*
* Prints statistics after the factorization and after each solve.
*
* @param print_lvl Print level
*/
void SetPrintLevel(int print_lvl);
/**
* @brief Set the matrix type
*
* The matrix type supported is either real and symmetric or real and
* non-symmetric.
*
* @param mat_type Matrix type
*/
void SetMatrixType(MatType mat_type);
~CPardisoSolver();
private:
MPI_Comm comm_;
// Global number of rows
int m;
// First row index of the global matrix on the local MPI rank
int first_row;
// Local number of nonzero entries
int nnz_loc;
// Local number of rows, obtained from a ParCSR matrix
int m_loc;
// CSR data structure for the copy data of the local CSR matrix
int *csr_rowptr = nullptr;
double *reordered_csr_nzval = nullptr;
int *reordered_csr_colind = nullptr;
// Internal solver memory pointer pt,
// 32-bit: int pt[64]
// 64-bit: long int pt[64] or void *pt[64] should be OK on both architectures
mutable void *pt[64] = {0};
// Solver control parameters, detailed description can be found in the
// constructor.
mutable int iparm[64] = {0};
mutable int maxfct, mnum, msglvl, phase, error;
int mtype;
int nrhs;
// Dummy variables
mutable int idum;
mutable double ddum;
};
} // namespace mfem
#endif
#endif // MFEM_USE_MKL_CPARDISO
#endif // MFEM_USE_MPI
+26 -26
View File
@@ -373,7 +373,7 @@ void DenseMatrix::SymmetricScaling(const Vector & s)
{
if (height != width || s.Size() != height)
{
mfem_error("DenseMatrix::SymmetricScaling: dimension mismatch");
mfem_error("DenseMatrix::SymmetricScaling");
}
double * ss = new double[width];
@@ -401,7 +401,7 @@ void DenseMatrix::InvSymmetricScaling(const Vector & s)
{
if (height != width || s.Size() != width)
{
mfem_error("DenseMatrix::InvSymmetricScaling: dimension mismatch");
mfem_error("DenseMatrix::SymmetricScaling");
}
double * ss = new double[width];
@@ -528,7 +528,7 @@ double DenseMatrix::Weight() const
double F = d[0] * d[3] + d[1] * d[4] + d[2] * d[5];
return sqrt(E * G - F * F);
}
mfem_error("DenseMatrix::Weight(): mismatched or unsupported dimensions");
mfem_error("DenseMatrix::Weight()");
return 0.0;
}
@@ -639,7 +639,7 @@ void DenseMatrix::Invert()
#ifdef MFEM_DEBUG
if (Height() <= 0 || Height() != Width())
{
mfem_error("DenseMatrix::Invert(): dimension mismatch");
mfem_error("DenseMatrix::Invert()");
}
#endif
@@ -1083,7 +1083,7 @@ void DenseMatrix::Eigensystem(Vector &ev, DenseMatrix *evect)
MFEM_CONTRACT_VAR(ev);
MFEM_CONTRACT_VAR(evect);
mfem_error("DenseMatrix::Eigensystem: Compiled without LAPACK");
mfem_error("DenseMatrix::Eigensystem");
#endif
}
@@ -1164,7 +1164,7 @@ void DenseMatrix::Eigensystem(DenseMatrix &b, Vector &ev,
MFEM_CONTRACT_VAR(b);
MFEM_CONTRACT_VAR(ev);
MFEM_CONTRACT_VAR(evect);
mfem_error("DenseMatrix::Eigensystem(generalized): Compiled without LAPACK");
mfem_error("DenseMatrix::Eigensystem for generalized eigenvalues");
#endif
}
@@ -1204,7 +1204,7 @@ void DenseMatrix::SingularValues(Vector &sv) const
#else
MFEM_CONTRACT_VAR(sv);
// compiling without lapack
mfem_error("DenseMatrix::SingularValues: Compiled without LAPACK");
mfem_error("DenseMatrix::SingularValues");
#endif
}
@@ -1441,7 +1441,7 @@ void DenseMatrix::GradToCurl(DenseMatrix &curl)
if ((Width() != 2 || curl.Width() != 1 || 2*n != curl.Height()) &&
(Width() != 3 || curl.Width() != 3 || 3*n != curl.Height()))
{
mfem_error("DenseMatrix::GradToCurl(...): dimension mismatch");
mfem_error("DenseMatrix::GradToCurl(...)");
}
#endif
@@ -1676,7 +1676,7 @@ void DenseMatrix::AddMatrix(DenseMatrix &A, int ro, int co)
#ifdef MFEM_DEBUG
if (co+aw > Width() || ro+ah > h)
{
mfem_error("DenseMatrix::AddMatrix(...) 1 : dimension mismatch");
mfem_error("DenseMatrix::AddMatrix(...) 1");
}
#endif
@@ -1706,7 +1706,7 @@ void DenseMatrix::AddMatrix(double a, const DenseMatrix &A, int ro, int co)
#ifdef MFEM_DEBUG
if (co+aw > Width() || ro+ah > h)
{
mfem_error("DenseMatrix::AddMatrix(...) 2 : dimension mismatch");
mfem_error("DenseMatrix::AddMatrix(...) 2");
}
#endif
@@ -1753,7 +1753,7 @@ void DenseMatrix::AdjustDofDirection(Array<int> &dofs)
#ifdef MFEM_DEBUG
if (dofs.Size() != n || Width() != n)
{
mfem_error("DenseMatrix::AdjustDofDirection(...): dimension mismatch");
mfem_error("DenseMatrix::AdjustDofDirection(...)");
}
#endif
@@ -2093,11 +2093,11 @@ void CalcAdjugate(const DenseMatrix &a, DenseMatrix &adja)
#ifdef MFEM_DEBUG
if (a.Width() > a.Height() || a.Width() < 1 || a.Height() > 3)
{
mfem_error("CalcAdjugate(...): unsupported dimensions");
mfem_error("CalcAdjugate(...)");
}
if (a.Width() != adja.Height() || a.Height() != adja.Width())
{
mfem_error("CalcAdjugate(...): dimension mismatch");
mfem_error("CalcAdjugate(...)");
}
#endif
@@ -2166,7 +2166,7 @@ void CalcAdjugateTranspose(const DenseMatrix &a, DenseMatrix &adjat)
if (a.Height() != a.Width() || adjat.Height() != adjat.Width() ||
a.Width() != adjat.Width() || a.Width() < 1 || a.Width() > 3)
{
mfem_error("CalcAdjugateTranspose(...): dimension mismatch");
mfem_error("CalcAdjugateTranspose(...)");
}
#endif
if (a.Width() == 1)
@@ -2269,7 +2269,7 @@ void CalcInverseTranspose(const DenseMatrix &a, DenseMatrix &inva)
if ( (a.Width() != a.Height()) || ( (a.Height()!= 1) && (a.Height()!= 2)
&& (a.Height()!= 3) ) )
{
mfem_error("CalcInverseTranspose(...): dimension mismatch");
mfem_error("CalcInverseTranspose(...)");
}
#endif
@@ -2396,7 +2396,7 @@ void MultABt(const DenseMatrix &A, const DenseMatrix &B, DenseMatrix &ABt)
if (A.Height() != ABt.Height() || B.Height() != ABt.Width() ||
A.Width() != B.Width())
{
mfem_error("MultABt(...): dimension mismatch");
mfem_error("MultABt(...)");
}
#endif
@@ -2462,7 +2462,7 @@ void MultADBt(const DenseMatrix &A, const Vector &D,
if (A.Height() != ADBt.Height() || B.Height() != ADBt.Width() ||
A.Width() != B.Width() || A.Width() != D.Size())
{
mfem_error("MultADBt(...): dimension mismatch");
mfem_error("MultADBt(...)");
}
#endif
@@ -2501,7 +2501,7 @@ void AddMultABt(const DenseMatrix &A, const DenseMatrix &B, DenseMatrix &ABt)
if (A.Height() != ABt.Height() || B.Height() != ABt.Width() ||
A.Width() != B.Width())
{
mfem_error("AddMultABt(...): dimension mismatch");
mfem_error("AddMultABt(...)");
}
#endif
@@ -2559,7 +2559,7 @@ void AddMultADBt(const DenseMatrix &A, const Vector &D,
if (A.Height() != ADBt.Height() || B.Height() != ADBt.Width() ||
A.Width() != B.Width() || A.Width() != D.Size())
{
mfem_error("AddMultADBt(...): dimension mismatch");
mfem_error("AddMultADBt(...)");
}
#endif
@@ -2595,7 +2595,7 @@ void AddMult_a_ABt(double a, const DenseMatrix &A, const DenseMatrix &B,
if (A.Height() != ABt.Height() || B.Height() != ABt.Width() ||
A.Width() != B.Width())
{
mfem_error("AddMult_a_ABt(...): dimension mismatch");
mfem_error("AddMult_a_ABt(...)");
}
#endif
@@ -2653,7 +2653,7 @@ void MultAtB(const DenseMatrix &A, const DenseMatrix &B, DenseMatrix &AtB)
if (A.Width() != AtB.Height() || B.Width() != AtB.Width() ||
A.Height() != B.Height())
{
mfem_error("MultAtB(...): dimension mismatch");
mfem_error("MultAtB(...)");
}
#endif
@@ -2761,7 +2761,7 @@ void MultVWt(const Vector &v, const Vector &w, DenseMatrix &VWt)
#ifdef MFEM_DEBUG
if (v.Size() != VWt.Height() || w.Size() != VWt.Width())
{
mfem_error("MultVWt(...): dimension mismatch");
mfem_error("MultVWt(...)");
}
#endif
@@ -2782,7 +2782,7 @@ void AddMultVWt(const Vector &v, const Vector &w, DenseMatrix &VWt)
#ifdef MFEM_DEBUG
if (VWt.Height() != m || VWt.Width() != n)
{
mfem_error("AddMultVWt(...): dimension mismatch");
mfem_error("AddMultVWt(...)");
}
#endif
@@ -2803,7 +2803,7 @@ void AddMultVVt(const Vector &v, DenseMatrix &VVt)
#ifdef MFEM_DEBUG
if (VVt.Height() != n || VVt.Width() != n)
{
mfem_error("AddMultVVt(...): dimension mismatch");
mfem_error("AddMultVVt(...)");
}
#endif
@@ -2828,7 +2828,7 @@ void AddMult_a_VWt(const double a, const Vector &v, const Vector &w,
#ifdef MFEM_DEBUG
if (VWt.Height() != m || VWt.Width() != n)
{
mfem_error("AddMult_a_VWt(...): dimension mismatch");
mfem_error("AddMult_a_VWt(...)");
}
#endif
@@ -3353,7 +3353,7 @@ void DenseMatrixEigensystem::Eval()
#ifdef MFEM_DEBUG
if (mat.Width() != n)
{
mfem_error("DenseMatrixEigensystem::Eval(): dimension mismatch");
mfem_error("DenseMatrixEigensystem::Eval()");
}
#endif
+1 -52
View File
@@ -1048,36 +1048,6 @@ HYPRE_Int HypreParMatrix::MultTranspose(HypreParVector & x, HypreParVector & y,
return hypre_ParCSRMatrixMatvecT(a, A, x, b, y);
}
void HypreParMatrix::AbsMult(double a, const Vector &x,
double b, Vector &y) const
{
MFEM_ASSERT(x.Size() == Width(), "invalid x.Size() = " << x.Size()
<< ", expected size = " << Width());
MFEM_ASSERT(y.Size() == Height(), "invalid y.Size() = " << y.Size()
<< ", expected size = " << Height());
auto x_data = x.HostRead();
auto y_data = (b == 0.0) ? y.HostWrite() : y.HostReadWrite();
internal::hypre_ParCSRMatrixAbsMatvec(A, a, const_cast<double*>(x_data),
b, y_data);
}
void HypreParMatrix::AbsMultTranspose(double a, const Vector &x,
double b, Vector &y) const
{
MFEM_ASSERT(x.Size() == Height(), "invalid x.Size() = " << x.Size()
<< ", expected size = " << Height());
MFEM_ASSERT(y.Size() == Width(), "invalid y.Size() = " << y.Size()
<< ", expected size = " << Width());
auto x_data = x.HostRead();
auto y_data = (b == 0.0) ? y.HostWrite() : y.HostReadWrite();
internal::hypre_ParCSRMatrixAbsMatvecT(A, a, const_cast<double*>(x_data),
b, y_data);
}
HypreParMatrix* HypreParMatrix::LeftDiagMult(const SparseMatrix &D,
HYPRE_Int* row_starts) const
{
@@ -3215,31 +3185,10 @@ void HypreBoomerAMG::SetOperator(const Operator &op)
B = X = NULL;
}
void HypreBoomerAMG::SetSystemsOptions(int dim, bool order_bynodes)
void HypreBoomerAMG::SetSystemsOptions(int dim)
{
HYPRE_BoomerAMGSetNumFunctions(amg_precond, dim);
// The default "system" ordering in hypre is Ordering::byVDIM. When we are
// using Ordering::byNODES, we have to specify the ordering explicitly with
// HYPRE_BoomerAMGSetDofFunc as in the following code.
if (order_bynodes)
{
// hypre actually deletes the following pointer in HYPRE_BoomerAMGDestroy,
// so we don't need to track it
HYPRE_Int *mapping = mfem_hypre_CTAlloc(HYPRE_Int, height);
int h_nnodes = height / dim; // nodes owned in linear algebra (not fem)
MFEM_VERIFY(height % dim == 0, "Ordering does not work as claimed!");
int k = 0;
for (int i = 0; i < dim; ++i)
{
for (int j = 0; j < h_nnodes; ++j)
{
mapping[k++] = i;
}
}
HYPRE_BoomerAMGSetDofFunc(amg_precond, mapping);
}
// More robust options with respect to convergence
HYPRE_BoomerAMGSetAggNumLevels(amg_precond, 0);
HYPRE_BoomerAMGSetStrongThreshold(amg_precond, 0.5);
+5 -10
View File
@@ -446,12 +446,6 @@ public:
virtual void MultTranspose(const Vector &x, Vector &y) const
{ MultTranspose(1.0, x, 0.0, y); }
/// Computes y = a * |A| * x + b * y, using entry-wise absolute values of matrix A
void AbsMult(double a, const Vector &x, double b, Vector &y) const;
/// Computes y = a * |At| * x + b * y, using entry-wise absolute values of the transpose of matrix A
void AbsMultTranspose(double a, const Vector &x, double b, Vector &y) const;
/** The "Boolean" analog of y = alpha * A * x + beta * y, where elements in
the sparsity pattern of the matrix are treated as "true". */
void BooleanMult(int alpha, const int *x, int beta, int *y)
@@ -992,15 +986,16 @@ public:
virtual void SetOperator(const Operator &op);
/** More robust options for systems, such as elasticity. */
void SetSystemsOptions(int dim, bool order_bynodes=false);
/** More robust options for systems, such as elasticity. Note that BoomerAMG
assumes Ordering::byVDIM in the finite element space used to generate the
matrix A. */
void SetSystemsOptions(int dim);
/** A special elasticity version of BoomerAMG that takes advantage of
geometric rigid body modes and could perform better on some problems, see
"Improving algebraic multigrid interpolation operators for linear
elasticity problems", Baker, Kolev, Yang, NLAA 2009, DOI:10.1002/nla.688.
This solver assumes Ordering::byVDIM in the FiniteElementSpace used to
construct A. */
As with SetSystemsOptions(), this solver assumes Ordering::byVDIM. */
void SetElasticityOptions(ParFiniteElementSpace *fespace);
void SetPrintLevel(int print_level)
-328
View File
@@ -16,7 +16,6 @@
#include "hypre_parcsr.hpp"
#include <limits>
#include <cmath>
namespace mfem
{
@@ -978,196 +977,6 @@ void hypre_ParCSRMatrixSplit(hypre_ParCSRMatrix *A,
}
}
/* Based on hypre_CSRMatrixMatvec in hypre's csr_matvec.c */
void hypre_CSRMatrixAbsMatvec(hypre_CSRMatrix *A,
HYPRE_Real alpha,
HYPRE_Real *x,
HYPRE_Real beta,
HYPRE_Real *y)
{
HYPRE_Real *A_data = hypre_CSRMatrixData(A);
HYPRE_Int *A_i = hypre_CSRMatrixI(A);
HYPRE_Int *A_j = hypre_CSRMatrixJ(A);
HYPRE_Int num_rows = hypre_CSRMatrixNumRows(A);
HYPRE_Int *A_rownnz = hypre_CSRMatrixRownnz(A);
HYPRE_Int num_rownnz = hypre_CSRMatrixNumRownnz(A);
HYPRE_Real *x_data = x;
HYPRE_Real *y_data = y;
HYPRE_Real temp, tempx;
HYPRE_Int i, jj;
HYPRE_Int m;
HYPRE_Real xpar=0.7;
/*-----------------------------------------------------------------------
* Do (alpha == 0.0) computation - RDF: USE MACHINE EPS
*-----------------------------------------------------------------------*/
if (alpha == 0.0)
{
for (i = 0; i < num_rows; i++)
{
y_data[i] *= beta;
}
return;
}
/*-----------------------------------------------------------------------
* y = (beta/alpha)*y
*-----------------------------------------------------------------------*/
temp = beta / alpha;
if (temp != 1.0)
{
if (temp == 0.0)
{
for (i = 0; i < num_rows; i++)
{
y_data[i] = 0.0;
}
}
else
{
for (i = 0; i < num_rows; i++)
{
y_data[i] *= temp;
}
}
}
/*-----------------------------------------------------------------
* y += abs(A)*x
*-----------------------------------------------------------------*/
/* use rownnz pointer to do the abs(A)*x multiplication
when num_rownnz is smaller than num_rows */
if (num_rownnz < xpar*(num_rows))
{
for (i = 0; i < num_rownnz; i++)
{
m = A_rownnz[i];
tempx = 0;
for (jj = A_i[m]; jj < A_i[m+1]; jj++)
{
tempx += std::abs(A_data[jj])*x_data[A_j[jj]];
}
y_data[m] += tempx;
}
}
else
{
for (i = 0; i < num_rows; i++)
{
tempx = 0;
for (jj = A_i[i]; jj < A_i[i+1]; jj++)
{
tempx += std::abs(A_data[jj])*x_data[A_j[jj]];
}
y_data[i] += tempx;
}
}
/*-----------------------------------------------------------------
* y = alpha*y
*-----------------------------------------------------------------*/
if (alpha != 1.0)
{
for (i = 0; i < num_rows; i++)
{
y_data[i] *= alpha;
}
}
}
/* Based on hypre_CSRMatrixMatvecT in hypre's csr_matvec.c */
void hypre_CSRMatrixAbsMatvecT(hypre_CSRMatrix *A,
HYPRE_Real alpha,
HYPRE_Real *x,
HYPRE_Real beta,
HYPRE_Real *y)
{
HYPRE_Real *A_data = hypre_CSRMatrixData(A);
HYPRE_Int *A_i = hypre_CSRMatrixI(A);
HYPRE_Int *A_j = hypre_CSRMatrixJ(A);
HYPRE_Int num_rows = hypre_CSRMatrixNumRows(A);
HYPRE_Int num_cols = hypre_CSRMatrixNumCols(A);
HYPRE_Real *x_data = x;
HYPRE_Real *y_data = y;
HYPRE_Int i, j, jj;
HYPRE_Real temp;
if (alpha == 0.0)
{
for (i = 0; i < num_cols; i++)
{
y_data[i] *= beta;
}
return;
}
/*-----------------------------------------------------------------------
* y = (beta/alpha)*y
*-----------------------------------------------------------------------*/
temp = beta / alpha;
if (temp != 1.0)
{
if (temp == 0.0)
{
for (i = 0; i < num_cols; i++)
{
y_data[i] = 0.0;
}
}
else
{
for (i = 0; i < num_cols; i++)
{
y_data[i] *= temp;
}
}
}
/*-----------------------------------------------------------------
* y += abs(A)^T*x
*-----------------------------------------------------------------*/
for (i = 0; i < num_rows; i++)
{
for (jj = A_i[i]; jj < A_i[i+1]; jj++)
{
j = A_j[jj];
y_data[j] += std::abs(A_data[jj]) * x_data[i];
}
}
/*-----------------------------------------------------------------
* y = alpha*y
*-----------------------------------------------------------------*/
if (alpha != 1.0)
{
for (i = 0; i < num_cols; i++)
{
y_data[i] *= alpha;
}
}
}
/* Based on hypre_CSRMatrixMatvec in hypre's csr_matvec.c */
void hypre_CSRMatrixBooleanMatvec(hypre_CSRMatrix *A,
HYPRE_Bool alpha,
@@ -1427,143 +1236,6 @@ hypre_ParCSRCommHandleCreate_bool(HYPRE_Int job,
return comm_handle;
}
/* Based on hypre_ParCSRMatrixMatvec in par_csr_matvec.c */
void hypre_ParCSRMatrixAbsMatvec(hypre_ParCSRMatrix *A,
HYPRE_Real alpha,
HYPRE_Real *x,
HYPRE_Real beta,
HYPRE_Real *y)
{
hypre_ParCSRCommHandle *comm_handle;
hypre_ParCSRCommPkg *comm_pkg = hypre_ParCSRMatrixCommPkg(A);
hypre_CSRMatrix *diag = hypre_ParCSRMatrixDiag(A);
hypre_CSRMatrix *offd = hypre_ParCSRMatrixOffd(A);
HYPRE_Int num_cols_offd = hypre_CSRMatrixNumCols(offd);
HYPRE_Int num_sends, i, j, index;
HYPRE_Real *x_tmp, *x_buf;
x_tmp = mfem_hypre_CTAlloc(HYPRE_Real, num_cols_offd);
/*---------------------------------------------------------------------
* If there exists no CommPkg for A, a CommPkg is generated using
* equally load balanced partitionings
*--------------------------------------------------------------------*/
if (!comm_pkg)
{
hypre_MatvecCommPkgCreate(A);
comm_pkg = hypre_ParCSRMatrixCommPkg(A);
}
num_sends = hypre_ParCSRCommPkgNumSends(comm_pkg);
x_buf = mfem_hypre_CTAlloc(
HYPRE_Real, hypre_ParCSRCommPkgSendMapStart(comm_pkg, num_sends));
index = 0;
for (i = 0; i < num_sends; i++)
{
j = hypre_ParCSRCommPkgSendMapStart(comm_pkg, i);
for ( ; j < hypre_ParCSRCommPkgSendMapStart(comm_pkg, i+1); j++)
{
x_buf[index++] = x[hypre_ParCSRCommPkgSendMapElmt(comm_pkg, j)];
}
}
comm_handle = hypre_ParCSRCommHandleCreate(1, comm_pkg, x_buf, x_tmp);
hypre_CSRMatrixAbsMatvec(diag, alpha, x, beta, y);
hypre_ParCSRCommHandleDestroy(comm_handle);
if (num_cols_offd)
{
hypre_CSRMatrixAbsMatvec(offd, alpha, x_tmp, 1.0, y);
}
mfem_hypre_TFree(x_buf);
mfem_hypre_TFree(x_tmp);
}
/* Based on hypre_ParCSRMatrixMatvecT in par_csr_matvec.c */
void hypre_ParCSRMatrixAbsMatvecT(hypre_ParCSRMatrix *A,
HYPRE_Real alpha,
HYPRE_Real *x,
HYPRE_Real beta,
HYPRE_Real *y)
{
hypre_ParCSRCommHandle *comm_handle;
hypre_ParCSRCommPkg *comm_pkg = hypre_ParCSRMatrixCommPkg(A);
hypre_CSRMatrix *diag = hypre_ParCSRMatrixDiag(A);
hypre_CSRMatrix *offd = hypre_ParCSRMatrixOffd(A);
HYPRE_Real *y_tmp;
HYPRE_Real *y_buf;
HYPRE_Int num_cols_offd = hypre_CSRMatrixNumCols(offd);
HYPRE_Int i, j, jj, end, num_sends;
y_tmp = mfem_hypre_TAlloc(HYPRE_Real, num_cols_offd);
/*---------------------------------------------------------------------
* If there exists no CommPkg for A, a CommPkg is generated using
* equally load balanced partitionings
*--------------------------------------------------------------------*/
if (!comm_pkg)
{
hypre_MatvecCommPkgCreate(A);
comm_pkg = hypre_ParCSRMatrixCommPkg(A);
}
num_sends = hypre_ParCSRCommPkgNumSends(comm_pkg);
y_buf = mfem_hypre_CTAlloc(
HYPRE_Real, hypre_ParCSRCommPkgSendMapStart(comm_pkg, num_sends));
if (num_cols_offd)
{
#if MFEM_HYPRE_VERSION >= 21100
if (A->offdT)
{
// offdT is optional. Used only if it's present.
hypre_CSRMatrixAbsMatvec(A->offdT, alpha, x, 0., y_tmp);
}
else
#endif
{
hypre_CSRMatrixAbsMatvecT(offd, alpha, x, 0., y_tmp);
}
}
comm_handle = hypre_ParCSRCommHandleCreate(2, comm_pkg, y_tmp, y_buf);
#if MFEM_HYPRE_VERSION >= 21100
if (A->diagT)
{
// diagT is optional. Used only if it's present.
hypre_CSRMatrixAbsMatvec(A->diagT, alpha, x, beta, y);
}
else
#endif
{
hypre_CSRMatrixAbsMatvecT(diag, alpha, x, beta, y);
}
hypre_ParCSRCommHandleDestroy(comm_handle);
for (i = 0; i < num_sends; i++)
{
end = hypre_ParCSRCommPkgSendMapStart(comm_pkg, i+1);
for (j = hypre_ParCSRCommPkgSendMapStart(comm_pkg, i); j < end; j++)
{
jj = hypre_ParCSRCommPkgSendMapElmt(comm_pkg, j);
y[jj] += y_buf[j];
}
}
mfem_hypre_TFree(y_buf);
mfem_hypre_TFree(y_tmp);
}
/* Based on hypre_ParCSRMatrixMatvec in par_csr_matvec.c */
void hypre_ParCSRMatrixBooleanMatvec(hypre_ParCSRMatrix *A,
HYPRE_Bool alpha,
-28
View File
@@ -118,34 +118,6 @@ void hypre_ParCSRMatrixSplit(hypre_ParCSRMatrix *A,
typedef int HYPRE_Bool;
#define HYPRE_MPI_BOOL MPI_INT
/// Computes y = alpha * |A| * x + beta * y, using entry-wise absolute values of matrix A
void hypre_CSRMatrixAbsMatvec(hypre_CSRMatrix *A,
HYPRE_Real alpha,
HYPRE_Real *x,
HYPRE_Real beta,
HYPRE_Real *y);
/// Computes y = alpha * |At| * x + beta * y, using entry-wise absolute values of the transpose of matrix A
void hypre_CSRMatrixAbsMatvecT(hypre_CSRMatrix *A,
HYPRE_Real alpha,
HYPRE_Real *x,
HYPRE_Real beta,
HYPRE_Real *y);
/// Computes y = alpha * |A| * x + beta * y, using entry-wise absolute values of matrix A
void hypre_ParCSRMatrixAbsMatvec(hypre_ParCSRMatrix *A,
HYPRE_Real alpha,
HYPRE_Real *x,
HYPRE_Real beta,
HYPRE_Real *y);
/// Computes y = alpha * |At| * x + beta * y, using entry-wise absolute values of the transpose of matrix A
void hypre_ParCSRMatrixAbsMatvecT(hypre_ParCSRMatrix *A,
HYPRE_Real alpha,
HYPRE_Real *x,
HYPRE_Real beta,
HYPRE_Real *y);
/** The "Boolean" analog of y = alpha * A * x + beta * y, where elements in the
sparsity pattern of the CSR matrix A are treated as "true". */
void hypre_CSRMatrixBooleanMatvec(hypre_CSRMatrix *A,
-8
View File
@@ -45,10 +45,6 @@
#include "hypre_parcsr.hpp"
#include "hypre.hpp"
#ifdef MFEM_USE_MUMPS
#include "mumps.hpp"
#endif
#ifdef MFEM_USE_PETSC
#include "petsc.hpp"
#endif
@@ -65,10 +61,6 @@
#include "strumpack.hpp"
#endif
#ifdef MFEM_USE_MKL_CPARDISO
#include "cpardiso.hpp"
#endif
#endif // MFEM_USE_MPI
#endif
-387
View File
@@ -1,387 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../config/config.hpp"
#ifdef MFEM_USE_MUMPS
#ifdef MFEM_USE_MPI
#include "mumps.hpp"
namespace mfem
{
MUMPSSolver::~MUMPSSolver()
{
if (id)
{
id->job = -2;
dmumps_c(id);
delete[] J;
delete[] I;
delete [] data;
}
}
void MUMPSSolver::SetParameters()
{
// output messages
id->ICNTL(1) = -1;
// Diagnosting printing
id->ICNTL(2) = -1;
// Global info on host
id->ICNTL(3) = -1;
// Level of error printing
id->ICNTL(4) = 0;
//input matrix format (assembled)
id->ICNTL(5) = 0;
// Use A or A^T
id->ICNTL(9) = 1;
// Iterative refinement (disabled)
id->ICNTL(10) = 0;
// Error analysis-statistics (disabled)
id->ICNTL(11) = 0;
// Use of ScaLAPACK (Parallel factorization on root)
id->ICNTL(13) = 0;
// Percentage increase of estimated workspace (default = 20%)
id->ICNTL(14) = 20;
// Number of OpenMP threads (default)
id->ICNTL(16) = 0;
// Matrix input format (distributed)
id->ICNTL(18) = 3;
// Schur complement (no Schur complement matrix returned)
id->ICNTL(19) = 0;
#if MFEM_MUMPS_VERSION >= 530
// Distributed RHS and Sol
id->ICNTL(20) = 10;
id->ICNTL(21) = 1;
#else
// Centralized RHS and Sol
id->ICNTL(20) = 0;
id->ICNTL(21) = 0;
#endif
// Out of core factorization and solve (disabled)
id->ICNTL(22) = 0;
// Max size of working memory (default = based on estimates)
id->ICNTL(23) = 0;
}
void MUMPSSolver::SetOperator(const Operator &op)
{
// Verify that the operator is a HypreParMatrix
auto APtr = dynamic_cast<const HypreParMatrix *>(&op);
MFEM_VERIFY(APtr, "Not compatible matrix type");
height = op.Height();
width = op.Width();
comm = APtr->GetComm();
MPI_Comm_size(comm, &numProcs);
MPI_Comm_rank(comm, &myid);
hypre_ParCSRMatrix *parcsr_op
= (hypre_ParCSRMatrix *) const_cast<HypreParMatrix &>(*APtr);
hypre_CSRMatrix *csr_op = hypre_MergeDiagAndOffd(parcsr_op);
#if MFEM_HYPRE_VERSION >= 21600
hypre_CSRMatrixBigJtoJ(csr_op);
#endif
int *Iptr = csr_op->i;
int *Jptr = csr_op->j;
int n_loc = csr_op->num_rows;
row_start = parcsr_op->first_row_index;
int nnz;
if (sym)
{
// count nnz;
nnz = 0;
int k = 0;
for (int i = 0; i < n_loc; i++)
{
for (int j = Iptr[i]; j < Iptr[i + 1]; j++)
{
int ii = row_start + i + 1;
int jj = Jptr[k] + 1;
k++;
if (ii>=jj) { nnz++; }
}
}
}
else
{
nnz = csr_op->num_nonzeros;
}
I = new int[nnz];
J = new int[nnz];
int k = 0;
if (sym)
{
int l = 0;
data = new double[nnz];
for (int i = 0; i < n_loc; i++)
{
for (int j = Iptr[i]; j < Iptr[i + 1]; j++)
{
// Global I and J indices in 1-based index (for fortran)
int ii = row_start + i + 1;
int jj = Jptr[k] + 1;
if (ii>=jj)
{
I[l] = ii;
J[l] = jj;
data[l++] = csr_op->data[k];
}
k++;
}
}
}
else
{
for (int i = 0; i < n_loc; i++)
{
for (int j = Iptr[i]; j < Iptr[i + 1]; j++)
{
// Global I and J indices in 1-based index (for fortran)
I[k] = row_start + i + 1;
J[k] = Jptr[k] + 1;
k++;
}
}
data = csr_op->data;
}
// new MUMPS object
id = new DMUMPS_STRUC_C;
// C to Fortran communicator
id->comm_fortran = (MUMPS_INT) MPI_Comm_c2f(comm);
// Host is involved in computation
id->par = 1;
// Unsymmetric matrix
id->sym = sym;
// Mumps init
id->job = -1;
dmumps_c(id);
// Set MUMPS default parameters
SetParameters();
// Global number of rows
id->n = parcsr_op->global_num_rows;
// Number of non zeros on the processor
id->nnz_loc = nnz;
// Distributed row array
id->irn_loc = I;
// Distributed column array
id->jcn_loc = J;
// Distributed data array
id->a_loc = data;
// MUMPS Analysis
id->job = 1;
dmumps_c(id);
// MUMPS Factorization
id->job = 2;
dmumps_c(id);
// matrix can be destroyed now
hypre_CSRMatrixDestroy(csr_op);
if (!sym) { data = nullptr; }
#if MFEM_MUMPS_VERSION >= 530
irhs_loc.SetSize(n_loc);
for (int i = 0; i < n_loc; i++)
{
irhs_loc[i] = row_start + i + 1;
}
row_starts.SetSize(numProcs);
MPI_Allgather(&row_start, 1, MPI_INT, row_starts, 1, MPI_INT, comm);
sol_loc.SetSize(id->INFO(23));
isol_loc.SetSize(id->INFO(23));
#else
if (myid == 0)
{
rhs_glob.SetSize(parcsr_op->global_num_rows);
recv_counts.SetSize(numProcs);
}
MPI_Gather(&n_loc, 1, MPI_INT, recv_counts, 1, MPI_INT, 0, comm);
if (myid == 0)
{
displs.SetSize(numProcs); displs[0] = 0;
int s = 0;
for (int k = 0; k < numProcs-1; k++)
{
s += recv_counts[k];
displs[k+1] = s;
}
}
#endif
}
void MUMPSSolver::Mult(const Vector &x, Vector &y) const
{
#if MFEM_MUMPS_VERSION >= 530
id->nloc_rhs = x.Size();
id->lrhs_loc = x.Size();
id->rhs_loc = x.GetData();
id->irhs_loc = const_cast<int *>(irhs_loc.GetData());
id->sol_loc = sol_loc.GetData();
id->lsol_loc = id->INFO(23);
id->isol_loc = const_cast<int *>(isol_loc.GetData());
id->job = 3;
dmumps_c(id);
RedistributeSol(isol_loc, sol_loc, y);
#else
MPI_Gatherv(x.GetData(), x.Size(), MPI_DOUBLE,
rhs_glob.GetData(), recv_counts,
displs, MPI_DOUBLE, 0, comm);
if (myid == 0)
{
id->rhs = rhs_glob.GetData();
}
id->job = 3;
dmumps_c(id);
MPI_Scatterv(rhs_glob.GetData(), recv_counts, displs,
MPI_DOUBLE, y.GetData(), y.Size(),
MPI_DOUBLE, 0, comm);
#endif
}
void MUMPSSolver::MultTranspose(const Vector &x, Vector &y) const
{
id->ICNTL(9) = 0;
Mult(x,y);
}
#if MFEM_MUMPS_VERSION >= 530
int MUMPSSolver::GetRowRank(int i, const Array<int> &row_starts_) const
{
if (row_starts_.Size() == 1)
{
return 0;
}
auto up = std::upper_bound(row_starts_.begin(), row_starts_.end(), i);
return std::distance(row_starts_.begin(), up) - 1;
}
void MUMPSSolver::RedistributeSol(const Array<int> &row_map,
const Vector &x,
Vector &y) const
{
MFEM_VERIFY(row_map.Size() == x.Size(), "Inconcistent sizes");
int size = x.Size();
// compute send_count
Array<int> send_count(numProcs);
send_count = 0;
for (int i = 0; i < size; i++)
{
int j = row_map[i] - 1; //fix to 0-based indexing
int row_rank = GetRowRank(j, row_starts);
send_count[row_rank]++; // both for val and global index
}
// compute recv_count
Array<int> recv_count(numProcs);
MPI_Alltoall(send_count, 1, MPI_INT, recv_count, 1, MPI_INT, comm);
// compute offsets
Array<int> send_displ(numProcs);
send_displ[0] = 0;
Array<int> recv_displ(numProcs);
recv_displ[0] = 0;
for (int k = 0; k < numProcs - 1; k++)
{
send_displ[k + 1] = send_displ[k] + send_count[k];
recv_displ[k + 1] = recv_displ[k] + recv_count[k];
}
int sbuff_size = send_count.Sum();
int rbuff_size = recv_count.Sum();
Array<int> sendbuf_index(sbuff_size);
sendbuf_index = 0;
Array<double> sendbuf_value(sbuff_size);
sendbuf_value = 0;
Array<int> soffs(numProcs);
soffs = 0;
// Fill in send buffers
for (int i = 0; i < size; i++)
{
int j = row_map[i] - 1; //fix to 0-based indexing
int row_rank = GetRowRank(j, row_starts);
int k = send_displ[row_rank] + soffs[row_rank];
sendbuf_index[k] = j;
sendbuf_value[k] = x(i);
soffs[row_rank]++;
}
// communicate
Array<int> recvbuf_index(rbuff_size);
Array<double> recvbuf_value(rbuff_size);
MPI_Alltoallv(sendbuf_index,
send_count,
send_displ,
MPI_INT,
recvbuf_index,
recv_count,
recv_displ,
MPI_INT,
comm);
MPI_Alltoallv(sendbuf_value,
send_count,
send_displ,
MPI_DOUBLE,
recvbuf_value,
recv_count,
recv_displ,
MPI_DOUBLE,
comm);
// Unpack recv buffer
for (int i = 0; i < rbuff_size; i++)
{
int local_index = recvbuf_index[i] - row_start;
y(local_index) = recvbuf_value[i];
}
}
#endif
} // namespace mfem
#endif // MFEM_USE_MPI
#endif // MFEM_USE_MUMPS
-104
View File
@@ -1,104 +0,0 @@
// Copyright (c) 2010-2020, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_MUMPS
#define MFEM_MUMPS
#include "../config/config.hpp"
#ifdef MFEM_USE_MUMPS
#ifdef MFEM_USE_MPI
#include "operator.hpp"
#include "hypre.hpp"
#include <mpi.h>
#include "dmumps_c.h"
#include <vector>
namespace mfem
{
class MUMPSSolver : public mfem::Solver
{
public:
// Default Constructor.
MUMPSSolver() {}
void SetMatrixSymType(int sym_) { sym = (sym_>2) ? 0 : sym_ ; }
// Factor and solve the linear system y = Op^{-1} x.
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const;
// Set the operator.
void SetOperator(const Operator &op);
// Default destructor.
~MUMPSSolver();
private:
MPI_Comm comm;
int numProcs;
int myid;
int sym=0;
int row_start;
int *I;
int *J;
double * data;
// MUMPS workspace
// macro s.t. indices match MUMPS documentation
#define ICNTL(I) icntl[(I) -1]
#define INFO(I) info[(I) -1]
DMUMPS_STRUC_C *id=nullptr;
void SetParameters();
#if MFEM_MUMPS_VERSION >= 530
Array<int> row_starts;
Array<int> irhs_loc;
Array<int> isol_loc;
Vector sol_loc;
int GetRowRank(int i, const Array<int> &row_starts_) const;
void RedistributeSol(const Array<int> &row_map,
const Vector &x,
Vector &y) const;
#else
Array<int> recv_counts;
Array<int> displs;
Vector rhs_glob;
#endif
}; // mfem::MUMPSSolver class
} // namespace mfem
#endif // MFEM_USE_MPI
#endif // MFEM_USE_MUMPS
#endif // MFEM_MUMPS
+20 -42
View File
@@ -134,7 +134,7 @@ OperatorJacobiSmoother::OperatorJacobiSmoother(const BilinearForm &a,
OperatorJacobiSmoother::OperatorJacobiSmoother(const Vector &d,
const Array<int> &ess_tdofs,
const double dmpng)
const double dmpng, const bool inverse)
:
Solver(d.Size()),
N(d.Size()),
@@ -143,16 +143,30 @@ OperatorJacobiSmoother::OperatorJacobiSmoother(const Vector &d,
ess_tdof_list(ess_tdofs),
residual(N)
{
Setup(d);
Setup(d, inverse);
}
void OperatorJacobiSmoother::Setup(const Vector &diag)
void OperatorJacobiSmoother::Setup(const Vector &diag, const bool inverse)
{
residual.UseDevice(true);
const double delta = damping;
auto D = diag.Read();
auto DI = dinv.Write();
MFEM_FORALL(i, N, DI[i] = delta / D[i]; );
if (inverse)
{
if (delta > 0.0)
{
MFEM_FORALL(i, N, DI[i] = delta * D[i]; );
}
else
{
MFEM_FORALL(i, N, DI[i] = D[i]; );
}
}
else
{
MFEM_FORALL(i, N, DI[i] = delta / D[i]; );
}
auto I = ess_tdof_list.Read();
MFEM_FORALL(i, ess_tdof_list.Size(), DI[I[i]] = delta; );
}
@@ -2284,7 +2298,7 @@ void MinimumDiscardedFillOrdering(SparseMatrix &C, Array<int> &p)
{
int i = J[ii];
// Find value of (i,k)
double C_ik = 0.0;
double C_ik;
for (int kk=I[i]; kk<I[i+1]; ++kk)
{
if (J[kk] == k)
@@ -2334,7 +2348,7 @@ void MinimumDiscardedFillOrdering(SparseMatrix &C, Array<int> &p)
int i = J[ii2];
if (w_heap.picked(i)) { continue; }
// Find value of (i,k)
double C_ik = 0.0;
double C_ik;
for (int kk2=I[i]; kk2<I[i+1]; ++kk2)
{
if (J[kk2] == k)
@@ -2665,42 +2679,6 @@ void BlockILU::Mult(const Vector &b, Vector &x) const
}
}
void ResidualBCMonitor::MonitorResidual(
int it, double norm, const Vector &r, bool final)
{
if (!ess_dofs_list) { return; }
double bc_norm_squared = 0.0;
r.HostRead();
ess_dofs_list->HostRead();
for (int i = 0; i < ess_dofs_list->Size(); i++)
{
const double r_entry = r((*ess_dofs_list)[i]);
bc_norm_squared += r_entry*r_entry;
}
bool print = true;
#ifdef MFEM_USE_MPI
MPI_Comm comm = iter_solver->GetComm();
if (comm != MPI_COMM_NULL)
{
double glob_bc_norm_squared = 0.0;
MPI_Reduce(&bc_norm_squared, &glob_bc_norm_squared, 1, MPI_DOUBLE,
MPI_SUM, 0, comm);
bc_norm_squared = glob_bc_norm_squared;
int rank;
MPI_Comm_rank(comm, &rank);
print = (rank == 0);
}
#endif
if ((it == 0 || final || bc_norm_squared > 0.0) && print)
{
mfem::out << " ResidualBCMonitor : b.c. residual norm = "
<< sqrt(bc_norm_squared) << endl;
}
}
#ifdef MFEM_USE_SUITESPARSE
void UMFPackSolver::Init()
+5 -40
View File
@@ -33,12 +33,8 @@ class BilinearForm;
/// Abstract base class for an iterative solver monitor
class IterativeSolverMonitor
{
protected:
/// The last IterativeSolver to which this monitor was attached.
const class IterativeSolver *iter_solver;
public:
IterativeSolverMonitor() : iter_solver(nullptr) {}
IterativeSolverMonitor() {}
virtual ~IterativeSolverMonitor() {}
@@ -53,11 +49,6 @@ public:
bool final)
{
}
/** @brief This method is invoked by ItertiveSolver::SetMonitor, informing
the monitor which IterativeSolver is using it. */
void SetIterativeSolver(const IterativeSolver &solver)
{ iter_solver = &solver; }
};
/// Abstract base class for iterative solver
@@ -109,15 +100,7 @@ public:
virtual void SetOperator(const Operator &op);
/// Set the iterative solver monitor
void SetMonitor(IterativeSolverMonitor &m)
{ monitor = &m; m.SetIterativeSolver(*this); }
#ifdef MFEM_USE_MPI
/** @brief Return the associated MPI communicator, or MPI_COMM_NULL if no
communicator is set. */
MPI_Comm GetComm() const
{ return dot_prod_type == 0 ? MPI_COMM_NULL : comm; }
#endif
void SetMonitor(IterativeSolverMonitor &m) { monitor = &m; }
};
@@ -142,13 +125,14 @@ public:
the matrix-free setting. */
OperatorJacobiSmoother(const Vector &d,
const Array<int> &ess_tdof_list,
const double damping=1.0);
const double damping=1.0,
const bool inverse=false);
~OperatorJacobiSmoother() {}
void Mult(const Vector &x, Vector &y) const;
void MultTranspose(const Vector &x, Vector &y) const { Mult(x, y); }
void SetOperator(const Operator &op) { oper = &op; }
void Setup(const Vector &diag);
void Setup(const Vector &diag, const bool inverse=false);
private:
const int N;
@@ -706,25 +690,6 @@ private:
mutable Array<int> ipiv;
};
/// Monitor that checks whether the residual is zero at a given set of dofs.
/** This monitor is useful for checking if the initial guess, rhs, operator, and
preconditioner are properly setup for solving in the subspace with imposed
essential boundary conditions. */
class ResidualBCMonitor : public IterativeSolverMonitor
{
protected:
const Array<int> *ess_dofs_list; ///< Not owned
public:
ResidualBCMonitor(const Array<int> &ess_dofs_list_)
: ess_dofs_list(&ess_dofs_list_) { }
void MonitorResidual(int it, double norm, const Vector &r,
bool final) override;
};
#ifdef MFEM_USE_SUITESPARSE
/// Direct sparse solver using UMFPACK
+8 -210
View File
@@ -28,25 +28,6 @@ namespace mfem
using namespace std;
#ifdef MFEM_USE_CUDA
int SparseMatrix::SparseMatrixCount = 0;
cusparseHandle_t SparseMatrix::handle;
size_t SparseMatrix::bufferSize = 0;
void * SparseMatrix::dBuffer = nullptr;
#endif
void SparseMatrix::InitCuSparse()
{
// Initialize cuSPARSE library
#ifdef MFEM_USE_CUDA
SparseMatrixCount++;
if (SparseMatrixCount == 1 && Device::Allows(Backend::CUDA_MASK))
{
cusparseCreate(&handle);
}
#endif
}
SparseMatrix::SparseMatrix(int nrows, int ncols)
: AbstractSparseMatrix(nrows, (ncols >= 0) ? ncols : nrows),
Rows(new RowNode *[nrows]),
@@ -69,8 +50,6 @@ SparseMatrix::SparseMatrix(int nrows, int ncols)
#ifdef MFEM_USE_MEMALLOC
NodesMem = new RowNodeAlloc;
#endif
InitCuSparse();
}
SparseMatrix::SparseMatrix(int *i, int *j, double *data, int m, int n)
@@ -88,8 +67,6 @@ SparseMatrix::SparseMatrix(int *i, int *j, double *data, int m, int n)
#ifdef MFEM_USE_MEMALLOC
NodesMem = NULL;
#endif
InitCuSparse();
}
SparseMatrix::SparseMatrix(int *i, int *j, double *data, int m, int n,
@@ -121,8 +98,6 @@ SparseMatrix::SparseMatrix(int *i, int *j, double *data, int m, int n,
A[i] = 0.0;
}
}
InitCuSparse();
}
SparseMatrix::SparseMatrix(int nrows, int ncols, int rowsize)
@@ -144,8 +119,6 @@ SparseMatrix::SparseMatrix(int nrows, int ncols, int rowsize)
{
I[i] = i * rowsize;
}
InitCuSparse();
}
SparseMatrix::SparseMatrix(const SparseMatrix &mat, bool copy_graph)
@@ -211,8 +184,6 @@ SparseMatrix::SparseMatrix(const SparseMatrix &mat, bool copy_graph)
ColPtrNode = NULL;
At = NULL;
isSorted = mat.isSorted;
InitCuSparse();
}
SparseMatrix::SparseMatrix(const Vector &v)
@@ -240,8 +211,6 @@ SparseMatrix::SparseMatrix(const Vector &v)
J[r] = r;
A[r] = v[r];
}
InitCuSparse();
}
SparseMatrix& SparseMatrix::operator=(const SparseMatrix &rhs)
@@ -281,16 +250,6 @@ void SparseMatrix::SetEmpty()
NodesMem = NULL;
#endif
isSorted = false;
#ifdef MFEM_USE_CUDA
if (initBuffers)
{
cusparseDestroySpMat(matA_descr);
cusparseDestroyDnVec(vecX_descr);
cusparseDestroyDnVec(vecY_descr);
initBuffers = false;
}
#endif
}
int SparseMatrix::RowSize(const int i) const
@@ -610,7 +569,7 @@ void SparseMatrix::AddMult(const Vector &x, Vector &y, const double a) const
const double *xp = x.HostRead();
double *yp = y.HostReadWrite();
// The matrix is not finalized, but multiplication is still possible
// The matrix is not finalized, but multiplication is still possible
for (int i = 0; i < height; i++)
{
RowNode *row = Rows[i];
@@ -633,72 +592,16 @@ void SparseMatrix::AddMult(const Vector &x, Vector &y, const double a) const
auto d_A = Read(A, nnz);
auto d_x = x.Read();
auto d_y = y.ReadWrite();
// Skip if matrix has no non-zeros
if (nnz == 0) {return;}
if (Device::Allows(Backend::CUDA_MASK) && useCuSparse)
MFEM_FORALL(i, height,
{
#ifdef MFEM_USE_CUDA
const double alpha = a;
const double beta = 1.0;
// Setup descriptors
if (!initBuffers)
double d = 0.0;
const int end = d_I[i+1];
for (int j = d_I[i]; j < end; j++)
{
// Setup matrix descriptor
cusparseCreateCsr(&matA_descr,Height(), Width(), J.Capacity(),
const_cast<int *>(d_I),
const_cast<int *>(d_J), const_cast<double *>(d_A), CUSPARSE_INDEX_32I,
CUSPARSE_INDEX_32I, CUSPARSE_INDEX_BASE_ZERO, CUDA_R_64F);
// Create handles for input/output vectors
cusparseCreateDnVec(&vecX_descr, x.Size(), const_cast<double *>(d_x),
CUDA_R_64F);
cusparseCreateDnVec(&vecY_descr, y.Size(), d_y, CUDA_R_64F);
initBuffers = true;
d += d_A[j] * d_x[d_J[j]];
}
// Allocate kernel space. Buffer is shared between different sparsemats
size_t newBufferSize = 0;
cusparseSpMV_bufferSize(handle, CUSPARSE_OPERATION_NON_TRANSPOSE, &alpha,
matA_descr,
vecX_descr, &beta, vecY_descr, CUDA_R_64F,
CUSPARSE_CSRMV_ALG1, &newBufferSize);
// Check if we need to resize
if (newBufferSize > bufferSize)
{
bufferSize = newBufferSize;
if (dBuffer != NULL) { CuMemFree(dBuffer); }
CuMemAlloc(&dBuffer, bufferSize);
}
// Update input/output vectors
cusparseDnVecSetValues(vecX_descr, const_cast<double *>(d_x));
cusparseDnVecSetValues(vecY_descr, d_y);
// Y = alpha A * X + beta * Y
cusparseSpMV(handle, CUSPARSE_OPERATION_NON_TRANSPOSE, &alpha, matA_descr,
vecX_descr, &beta, vecY_descr, CUDA_R_64F, CUSPARSE_CSRMV_ALG1, dBuffer);
#endif
}
else
{
// Native version
MFEM_FORALL(i, height,
{
double d = 0.0;
const int end = d_I[i+1];
for (int j = d_I[i]; j < end; j++)
{
d += d_A[j] * d_x[d_J[j]];
}
d_y[i] += a * d;
});
}
d_y[i] += a * d;
});
#else
const double *Ap = A, *xp = x.GetData();
double *yp = y.GetData();
@@ -881,101 +784,6 @@ void SparseMatrix::BooleanMultTranspose(const Array<int> &x,
}
}
void SparseMatrix::AbsMult(const Vector &x, Vector &y) const
{
MFEM_ASSERT(width == x.Size(), "Input vector size (" << x.Size()
<< ") must match matrix width (" << width << ")");
MFEM_ASSERT(height == y.Size(), "Output vector size (" << y.Size()
<< ") must match matrix height (" << height << ")");
if (Finalized()) { y.UseDevice(true); }
y = 0.0;
if (!Finalized())
{
const double *xp = x.HostRead();
double *yp = y.HostReadWrite();
// The matrix is not finalized, but multiplication is still possible
for (int i = 0; i < height; i++)
{
RowNode *row = Rows[i];
double b = 0.0;
for ( ; row != NULL; row = row->Prev)
{
b += std::abs(row->Value) * xp[row->Column];
}
*yp += b;
yp++;
}
return;
}
const int height = this->height;
const int nnz = J.Capacity();
auto d_I = Read(I, height+1);
auto d_J = Read(J, nnz);
auto d_A = Read(A, nnz);
auto d_x = x.Read();
auto d_y = y.ReadWrite();
MFEM_FORALL(i, height,
{
double d = 0.0;
const int end = d_I[i+1];
for (int j = d_I[i]; j < end; j++)
{
d += std::abs(d_A[j]) * d_x[d_J[j]];
}
d_y[i] += d;
});
}
void SparseMatrix::AbsMultTranspose(const Vector &x, Vector &y) const
{
MFEM_ASSERT(height == x.Size(), "Input vector size (" << x.Size()
<< ") must match matrix height (" << height << ")");
MFEM_ASSERT(width == y.Size(), "Output vector size (" << y.Size()
<< ") must match matrix width (" << width << ")");
y = 0.0;
if (!Finalized())
{
double *yp = y.GetData();
// The matrix is not finalized, but multiplication is still possible
for (int i = 0; i < height; i++)
{
RowNode *row = Rows[i];
double b = x(i);
for ( ; row != NULL; row = row->Prev)
{
yp[row->Column] += fabs(row->Value) * b;
}
}
return;
}
if (At)
{
At->AbsMult(x, y);
}
else
{
MFEM_VERIFY(Device::IsDisabled(), "transpose action on device is not "
"enabled; see BuildTranspose() for details.");
for (int i = 0; i < height; i++)
{
const double xi = x[i];
const int end = I[i+1];
for (int j = I[i]; j < end; j++)
{
const int Jj = J[j];
y[Jj] += std::abs(A[j]) * xi;
}
}
}
}
double SparseMatrix::InnerProduct(const Vector &x, const Vector &y) const
{
MFEM_ASSERT(x.Size() == Width(), "x.Size() = " << x.Size()
@@ -3154,16 +2962,6 @@ void SparseMatrix::Destroy()
delete NodesMem;
#endif
delete At;
#ifdef MFEM_USE_CUDA
if (initBuffers)
{
cusparseDestroySpMat(matA_descr);
cusparseDestroyDnVec(vecX_descr);
cusparseDestroyDnVec(vecY_descr);
initBuffers = false;
}
#endif
}
int SparseMatrix::ActualWidth() const

Some files were not shown because too many files have changed in this diff Show More