Compare commits
248
Commits
dfem-dev
...
dfem-bench
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
757b591c2a | ||
|
|
b0bd96752f | ||
|
|
4c52ecefcd | ||
|
|
aa3d79e2d5 | ||
|
|
fd587a9a0f | ||
|
|
faa556ad42 | ||
|
|
c8a48e768d | ||
|
|
22edcd0f90 | ||
|
|
81963e09c0 | ||
|
|
88b72a47a7 | ||
|
|
05c5e98a90 | ||
|
|
ee0d1fa0b7 | ||
|
|
bf3a40f73e | ||
|
|
856d13e9ff | ||
|
|
7763785ed7 | ||
|
|
2baa889917 | ||
|
|
2ed1a9eaad | ||
|
|
1545f03a94 | ||
|
|
59a5c9fc79 | ||
|
|
c389a3c434 | ||
|
|
ec96a85f86 | ||
|
|
81b6b7eeb2 | ||
|
|
4b5974f600 | ||
|
|
a6926f4ce6 | ||
|
|
b6e972af79 | ||
|
|
e76ec19775 | ||
|
|
d797322fea | ||
|
|
5a5e34a744 | ||
|
|
93db7052ff | ||
|
|
f7170af7bd | ||
|
|
b78eef3eaa | ||
|
|
d5decea85c | ||
|
|
dea3ae3317 | ||
|
|
33c1e50235 | ||
|
|
5718ad1b53 | ||
|
|
4f3671e253 | ||
|
|
4e08bb1b66 | ||
|
|
69c5016b63 | ||
|
|
ce1bf58dc0 | ||
|
|
5eb00c9ee6 | ||
|
|
1f3b6b95aa | ||
|
|
118db41049 | ||
|
|
8390c3e50b | ||
|
|
168b5179e6 | ||
|
|
7697f6d400 | ||
|
|
235ebce5d5 | ||
|
|
e8a09d6499 | ||
|
|
edc67827d8 | ||
|
|
9e5cdef2ef | ||
|
|
c2f4a5e248 | ||
|
|
72d811b289 | ||
|
|
43731aa990 | ||
|
|
45f59fff3a | ||
|
|
58a4cfa132 | ||
|
|
333dd3f2fd | ||
|
|
c4f7dd77b1 | ||
|
|
f442b83573 | ||
|
|
768aaae25d | ||
|
|
eab997c557 | ||
|
|
9d73dc487d | ||
|
|
2575ac61ba | ||
|
|
6130144da1 | ||
|
|
68db31da44 | ||
|
|
1acbce733c | ||
|
|
b44316049b | ||
|
|
2e133e8ecb | ||
|
|
dfb2f4d7f2 | ||
|
|
10e9e4215f | ||
|
|
8125a211d3 | ||
|
|
818b8db433 | ||
|
|
ad4626edfc | ||
|
|
8d7e8933cf | ||
|
|
3ad21a409f | ||
|
|
b16b550150 | ||
|
|
102dc8bd02 | ||
|
|
e306ba0c85 | ||
|
|
4b88ad2b0a | ||
|
|
d0fb4e342e | ||
|
|
b53d0529db | ||
|
|
8c7988b525 | ||
|
|
dfffe4b5e8 | ||
|
|
538aa11904 | ||
|
|
6fa978af9a | ||
|
|
3f81af72f6 | ||
|
|
97f1cf08fb | ||
|
|
d3f1379dc8 | ||
|
|
ea6fb52698 | ||
|
|
07a87e369c | ||
|
|
53bc415268 | ||
|
|
be537728df | ||
|
|
4e5b98b10f | ||
|
|
d4acd906bf | ||
|
|
d751ce66a3 | ||
|
|
3f0abd4dfd | ||
|
|
44a423d804 | ||
|
|
3e61e0490e | ||
|
|
def4919592 | ||
|
|
2d147d70e0 | ||
|
|
e29e64dffe | ||
|
|
a7ec259bd5 | ||
|
|
3e93e19767 | ||
|
|
532b065596 | ||
|
|
82c1e2315b | ||
|
|
8cc9eec535 | ||
|
|
dece65be31 | ||
|
|
3e6d29b3dd | ||
|
|
487135b497 | ||
|
|
91f648aa95 | ||
|
|
999931ded2 | ||
|
|
15dbcae725 | ||
|
|
01efb623da | ||
|
|
f854c5262d | ||
|
|
a91b754aaa | ||
|
|
b1623ff3d4 | ||
|
|
c2426ca45a | ||
|
|
276f419a3d | ||
|
|
c91b8bea01 | ||
|
|
7bdceca6ce | ||
|
|
6f9a263435 | ||
|
|
28a7865ed1 | ||
|
|
6e7335ac52 | ||
|
|
8115383dec | ||
|
|
3d1b017a60 | ||
|
|
bf14e5b018 | ||
|
|
fb3517453f | ||
|
|
bfca6beb28 | ||
|
|
f51e46d3d8 | ||
|
|
78a60cc1d9 | ||
|
|
935d3a9e42 | ||
|
|
35866f8485 | ||
|
|
6b4b644355 | ||
|
|
b9ec58e7a1 | ||
|
|
4644aed322 | ||
|
|
80da896859 | ||
|
|
b96dcb4401 | ||
|
|
5054f1784d | ||
|
|
788c0efda0 | ||
|
|
4d49d42702 | ||
|
|
f5192230e0 | ||
|
|
400e3eca7d | ||
|
|
b90c8d80fe | ||
|
|
a0491f6bfc | ||
|
|
a8df54cf5d | ||
|
|
9c4e43ee12 | ||
|
|
7a1887c525 | ||
|
|
907783f9ca | ||
|
|
f4f68fa021 | ||
|
|
b76e9e80a7 | ||
|
|
b8f677b2fe | ||
|
|
6e42fbae4d | ||
|
|
b8c0008061 | ||
|
|
cdce090c2a | ||
|
|
e246c0852b | ||
|
|
4db86286ee | ||
|
|
9308946715 | ||
|
|
d28eca6b7f | ||
|
|
8e26105232 | ||
|
|
c674f9f7ad | ||
|
|
537d30120a | ||
|
|
519267e1cb | ||
|
|
2c495fb70d | ||
|
|
401d1aec7b | ||
|
|
8299b1c036 | ||
|
|
9dd1e4dbdb | ||
|
|
47a3534eff | ||
|
|
2f39ff66f3 | ||
|
|
ddca183704 | ||
|
|
078ce6130c | ||
|
|
6d15c2a156 | ||
|
|
7d705c0677 | ||
|
|
c027328b91 | ||
|
|
65cb67e1c1 | ||
|
|
494f27c14c | ||
|
|
3c02b72084 | ||
|
|
75e2be35ba | ||
|
|
b0f9cbfd26 | ||
|
|
6afea18cde | ||
|
|
2e69ff4b97 | ||
|
|
6c70fe9334 | ||
|
|
4ccbd4581e | ||
|
|
6e262f6c3f | ||
|
|
52e10475a5 | ||
|
|
c0299a5a4b | ||
|
|
06eecb0dce | ||
|
|
96261a7742 | ||
|
|
d7c479fa1e | ||
|
|
8b01d8f13b | ||
|
|
710da275c8 | ||
|
|
fd481eb725 | ||
|
|
b5bbdbbed5 | ||
|
|
6a26200314 | ||
|
|
a485121526 | ||
|
|
1f9e1cf175 | ||
|
|
ec402882da | ||
|
|
e7633e0e2c | ||
|
|
30aeb465b7 | ||
|
|
ff4993fc51 | ||
|
|
0a42ea8021 | ||
|
|
b8d024b59b | ||
|
|
9e1ccf4543 | ||
|
|
075ebb255d | ||
|
|
3eb6a5b3b2 | ||
|
|
8ba1f17f72 | ||
|
|
e5f5a79e66 | ||
|
|
43f1b19767 | ||
|
|
7bebe4528f | ||
|
|
da63657cdd | ||
|
|
2b1d271888 | ||
|
|
47fb8a4fda | ||
|
|
ee7d9726df | ||
|
|
44b560a916 | ||
|
|
5657f6ebe8 | ||
|
|
19543b6b16 | ||
|
|
94a832a0c6 | ||
|
|
b56e994ecd | ||
|
|
8be11cdfdb | ||
|
|
d71a9602b5 | ||
|
|
1108bb7e85 | ||
|
|
ae8e5aa88d | ||
|
|
17f4acf6b1 | ||
|
|
2ce3f3037c | ||
|
|
29189a6d4a | ||
|
|
08f3c86b8a | ||
|
|
c6eb171b5b | ||
|
|
d26695cd2a | ||
|
|
01ab390b06 | ||
|
|
43c42295d3 | ||
|
|
52bc915120 | ||
|
|
cd9cabb955 | ||
|
|
e66a61c198 | ||
|
|
f8b3c78b19 | ||
|
|
4749746171 | ||
|
|
1ddd01c2a0 | ||
|
|
87ec3850b5 | ||
|
|
1b25a61c9e | ||
|
|
7bee8e8161 | ||
|
|
a545ff8264 | ||
|
|
5352234aef | ||
|
|
c3732f9d86 | ||
|
|
b95f3809fe | ||
|
|
e6a28b7753 | ||
|
|
62adea8b46 | ||
|
|
7d11db33c0 | ||
|
|
11fce4235b | ||
|
|
f500b4875f | ||
|
|
ba212c583e | ||
|
|
fd341e07da | ||
|
|
d59e2a229c |
@@ -144,6 +144,8 @@ examples/amgx/sol.gf
|
||||
examples/amgx/mesh.*
|
||||
examples/amgx/sol.*
|
||||
|
||||
examples/dfem/minimal_surface
|
||||
|
||||
examples/caliper/ex1
|
||||
examples/caliper/ex1p
|
||||
examples/caliper/refined.mesh
|
||||
|
||||
@@ -157,4 +157,18 @@ constexpr real_t operator""_r(unsigned long long v)
|
||||
#endif
|
||||
#endif // MFEM_USE_MPI not defined
|
||||
|
||||
#ifdef NVTX_DBG_HPP
|
||||
#include NVTX_DBG_HPP
|
||||
#else
|
||||
#define db1(...)
|
||||
#define dbg(...)
|
||||
#define dbl(...)
|
||||
#define dba(...)
|
||||
#define dbc(...)
|
||||
#define NVTX_MARK_FUNCTION
|
||||
#define NVTX_MARK_BEGIN(...)
|
||||
#define NVTX_MARK_END(...)
|
||||
#define NVTX(...)
|
||||
#endif
|
||||
|
||||
#endif // MFEM_CONFIG_HPP
|
||||
|
||||
@@ -522,6 +522,8 @@ GSLIB_LIB = -L$(GSLIB_DIR)/lib -lgs
|
||||
|
||||
# CUDA library configuration
|
||||
CUDA_OPT =
|
||||
# base CUDA install directory, only needed if building with clang+cuda
|
||||
CUDA_DIR = /usr/local/cuda/
|
||||
CUDA_LIB = -lcusparse -lcublas
|
||||
CLANG_CUDA_LIB = -L$(CUDA_DIR)/lib64 -L$(CUDA_DIR)/lib \
|
||||
$(XLINKER)-rpath,$(CUDA_DIR)/lib64,-rpath,$(CUDA_DIR)/lib \
|
||||
|
||||
@@ -2209,6 +2209,7 @@ private:
|
||||
const FiniteElementSpace *fespace;
|
||||
const DofToQuad *maps; ///< Not owned
|
||||
const GeometricFactors *geom; ///< Not owned
|
||||
public:
|
||||
int dim, ne, dofs1D, quad1D;
|
||||
Vector pa_data;
|
||||
bool symmetric = true; ///< False if using a nonsymmetric matrix coefficient
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
export LC_USER=andrej1
|
||||
module load rocmcc/6.3.1-cce-19.0.0-magic cmake/3.29.2
|
||||
|
||||
export MPICH_CC=amdclang
|
||||
export MPICH_CXX=amdclang++
|
||||
export ROCM_PATH=/opt/rocm-6.3.1
|
||||
export LLVM_DIR=$ROCM_PATH/lib/llvm
|
||||
export MPI_DIR=/usr/tce/packages/cray-mpich/cray-mpich-8.1.32-rocmcc-6.3.1-cce-19.0.0-magic
|
||||
|
||||
export CMAKE_PREFIX_PATH=$CMAKE_PREFIX_PATH:$ROCM_PATH/lib/cmake/hip:$ROCM_PATH/lib/cmake/hipblas:$ROCM_PATH/lib/cmake/hipblas-common:$ROCM_PATH/lib/cmake/hipsparse:$ROCM_PATH/lib/cmake/rocsparse:$ROCM_PATH/lib/cmake/rocrand
|
||||
|
||||
export BASE_DIR=/usr/workspace/$LC_USER/dfem-tuo-magic
|
||||
export LOCAL_DIR=/usr/workspace/$LC_USER/dfem-tuo-magic/local
|
||||
mkdir -p $LOCAL_DIR
|
||||
export PATH=$LOCAL_DIR/bin:$PATH
|
||||
cd $BASE_DIR
|
||||
|
||||
## Enzyme
|
||||
git clone --depth 1 https://github.com/EnzymeAD/Enzyme.git
|
||||
pushd Enzyme/enzyme
|
||||
CC=amdclang CXX=amdclang++ cmake -B build -DLLVM_DIR=$LLVM_DIR -DCMAKE_INSTALL_PREFIX=$LOCAL_DIR
|
||||
cmake --build build -j && cmake --install build
|
||||
popd
|
||||
|
||||
## hypre
|
||||
curl https://github.com/hypre-space/hypre/archive/refs/tags/v2.32.0.tar.gz -o hypre-v2.32.0.tar.gz -L
|
||||
tar xzf hypre-v2.32.0.tar.gz
|
||||
pushd hypre-2.32.0/src
|
||||
CC=mpicc CXX=mpicxx CXXFLAGS="std=c++17 -fPIC" CFLAGS="-fPIC" ROCM_PATH=$ROCM_PATH ./configure --disable-fortran --prefix=$LOCAL_DIR --with-MPI-libs="mpi mpich" --with-MPI-lib-dirs=$MPI_DIR/lib --with-MPI-include=$MPI_DIR/include --enable-shared --with-hip
|
||||
make -j install
|
||||
popd
|
||||
|
||||
## metis
|
||||
curl -OL https://github.com/mfem/tpls/raw/gh-pages/parmetis-4.0.3.tar.gz
|
||||
tar xzf parmetis-4.0.3.tar.gz
|
||||
pushd parmetis-4.0.3
|
||||
cmake -B build -DCMAKE_CXX_FLAGS="-fPIC" -DCMAKE_C_FLAGS="-fPIC" -DGKLIB_PATH=$BASE_DIR/parmetis-4.0.3/metis/GKlib -DMETIS_PATH=$BASE_DIR/parmetis-4.0.3/metis -DCMAKE_INSTALL_PREFIX=$LOCAL_DIR -DSHARED=1 -DCMAKE_C_COMPILER=mpicc -DCMAKE_CXX_COMPILER=mpicxx
|
||||
cmake --build build -j && cmake --install build
|
||||
popd
|
||||
pushd parmetis-4.0.3/metis
|
||||
cmake -B build -DCMAKE_CXX_FLAGS="-fPIC" -DCMAKE_C_FLAGS="-fPIC" -DGKLIB_PATH=$BASE_DIR/parmetis-4.0.3/metis/GKlib -DCMAKE_INSTALL_PREFIX=$LOCAL_DIR -DSHARED=1 -DCMAKE_C_COMPILER=mpicc -DCMAKE_CXX_COMPILER=mpicxx
|
||||
cmake --build build -j && cmake --install build
|
||||
popd
|
||||
|
||||
git clone https://github.com/mfem/mfem.git
|
||||
git switch dfem-phase1-dev
|
||||
pushd mfem
|
||||
CXX=mpicxx cmake -B build-opt -DCMAKE_BUILD_TYPE=Release -DMFEM_USE_HIP=ON -DCMAKE_HIP_ARCHITECTURES="gfx942" -DCMAKE_HIP_PLATFORM="amd"
|
||||
cmake --build build-opt -j
|
||||
@@ -0,0 +1,31 @@
|
||||
if (NOT CMAKE_BUILD_TYPE)
|
||||
set(CMAKE_BUILD_TYPE "Release" CACHE STRING
|
||||
"Build type: Debug, Release, RelWithDebInfo, or MinSizeRel." FORCE)
|
||||
endif()
|
||||
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
# set(CMAKE_CXX_FLAGS "--save-temps -Rpass-analysis=kernel-resource-usage -mllvm -amdgpu-early-inline-all=true -mllvm -amdgpu-function-calls=false")
|
||||
|
||||
set(MFEM_PRECISION "double" CACHE STRING
|
||||
"Floating-point precision to use: single, or double")
|
||||
|
||||
option(BUILD_SHARED_LIBS "Enable shared library build of MFEM" ON)
|
||||
option(MFEM_USE_MPI "Enable MPI parallel build" ON)
|
||||
option(MFEM_USE_METIS "Enable METIS usage" ${MFEM_USE_MPI})
|
||||
option(MFEM_USE_ENZYME "Enable Enzyme" ON)
|
||||
option(MFEM_USE_HIP "Enable HIP" ON)
|
||||
|
||||
set(MFEM_MPI_NP 4 CACHE STRING "Number of processes used for MPI tests")
|
||||
|
||||
option(MFEM_ENABLE_TESTING ON)
|
||||
|
||||
set(HIP_ARCH "gfx942" CACHE STRING "Target HIP architecture.")
|
||||
|
||||
# Make sure all dirs are absolute
|
||||
set(ENZYME_DIR "/usr/workspace/andrej1/dfem-tuo-magic/local/cmake/Enzyme" CACHE PATH "Path to the Enzyme library.")
|
||||
set(HYPRE_DIR "/usr/workspace/andrej1/dfem-tuo-magic/local" CACHE PATH "Path to the hypre library.")
|
||||
set(METIS_DIR "/usr/workspace/andrej1/dfem-tuo-magic/local" CACHE PATH "Path to the METIS library.")
|
||||
|
||||
set(CMAKE_SKIP_PREPROCESSED_SOURCE_RULES ON) # Skip *.i rules
|
||||
set(CMAKE_SKIP_ASSEMBLY_SOURCE_RULES ON) # Skip *.s rules
|
||||
Symlink
+1
@@ -0,0 +1 @@
|
||||
../../stash/debug/nvtx.hpp
|
||||
@@ -119,7 +119,7 @@ $(if $(word 2,$(SRC)),$(error Spaces in SRC = "$(SRC)" are not supported))
|
||||
MFEM_GIT_STRING = $(shell [ -d $(MFEM_DIR)/.git ] && git -C $(MFEM_DIR) \
|
||||
describe --all --long --abbrev=40 --dirty --always 2> /dev/null)
|
||||
|
||||
EXAMPLE_SUBDIRS = amgx caliper ginkgo hiop petsc pumi sundials superlu moonolith
|
||||
EXAMPLE_SUBDIRS = amgx dfem caliper ginkgo hiop petsc pumi sundials superlu moonolith
|
||||
EXAMPLE_DIRS := examples $(addprefix examples/,$(EXAMPLE_SUBDIRS))
|
||||
EXAMPLE_TEST_DIRS := examples
|
||||
|
||||
@@ -807,7 +807,7 @@ FORMAT_EXCLUDE = general/tinyxml2.cpp tests/unit/catch.hpp
|
||||
FORMAT_LIST = $(filter-out $(FORMAT_EXCLUDE),$(wildcard $(FORMAT_FILES)))
|
||||
|
||||
COUT_CERR_FILES = $(foreach dir,$(DIRS),$(dir)/*.[ch]pp)
|
||||
COUT_CERR_EXCLUDE = '^general/error\.cpp' '^general/globals\.[ch]pp'
|
||||
COUT_CERR_EXCLUDE = '^general/error\.cpp' '^general/globals\.[ch]pp' '^general/nvtx\.hpp'
|
||||
|
||||
DEPRECATION_WARNING := \
|
||||
"This feature is planned for removal in the next release."\
|
||||
|
||||
@@ -52,4 +52,8 @@
|
||||
#include "fem/moonolith/transfer.hpp"
|
||||
#endif // MFEM_USE_MOONOLITH
|
||||
|
||||
#ifdef NVTX_FMT_HPP
|
||||
#include NVTX_FMT_HPP
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
@@ -51,6 +51,7 @@ endfunction(add_benchmark)
|
||||
#-------------------------------------------------------------------------------
|
||||
add_benchmark(assembly_levels)
|
||||
add_benchmark(ceed)
|
||||
add_benchmark(dfem)
|
||||
add_benchmark(dg_amr)
|
||||
add_benchmark(elasticity)
|
||||
add_benchmark(tmop)
|
||||
|
||||
@@ -217,7 +217,7 @@ struct BP : public BakeOff<BFI, VDIM, GLL>
|
||||
cg.SetRelTol(0.0);
|
||||
cg.SetMaxIter(max_it);
|
||||
cg.SetPrintLevel(print_lvl);
|
||||
|
||||
cg.iterative_mode = false;
|
||||
benchmark();
|
||||
mdofs = 0.0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,617 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
#define NVTX_COLOR ::nvtx::kNvidia
|
||||
|
||||
#include "bench.hpp" // IWYU pragma: keep
|
||||
|
||||
#ifdef MFEM_USE_BENCHMARK
|
||||
|
||||
#include <memory>
|
||||
|
||||
#include "fem/qinterp/det.hpp" // IWYU pragma: keep
|
||||
#include "fem/qinterp/grad.hpp" // IWYU pragma: keep
|
||||
#include "fem/integ/lininteg_domain_kernels.hpp" // IWYU pragma: keep
|
||||
#include "fem/integ/bilininteg_vecdiffusion_pa.hpp" // IWYU pragma: keep
|
||||
|
||||
#include <fem/dfem/doperator.hpp>
|
||||
#include <linalg/tensor.hpp>
|
||||
|
||||
#include "fem/kernels.hpp"
|
||||
namespace ker = kernels::internal;
|
||||
|
||||
using namespace mfem;
|
||||
|
||||
using mfem::future::tuple;
|
||||
using mfem::future::tensor;
|
||||
|
||||
using future::DifferentiableOperator;
|
||||
using future::UniformParameterSpace;
|
||||
using future::ParameterFunction;
|
||||
using future::FieldDescriptor;
|
||||
using future::Gradient;
|
||||
using future::Weight;
|
||||
using future::Identity;
|
||||
|
||||
/// info //////////////////////////////////////////////////////////////////////
|
||||
void info()
|
||||
{
|
||||
mfem::out << "\x1b[33m";
|
||||
mfem::out << "version 0: PA std" << std::endl;
|
||||
mfem::out << "version 1: PA new" << std::endl;
|
||||
mfem::out << "version 2: MF ∂fem-master" << std::endl;
|
||||
mfem::out << "version 3: PA ∂fem-master" << std::endl;
|
||||
mfem::out << "\x1b[m" << std::endl;
|
||||
}
|
||||
|
||||
// Custom benchmark arguments generator ///////////////////////////////////////
|
||||
static void CustomArguments(bm::Benchmark *b) noexcept
|
||||
{
|
||||
constexpr int MAX_NDOFS = 8 * 1024 * (mfem_use_gpu ? 1024 : 8);
|
||||
|
||||
const auto versions = { 0, /*1, 2,*/ 3 };
|
||||
|
||||
const auto orders = { /*6, 5,*/ 4, 3, 2, 1 };
|
||||
|
||||
constexpr auto ndofs = [](int n) constexpr noexcept -> int
|
||||
{
|
||||
return (n + 1) * (n + 1) * (n + 1);
|
||||
};
|
||||
|
||||
constexpr auto inc = [](int n) constexpr noexcept -> int
|
||||
{
|
||||
return n < 160 ? 4 : n < 240 ? 8 : n < 320 ? 16 : 32;
|
||||
};
|
||||
|
||||
for (auto k : versions)
|
||||
{
|
||||
for (auto p : orders)
|
||||
{
|
||||
for (int n = 16; ndofs(n) <= MAX_NDOFS; n += inc(n))
|
||||
{
|
||||
b->Args({k, p, n});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Register kernel specializations used in the benchmarks /////////////////////
|
||||
static void AddKernelSpecializations()
|
||||
{
|
||||
using DET = QuadratureInterpolator::DetKernels;
|
||||
DET::Specialization<3, 3, 2, 2>::Add();
|
||||
DET::Specialization<3, 3, 2, 3>::Add();
|
||||
DET::Specialization<3, 3, 2, 5>::Add();
|
||||
DET::Specialization<3, 3, 2, 6>::Add();
|
||||
DET::Specialization<3, 3, 5, 5>::Add();
|
||||
// Others might exceed memory limits
|
||||
|
||||
using GRAD = QuadratureInterpolator::GradKernels;
|
||||
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 2>::Add();
|
||||
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 7>::Add();
|
||||
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 8>::Add();
|
||||
GRAD::Specialization<3, QVectorLayout::byNODES, false, 3, 2, 9>::Add();
|
||||
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 3, 2, 3>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 3, 2, 4>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 3, 2, 5>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 3, 2, 6>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 3, 2, 7>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 3, 2, 8>::Add();
|
||||
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 1, 2, 3>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 1, 4, 5>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 1, 5, 6>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 1, 6, 7>::Add();
|
||||
// GRAD::Specialization<3, QVectorLayout::byVDIM, false, 1, 7, 8>::Add();
|
||||
|
||||
using LIN = DomainLFIntegrator::AssembleKernels;
|
||||
LIN::Specialization<3, 7, 7>::Add();
|
||||
LIN::Specialization<3, 6, 6>::Add();
|
||||
LIN::Specialization<3, 8, 8>::Add();
|
||||
|
||||
using VDIFF = VectorDiffusionIntegrator::ApplyPAKernels;
|
||||
VDIFF::Specialization<3, 3, 3, 3>::Add();
|
||||
VDIFF::Specialization<3, 3, 4, 4>::Add();
|
||||
VDIFF::Specialization<3, 3, 5, 5>::Add();
|
||||
VDIFF::Specialization<3, 3, 6, 6>::Add();
|
||||
VDIFF::Specialization<3, 3, 7, 7>::Add();
|
||||
VDIFF::Specialization<3, 3, 8, 8>::Add();
|
||||
}
|
||||
|
||||
/// Globals ///////////////////////////////////////////////////////////////////
|
||||
Device *device_ptr = nullptr;
|
||||
static int gD1D = 0, gQ1D = 0;
|
||||
|
||||
/// StiffnessIntegrator ///////////////////////////////////////////////////////
|
||||
struct StiffnessIntegrator : public BilinearFormIntegrator
|
||||
{
|
||||
using mfem::NonlinearFormIntegrator::AssemblePA;
|
||||
const FiniteElementSpace *fes;
|
||||
const real_t *B, *G, *DX;
|
||||
int ne, d1d, q1d;
|
||||
Vector J0, dx;
|
||||
public:
|
||||
StiffnessIntegrator()
|
||||
{
|
||||
dbg();
|
||||
NVTX();
|
||||
StiffnessKernels::Specialization<2, 3>::Add();
|
||||
StiffnessKernels::Specialization<3, 4>::Add();
|
||||
StiffnessKernels::Specialization<4, 5>::Add();
|
||||
StiffnessKernels::Specialization<5, 6>::Add();
|
||||
StiffnessKernels::Specialization<6, 7>::Add();
|
||||
StiffnessKernels::Specialization<7, 8>::Add();
|
||||
StiffnessKernels::Specialization<9, 10>::Add();
|
||||
}
|
||||
|
||||
void AssemblePA(const FiniteElementSpace &fespace) override
|
||||
{
|
||||
NVTX();
|
||||
fes = &fespace;
|
||||
auto *mesh = fes->GetMesh();
|
||||
const int DIM = mesh->Dimension();
|
||||
ne = mesh->GetNE();
|
||||
const auto p = fes->GetFE(0)->GetOrder();
|
||||
const auto q = 2 * p + mesh->GetElementTransformation(0)->OrderW();
|
||||
const auto type = mesh->GetElementBaseGeometry(0);
|
||||
const IntegrationRule &ir = IntRules.Get(type, q);
|
||||
const int NQPT = ir.GetNPoints();
|
||||
d1d = p + 1;
|
||||
q1d = IntRules.Get(Geometry::SEGMENT, ir.GetOrder()).GetNPoints();
|
||||
MFEM_VERIFY(d1d == gD1D, "D1D mismatch: " << d1d << " != " << gD1D);
|
||||
MFEM_VERIFY(q1d == gQ1D, "Q1D mismatch: " << q1d << " != " << gQ1D);
|
||||
MFEM_VERIFY(NQPT == q1d * q1d * q1d, "");
|
||||
const DofToQuad *maps =
|
||||
&fes->GetFE(0)->GetDofToQuad(ir, DofToQuad::TENSOR);
|
||||
const GridFunction *nodes = (mesh->EnsureNodes(), mesh->GetNodes());
|
||||
const FiniteElementSpace *nfes = nodes->FESpace();
|
||||
const int nVDIM = nfes->GetVDim();
|
||||
dx.SetSize(nVDIM * DIM * NQPT * ne, Device::GetDeviceMemoryType());
|
||||
J0.SetSize(nVDIM * DIM * NQPT * ne, Device::GetDeviceMemoryType());
|
||||
dx.UseDevice(true), J0.UseDevice(true);
|
||||
B = maps->B.Read(), G = maps->G.Read(), DX = dx.Read();
|
||||
|
||||
const Operator *NR =
|
||||
nfes->GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC);
|
||||
const QuadratureInterpolator *nqi = nfes->GetQuadratureInterpolator(ir);
|
||||
nqi->SetOutputLayout(QVectorLayout::byVDIM);
|
||||
const int nd = nfes->GetFE(0)->GetDof();
|
||||
Vector xe(nVDIM * nd * ne, Device::GetDeviceMemoryType());
|
||||
NR->Mult(*nodes, (xe.UseDevice(true), xe));
|
||||
nqi->Derivatives(xe, J0);
|
||||
|
||||
const int Q1D = q1d;
|
||||
const auto w_r = ir.GetWeights().Read();
|
||||
const auto W = Reshape(w_r, q1d, q1d, q1d);
|
||||
const auto J = Reshape(J0.Read(), 3, 3, q1d, q1d, q1d, ne);
|
||||
auto DX_w = Reshape(dx.Write(), 3, 3, q1d, q1d, q1d, ne);
|
||||
|
||||
mfem::forall_3D(ne, Q1D, Q1D, Q1D,[=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qz, z, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
const real_t w = W(qx, qy, qz);
|
||||
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
|
||||
const real_t detJ = kernels::Det<3>(Jtr);
|
||||
const real_t wd = w * detJ;
|
||||
const real_t D[9] = { wd, 0.0, 0.0,
|
||||
0.0, wd, 0.0,
|
||||
0.0, 0.0, wd
|
||||
};
|
||||
real_t Jrt[9], A[9];
|
||||
kernels::CalcInverse<3>(Jtr, Jrt);
|
||||
kernels::MultABt(3, 3, 3, D, Jrt, A);
|
||||
kernels::Mult(3, 3, 3, A, Jrt, &DX_w(0, 0, qx, qy, qz, e));
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
});
|
||||
}
|
||||
|
||||
template <int T_D1D = 0, int T_Q1D = 0>
|
||||
static void StiffnessMult(const int NE, const real_t *b, const real_t *g,
|
||||
const real_t *dx, const real_t *xe, real_t *ye,
|
||||
const int d1d, const int q1d)
|
||||
{
|
||||
NVTX();
|
||||
const int D1D = T_D1D ? T_D1D : d1d;
|
||||
const int Q1D = T_Q1D ? T_Q1D : q1d;
|
||||
|
||||
constexpr int DIM = 3, VDIM = 1;
|
||||
const auto XE = Reshape(xe, D1D, D1D, D1D, VDIM, NE);
|
||||
const auto DX = Reshape(dx, 3, 3, Q1D, Q1D, Q1D, NE);
|
||||
auto YE = Reshape(ye, D1D, D1D, D1D, VDIM, NE);
|
||||
|
||||
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
|
||||
{
|
||||
constexpr int MD1 = T_D1D > 0 ? kernels::internal::SetMaxOf(T_D1D) : 32;
|
||||
constexpr int MQ1 = T_Q1D > 0 ? kernels::internal::SetMaxOf(T_Q1D) : 32;
|
||||
|
||||
MFEM_SHARED real_t smem[MQ1][MQ1];
|
||||
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
|
||||
ker::vd_regs3d_t<VDIM, DIM, MQ1> r0, r1;
|
||||
|
||||
ker::LoadMatrix(D1D, Q1D, b, sB);
|
||||
ker::LoadMatrix(D1D, Q1D, g, sG);
|
||||
|
||||
ker::LoadDofs3d(e, D1D, XE, r0);
|
||||
ker::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1);
|
||||
|
||||
for (int qz = 0; qz < Q1D; qz++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
|
||||
{
|
||||
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
|
||||
{
|
||||
real_t v[3], u[3] = { r1[0][0][qz][qy][qx],
|
||||
r1[0][1][qz][qy][qx],
|
||||
r1[0][2][qz][qy][qx]
|
||||
};
|
||||
const real_t *dx = &DX(0, 0, qx, qy, qz, e);
|
||||
kernels::Mult(3, 3, dx, u, v);
|
||||
r0[0][0][qz][qy][qx] = v[0];
|
||||
r0[0][1][qz][qy][qx] = v[1];
|
||||
r0[0][2][qz][qy][qx] = v[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
ker::GradTranspose3d(D1D, Q1D, smem, sB, sG, r0, r1);
|
||||
ker::WriteDofs3d(e, D1D, r1, YE);
|
||||
});
|
||||
}
|
||||
|
||||
using StiffnessKernelType = decltype(&StiffnessMult<>);
|
||||
MFEM_REGISTER_KERNELS(StiffnessKernels, StiffnessKernelType, (int, int));
|
||||
|
||||
void AddMultPA(const Vector &x, Vector &y) const override
|
||||
{
|
||||
StiffnessKernels::Run(d1d, q1d, ne, B, G, DX, x.Read(), y.ReadWrite(),
|
||||
d1d, q1d);
|
||||
}
|
||||
};
|
||||
|
||||
template <int D1D, int Q1D>
|
||||
StiffnessIntegrator::StiffnessKernelType
|
||||
StiffnessIntegrator::StiffnessKernels::Kernel()
|
||||
{
|
||||
return StiffnessMult<D1D, Q1D>;
|
||||
}
|
||||
|
||||
StiffnessIntegrator::StiffnessKernelType
|
||||
StiffnessIntegrator::StiffnessKernels::Fallback(int d1d, int q1d)
|
||||
{
|
||||
dbg("\x1b[33mFallback d1d:{} q1d:{}", d1d, q1d);
|
||||
MFEM_ABORT("No kernel for d1d=" << d1d << " q1d=" << q1d);
|
||||
// return StiffnessMult<>;
|
||||
}
|
||||
|
||||
/// BakeOff ///////////////////////////////////////////////////////////////////
|
||||
template <int VDIM, bool GLL>
|
||||
struct BakeOff
|
||||
{
|
||||
static constexpr int DIM = 3;
|
||||
const int p, c, q, n, nx, ny, nz;
|
||||
Mesh smesh;
|
||||
ParMesh pmesh;
|
||||
H1_FECollection fec;
|
||||
ParFiniteElementSpace pfes;
|
||||
const Geometry::Type geom_type;
|
||||
IntegrationRules irs;
|
||||
const IntegrationRule *ir;
|
||||
ConstantCoefficient one;
|
||||
Vector uvec;
|
||||
VectorConstantCoefficient unit_vec;
|
||||
const int dofs;
|
||||
ParGridFunction *nodes;
|
||||
ParFiniteElementSpace& mfes;
|
||||
ParGridFunction x, y;
|
||||
ParBilinearForm a;
|
||||
std::unique_ptr<DifferentiableOperator> dop;
|
||||
const int elem_size, total_size, d1d, q1d;
|
||||
UniformParameterSpace qd_ps;
|
||||
ParameterFunction qdata;
|
||||
|
||||
double mdofs{};
|
||||
|
||||
BakeOff(int p, int side):
|
||||
p(p), c(side), q(2 * p + (GLL ? -1 : 3)),
|
||||
n((assert(c >= p), c / p)),
|
||||
nx(n + (p * (n + 1) * p * n * p * n < c * c * c ? 1 : 0)),
|
||||
ny(n + (p * (n + 1) * p * (n + 1) * p * n < c * c * c ? 1 : 0)),
|
||||
nz(n),
|
||||
smesh(Mesh::MakeCartesian3D(nx, ny, nz, Element::HEXAHEDRON)),
|
||||
pmesh(MPI_COMM_WORLD, (smesh.EnsureNodes(), smesh)),
|
||||
fec(p, DIM, BasisType::GaussLobatto),
|
||||
pfes(&pmesh, &fec, VDIM),
|
||||
geom_type(pmesh.GetTypicalElementGeometry()),
|
||||
irs(0, GLL ? Quadrature1D::GaussLobatto : Quadrature1D::GaussLegendre),
|
||||
ir(&irs.Get(geom_type, q)),
|
||||
one(1.0),
|
||||
uvec(DIM),
|
||||
unit_vec((uvec = 1.0, uvec /= uvec.Norml2(), uvec)),
|
||||
dofs(pfes.GetTrueVSize()),
|
||||
nodes(static_cast<ParGridFunction*>(pmesh.GetNodes())),
|
||||
mfes(*nodes->ParFESpace()),
|
||||
x(&pfes),
|
||||
y(&pfes),
|
||||
a(&pfes),
|
||||
elem_size(DIM * DIM * ir->GetNPoints()),
|
||||
total_size(elem_size * pmesh.GetNE()),
|
||||
d1d(p + 1),
|
||||
q1d(IntRules.Get(Geometry::SEGMENT, ir->GetOrder()).GetNPoints()),
|
||||
qd_ps(pmesh, *ir, DIM*DIM),
|
||||
qdata(qd_ps)
|
||||
{
|
||||
NVTX_MARK_FUNCTION;
|
||||
// pmesh.SetCurvature(p);
|
||||
smesh.Clear();
|
||||
x = 0.0;
|
||||
|
||||
gD1D = d1d, gQ1D = q1d;
|
||||
// dbg("D1D: {}, Q1D: {}", gD1D, gQ1D);
|
||||
qdata.UseDevice(true);
|
||||
assert(q1d*q1d*q1d == ir->GetNPoints());
|
||||
}
|
||||
|
||||
virtual void benchmark() = 0;
|
||||
|
||||
[[nodiscard]] double SumMdofs() const noexcept { return mdofs; }
|
||||
|
||||
[[nodiscard]] double MDofs() const noexcept { return 1e-6 * dofs; }
|
||||
};
|
||||
|
||||
/// Q-Functions ///////////////////////////////////////////////////////////////
|
||||
template<int DIM>
|
||||
struct MF
|
||||
{
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto operator()(const tensor<real_t, DIM>& Gu,
|
||||
const tensor<real_t, DIM, DIM>& J,
|
||||
const real_t& w) const
|
||||
{
|
||||
auto invJ = inv(J);
|
||||
return tuple{((Gu * invJ)) * transpose(invJ) * det(J) * w};
|
||||
}
|
||||
};
|
||||
|
||||
template<int DIM>
|
||||
struct PASetup
|
||||
{
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto operator()([[maybe_unused]] const real_t &u,
|
||||
const tensor<real_t, DIM, DIM> &J,
|
||||
const real_t &w)const
|
||||
{
|
||||
return tuple{inv(J) * transpose(inv(J)) * det(J) * w};
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<int DIM>
|
||||
struct PAApply
|
||||
{
|
||||
MFEM_HOST_DEVICE inline
|
||||
auto operator()(const tensor<real_t, DIM> &Gu,
|
||||
const tensor<real_t, DIM, DIM> &q) const
|
||||
{
|
||||
return tuple{q * Gu};
|
||||
};
|
||||
};
|
||||
|
||||
/// Diffusion /////////////////////////////////////////////////////////////////
|
||||
template <int VDIM = 1, bool GLL = false>
|
||||
struct Diffusion : public BakeOff<VDIM, GLL>
|
||||
{
|
||||
static constexpr int DIM = 3;
|
||||
static constexpr int U = 0, Ξ = 1, Q = 2;
|
||||
|
||||
const real_t rtol = 0.0;
|
||||
const int max_it = 32, print_lvl = -1;
|
||||
|
||||
Array<int> ess_tdof_list, ess_bdr, all_domain_attr;
|
||||
ParLinearForm b;
|
||||
FieldDescriptor u_fd, Ξ_fd, q_fd;
|
||||
std::vector<FieldDescriptor> u_sol, q_param, Ξ_q_params;
|
||||
OperatorPtr A;
|
||||
Operator *A_ptr;
|
||||
Vector B, X;
|
||||
CGSolver cg;
|
||||
|
||||
using base = BakeOff<VDIM, GLL>;
|
||||
using base::a;
|
||||
using base::ir;
|
||||
using base::one;
|
||||
using base::pmesh;
|
||||
using base::pfes;
|
||||
using base::mfes;
|
||||
using base::x;
|
||||
using base::y;
|
||||
using base::mdofs;
|
||||
using base::dop;
|
||||
using base::nodes;
|
||||
using base::qdata;
|
||||
using base::qd_ps;
|
||||
using base::dofs;
|
||||
|
||||
Diffusion(int version, int order, int side):
|
||||
BakeOff<VDIM, GLL>(order, side),
|
||||
ess_bdr(pmesh.bdr_attributes.Max()),
|
||||
all_domain_attr(pmesh.bdr_attributes.Max()),
|
||||
b(&pfes),
|
||||
u_fd{U, &pfes}, Ξ_fd{Ξ, &mfes}, q_fd{Q, &qd_ps},
|
||||
u_sol{u_fd},
|
||||
q_param {q_fd},
|
||||
Ξ_q_params {Ξ_fd, q_fd},
|
||||
cg(MPI_COMM_WORLD)
|
||||
{
|
||||
// dbg("pmesh.bdr_attributes.Max():{}",pmesh.bdr_attributes.Max());
|
||||
static_assert(VDIM == 1 && GLL == false);
|
||||
|
||||
ess_bdr = 1;
|
||||
all_domain_attr = 1;
|
||||
pfes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
|
||||
b.AddDomainIntegrator(new DomainLFIntegrator(this->one));
|
||||
b.UseFastAssembly(true);
|
||||
b.Assemble();
|
||||
|
||||
if (version < 2) // standard, new PA regs
|
||||
{
|
||||
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
if (version == 0) { a.AddDomainIntegrator(new DiffusionIntegrator(ir)); }
|
||||
if (version == 1) { a.AddDomainIntegrator(new StiffnessIntegrator()); }
|
||||
a.Assemble();
|
||||
a.FormLinearSystem(ess_tdof_list, x, b, A, X, B);
|
||||
if (version == 0)
|
||||
{
|
||||
BilinearFormIntegrator *bfi = a.GetDBFI()->operator[](0);
|
||||
auto *di = dynamic_cast<DiffusionIntegrator*>(bfi);
|
||||
assert(di);
|
||||
const int d1d = di->dofs1D, q1d = di->quad1D;
|
||||
// dbg("\x1b[33md1d: {} q1d: {}", d1d, q1d);
|
||||
MFEM_VERIFY(d1d == gD1D, "D1D mismatch: " << d1d << " != " << gD1D);
|
||||
MFEM_VERIFY(q1d == gQ1D, "Q1D mismatch: " << q1d << " != " << gQ1D);
|
||||
}
|
||||
}
|
||||
else if (version == 2) // 2: MF ∂fem
|
||||
{
|
||||
dbg("MF ∂fem");
|
||||
auto solutions = std::vector{FieldDescriptor{U, &pfes}};
|
||||
auto parameters = std::vector{FieldDescriptor{Ξ, &mfes}};
|
||||
dop = std::make_unique<DifferentiableOperator>(solutions, parameters, pmesh);
|
||||
dop->SetParameters({nodes});
|
||||
MF<DIM> mf_apply;
|
||||
dop->AddDomainIntegrator(mf_apply,
|
||||
tuple{Gradient<U>{}, Gradient<Ξ>{}, Weight{}},
|
||||
tuple{Gradient<U>{}},
|
||||
*ir, ess_bdr);
|
||||
dop->FormLinearSystem(ess_tdof_list, x, b, A_ptr, X, B);
|
||||
A.Reset(A_ptr);
|
||||
}
|
||||
else if (version == 3) // PA ∂fem
|
||||
{
|
||||
dbg("PA ∂fem");
|
||||
auto w = Weight{};
|
||||
auto q = Identity<Q> {};
|
||||
auto u = Identity<U> {};
|
||||
auto Gu = Gradient<U> {};
|
||||
auto GΞ = Gradient<Ξ> {};
|
||||
tuple Gu_q = {Gu, q};
|
||||
tuple u_J_w = {u, GΞ, w};
|
||||
|
||||
DifferentiableOperator dSetup(u_sol, Ξ_q_params, pmesh);
|
||||
PASetup<DIM> pa_setup;
|
||||
dSetup.AddDomainIntegrator(pa_setup, u_J_w, tuple{q}, *ir, ess_bdr);
|
||||
dSetup.SetParameters({nodes, &qdata});
|
||||
X.SetSize(pfes.GetTrueVSize());
|
||||
pfes.GetRestrictionMatrix()->Mult(x, X);
|
||||
dSetup.Mult(X, qdata);
|
||||
dop = std::make_unique<DifferentiableOperator>(u_sol, q_param, pmesh);
|
||||
PAApply<DIM> pa_apply;
|
||||
dop->AddDomainIntegrator(pa_apply, Gu_q, tuple{Gu}, *ir, ess_bdr);
|
||||
dop->SetParameters({ &qdata });
|
||||
|
||||
dop->FormLinearSystem(ess_tdof_list, x, b, A_ptr, X, B);
|
||||
A.Reset(A_ptr);
|
||||
}
|
||||
else { MFEM_ABORT("Invalid version"); }
|
||||
|
||||
cg.SetOperator(*A);
|
||||
cg.iterative_mode = false;
|
||||
cg.SetAbsTol(0.0);
|
||||
if (dofs < 128 * 1024) // check
|
||||
{
|
||||
dbg("check");
|
||||
cg.SetPrintLevel(-1);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetRelTol(1e-8);
|
||||
cg.Mult(B, X);
|
||||
MFEM_VERIFY(cg.GetConverged(), "❌❌❌ CG solver did not converge.");
|
||||
// mfem::out << "✅" << std::endl;
|
||||
}
|
||||
cg.SetRelTol(rtol);
|
||||
cg.SetMaxIter(max_it);
|
||||
cg.SetPrintLevel(print_lvl);
|
||||
benchmark();
|
||||
mdofs = 0.0;
|
||||
}
|
||||
|
||||
void benchmark() override
|
||||
{
|
||||
cg.Mult(B, X);
|
||||
MFEM_DEVICE_SYNC;
|
||||
mdofs += this->MDofs() * cg.GetNumIterations();
|
||||
}
|
||||
};
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
#define BakeOff_Problem(i, Problem) \
|
||||
static void BP##i(bm::State &state) \
|
||||
{ \
|
||||
const auto version = static_cast<int>(state.range(0)); \
|
||||
const auto order = static_cast<int>(state.range(1)); \
|
||||
const auto side = static_cast<int>(state.range(2)); \
|
||||
Problem ker(version, order, side); \
|
||||
while (state.KeepRunning()) { ker.benchmark(); } \
|
||||
bm::Counter::Flags flags = bm::Counter::kIsRate; \
|
||||
state.counters["MDof/s"] = bm::Counter(ker.SumMdofs(), flags); \
|
||||
state.counters["Dofs"] = bm::Counter(ker.dofs); \
|
||||
state.counters["p"] = bm::Counter(order); \
|
||||
state.counters["version"] = bm::Counter(version); \
|
||||
} \
|
||||
BENCHMARK(BP##i) \
|
||||
->Apply(CustomArguments) \
|
||||
->Unit(bm::kMillisecond)
|
||||
|
||||
BakeOff_Problem(3, Diffusion);
|
||||
|
||||
/// main //////////////////////////////////////////////////////////////////////
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
dbg();
|
||||
static mfem::MPI_Session mpi(argc, argv);
|
||||
|
||||
bm::ConsoleReporter CR;
|
||||
bm::Initialize(&argc, argv);
|
||||
|
||||
AddKernelSpecializations();
|
||||
info();
|
||||
|
||||
// Device setup, cpu by default
|
||||
std::string device_config = "cpu";
|
||||
const auto global_context = bmi::GetGlobalContext();
|
||||
if (global_context != nullptr)
|
||||
{
|
||||
const auto device = global_context->find("device");
|
||||
if (device != global_context->end())
|
||||
{
|
||||
mfem::out << device->first << " : " << device->second << std::endl;
|
||||
device_config = device->second;
|
||||
}
|
||||
}
|
||||
dbg("device_config: {}", device_config);
|
||||
Device device(device_config.c_str());
|
||||
device_ptr = &device;
|
||||
device.Print();
|
||||
|
||||
if (bm::ReportUnrecognizedArguments(argc, argv)) { return EXIT_FAILURE; }
|
||||
|
||||
bm::RunSpecifiedBenchmarks(&CR);
|
||||
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_BENCHMARK
|
||||
@@ -20,8 +20,8 @@ CONFIG_MK = $(or $(wildcard $(MFEM_BUILD_DIR)/config/config.mk),\
|
||||
MFEM_LIB_FILE = mfem_is_not_built
|
||||
-include $(CONFIG_MK)
|
||||
|
||||
SEQ_TESTS = bench_assembly_levels bench_ceed bench_dg_amr bench_elasticity \
|
||||
bench_tmop bench_vector bench_virtuals
|
||||
SEQ_TESTS = bench_assembly_levels bench_ceed bench_dfem bench_dg_amr \
|
||||
bench_elasticity bench_tmop bench_vector bench_virtuals
|
||||
PAR_TESTS =
|
||||
ifeq ($(MFEM_USE_MPI),NO)
|
||||
TESTS = $(SEQ_TESTS)
|
||||
|
||||
Reference in New Issue
Block a user