Compare commits
69
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c6ee709eef | ||
|
|
0274b67ff6 | ||
|
|
38da503958 | ||
|
|
6c4f9c69a9 | ||
|
|
0702eb2fb2 | ||
|
|
7a62a7fd4a | ||
|
|
0f94d484a4 | ||
|
|
4b5798f905 | ||
|
|
87240f5619 | ||
|
|
3ccaa48cd4 | ||
|
|
a08928dd97 | ||
|
|
a0694d5825 | ||
|
|
a3ecbef0ec | ||
|
|
db5bc1725e | ||
|
|
6d50ebc9c3 | ||
|
|
e713913177 | ||
|
|
dd9898af3b | ||
|
|
0435ef5dac | ||
|
|
79e5020a18 | ||
|
|
adc81c7f5d | ||
|
|
18676c61b7 | ||
|
|
8f8deab121 | ||
|
|
8b0c779320 | ||
|
|
a012769434 | ||
|
|
ce5517b9af | ||
|
|
70ae37d5f0 | ||
|
|
feac718e95 | ||
|
|
9be8c15cf8 | ||
|
|
08ba45fca3 | ||
|
|
3aacfbfab0 | ||
|
|
2a60b998c7 | ||
|
|
91a168929f | ||
|
|
1369f5e189 | ||
|
|
993e4fbbe1 | ||
|
|
07d8a17abe | ||
|
|
5531b82dbc | ||
|
|
d9f60f401b | ||
|
|
fb876ba3a1 | ||
|
|
32a94b438f | ||
|
|
917978d310 | ||
|
|
a62302b4cb | ||
|
|
f5d2b82839 | ||
|
|
1382f8aa1f | ||
|
|
f76a884d15 | ||
|
|
5c56659e46 | ||
|
|
4bcd4586ba | ||
|
|
083d42c6ce | ||
|
|
dbc3458db0 | ||
|
|
4f694287ae | ||
|
|
5d8fbfee93 | ||
|
|
7799753053 | ||
|
|
cb6db58ad3 | ||
|
|
5da2bfc23d | ||
|
|
466a771ab0 | ||
|
|
a317e1a17d | ||
|
|
fcbde98cb6 | ||
|
|
80cb02328c | ||
|
|
1847e460cf | ||
|
|
1cd46aa768 | ||
|
|
11edee7aca | ||
|
|
28115e5de2 | ||
|
|
20954328c3 | ||
|
|
41e92219ee | ||
|
|
a0c3620618 | ||
|
|
9c791bed5a | ||
|
|
42d0fc17a1 | ||
|
|
2d2da417bb | ||
|
|
2e86ccb948 | ||
|
|
c7fe1ff1f4 |
+9
-1
@@ -522,7 +522,10 @@ endif()
|
||||
|
||||
# Enzyme
|
||||
if (MFEM_USE_ENZYME)
|
||||
find_package(ENZYME REQUIRED)
|
||||
find_package(Enzyme REQUIRED HINTS ${ENZYME_DIR})
|
||||
message(STATUS "Enzyme found in ${ENZYME_DIR}.")
|
||||
set(ENZYME_INCLUDE_DIRS ${ENZYME_DIR}/include)
|
||||
set(ENZYME_FOUND 1)
|
||||
endif()
|
||||
|
||||
# MFEM_TIMER_TYPE
|
||||
@@ -629,6 +632,11 @@ set(MFEM_INSTALL_DIR ${CMAKE_INSTALL_PREFIX} CACHE PATH
|
||||
mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
|
||||
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES})
|
||||
|
||||
if (MFEM_USE_ENZYME)
|
||||
target_link_libraries(mfem PUBLIC ClangEnzymeFlags)
|
||||
endif()
|
||||
|
||||
if (MINGW)
|
||||
target_link_libraries(mfem PRIVATE ws2_32)
|
||||
endif()
|
||||
|
||||
@@ -1,27 +0,0 @@
|
||||
# Copyright (c) 2010-2024, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
message(STATUS "Looking for ENZYME ...")
|
||||
message(STATUS " in ENZYME_DIR = ${ENZYME_DIR}")
|
||||
|
||||
# Make sure the directory and version combination works. Do nothing otherwise.
|
||||
if(EXISTS "${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so")
|
||||
message(STATUS "Found ENZYME: ${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so")
|
||||
|
||||
# Set ENZYME_FOUND
|
||||
set(ENZYME_FOUND TRUE CACHE BOOL "ENZYME was found." FORCE)
|
||||
|
||||
# Set CXX flags to accommodate the Enzyme Clang plugin
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Xclang -load -Xclang ${ENZYME_DIR}/ClangEnzyme-${ENZYME_VERSION}.so -mllvm -enzyme-loose-types=1")
|
||||
set(MFEM_USE_ENZYME YES)
|
||||
else()
|
||||
|
||||
endif()
|
||||
@@ -0,0 +1,35 @@
|
||||
MFEM mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see mesh/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
#
|
||||
|
||||
dimension
|
||||
2
|
||||
|
||||
elements
|
||||
1
|
||||
1 3 0 1 2 3
|
||||
|
||||
boundary
|
||||
4
|
||||
1 1 0 1
|
||||
2 1 1 2
|
||||
3 1 2 3
|
||||
4 1 3 0
|
||||
|
||||
vertices
|
||||
4
|
||||
2
|
||||
0 0
|
||||
1 0.3
|
||||
1.4 1.2
|
||||
0.25 1.34
|
||||
@@ -50,6 +50,27 @@ list(APPEND ALL_EXE_SRCS
|
||||
|
||||
if (MFEM_USE_MPI)
|
||||
list(APPEND ALL_EXE_SRCS
|
||||
dfem_poisson.cpp
|
||||
dfem_stokes.cpp
|
||||
enzyme_interface_smoketest.cpp
|
||||
test_dfem_dual.cpp
|
||||
test_dfem.cpp
|
||||
dfem_laghos.cpp
|
||||
dfem_minimal_example.cpp
|
||||
dfem_test_diffusion_2d.cpp
|
||||
dfem_test_diffusion_3d.cpp
|
||||
dfem_test_ordering.cpp
|
||||
dfem_test_vector_diffusion.cpp
|
||||
dfem_test_elasticity.cpp
|
||||
dfem_test_nonlinear_elasticity_3d.cpp
|
||||
dfem_test_nonlinear_diffusion_3d.cpp
|
||||
dfem_test_interpolate_linear_scalar.cpp
|
||||
dfem_test_interpolate_linear_scalar_3d.cpp
|
||||
dfem_test_interpolate_gradient_linear_scalar_3d.cpp
|
||||
dfem_test_mass_scalar_3d.cpp
|
||||
dfem_test_mass_scalar_2d.cpp
|
||||
dfem_test_interpolate_linear_vector.cpp
|
||||
dfem_test_interpolate_linear_vector_3d.cpp
|
||||
ex0p.cpp
|
||||
ex1p.cpp
|
||||
ex2p.cpp
|
||||
@@ -110,6 +131,16 @@ include_directories(BEFORE ${PROJECT_BINARY_DIR})
|
||||
# Add one executable per cpp file
|
||||
add_mfem_examples(ALL_EXE_SRCS)
|
||||
|
||||
target_link_libraries(dfem_poisson ClangEnzymeFlags)
|
||||
target_link_libraries(dfem_stokes ClangEnzymeFlags)
|
||||
target_link_libraries(enzyme_interface_smoketest ClangEnzymeFlags)
|
||||
target_link_libraries(test_dfem ClangEnzymeFlags)
|
||||
target_link_libraries(dfem_laghos ClangEnzymeFlags)
|
||||
target_link_libraries(dfem_minimal_example ClangEnzymeFlags)
|
||||
target_link_libraries(dfem_test_diffusion_3d ClangEnzymeFlags)
|
||||
target_link_libraries(dfem_test_nonlinear_diffusion_3d ClangEnzymeFlags)
|
||||
target_link_libraries(dfem_test_nonlinear_elasticity_3d ClangEnzymeFlags)
|
||||
|
||||
# Add a test for each example
|
||||
if (MFEM_ENABLE_TESTING)
|
||||
foreach(SRC_FILE ${ALL_EXE_SRCS})
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
#pragma once
|
||||
|
||||
#include "dfem_differentiable_operator.hpp"
|
||||
@@ -0,0 +1,232 @@
|
||||
template <
|
||||
typename kernels_tuple,
|
||||
size_t num_solutions,
|
||||
size_t num_parameters,
|
||||
size_t num_fields,
|
||||
size_t num_kernels
|
||||
>
|
||||
template <
|
||||
typename kernel_t
|
||||
>
|
||||
void DifferentiableOperator<kernels_tuple,
|
||||
num_solutions,
|
||||
num_parameters,
|
||||
num_fields,
|
||||
num_kernels>::Action::create_action_callback(
|
||||
kernel_t kernel,
|
||||
mult_func_t &func)
|
||||
{
|
||||
using entity_t = typename kernel_t::entity_t;
|
||||
|
||||
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
|
||||
|
||||
constexpr int hardcoded_output_idx = 0;
|
||||
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
|
||||
|
||||
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
|
||||
element_dof_ordering);
|
||||
|
||||
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
|
||||
|
||||
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
|
||||
const int num_entities = GetNumEntities<entity_t>(op.mesh);
|
||||
const int num_qp = op.integration_rule.GetNPoints();
|
||||
|
||||
// All solutions T-vector sizes make up the width of the operator, since
|
||||
// they are explicitly provided in Mult() for example.
|
||||
|
||||
op.width = GetTrueVSize(op.fields[test_space_field_idx]);
|
||||
op.residual_lsize = GetVSize(op.fields[test_space_field_idx]);
|
||||
|
||||
if constexpr (std::is_same_v<decltype(output_fop), One>)
|
||||
{
|
||||
op.height = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
op.height = op.residual_lsize;
|
||||
}
|
||||
|
||||
residual_l.SetSize(op.residual_lsize);
|
||||
|
||||
// assume only a single element type for now
|
||||
std::vector<const DofToQuad*> dtq;
|
||||
for (const auto &field : op.fields)
|
||||
{
|
||||
dtq.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
|
||||
doftoquad_mode));
|
||||
}
|
||||
const int q1d = (int)floor(pow(num_qp, 1.0/op.mesh.Dimension()) + 0.5);
|
||||
|
||||
residual_e.SetSize(R->Height());
|
||||
|
||||
const int residual_size_on_qp = GetSizeOnQP<entity_t>(
|
||||
mfem::get<hardcoded_output_idx>(kernel.outputs),
|
||||
op.fields[test_space_field_idx]);
|
||||
|
||||
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
|
||||
kinput_to_field);
|
||||
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
|
||||
koutput_to_field);
|
||||
|
||||
auto input_fops = create_bare_fops(kernel.inputs);
|
||||
auto output_fops = create_bare_fops(kernel.outputs);
|
||||
|
||||
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
|
||||
const int test_op_dim =
|
||||
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
|
||||
mfem::get<hardcoded_output_idx>(output_fops).vdim;
|
||||
const int num_test_dof = R->Height() /
|
||||
mfem::get<hardcoded_output_idx>(output_fops).vdim /
|
||||
num_entities;
|
||||
|
||||
auto ir_weights = Reshape(this->op.integration_rule.GetWeights().Read(),
|
||||
num_qp);
|
||||
|
||||
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
|
||||
output_dtq_maps,
|
||||
op.fields,
|
||||
num_entities,
|
||||
kernel.inputs,
|
||||
num_qp,
|
||||
input_size_on_qp,
|
||||
residual_size_on_qp);
|
||||
|
||||
Vector shmem_cache(shmem_info.total_size);
|
||||
|
||||
print_shared_memory_info(shmem_info);
|
||||
|
||||
func = [=](Vector &ye_mem) mutable
|
||||
{
|
||||
restriction<entity_t>(op.solutions, solutions_l, this->fields_e,
|
||||
op.element_dof_ordering);
|
||||
restriction<entity_t>(op.parameters, parameters_l, this->fields_e,
|
||||
op.element_dof_ordering,
|
||||
op.solutions.size());
|
||||
|
||||
auto ye = Reshape(ye_mem.ReadWrite(), test_vdim, num_test_dof, num_entities);
|
||||
auto wrapped_fields_e = wrap_fields(this->fields_e, shmem_info.field_sizes, num_entities);
|
||||
|
||||
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
|
||||
{
|
||||
// printf("\ne: %d\n", e);
|
||||
// tic();
|
||||
auto input_dtq_shmem = load_dtq_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
|
||||
shmem_info.input_dtq_sizes,
|
||||
input_dtq_maps);
|
||||
|
||||
auto output_dtq_shmem = load_dtq_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
|
||||
shmem_info.output_dtq_sizes,
|
||||
output_dtq_maps);
|
||||
|
||||
auto fields_shmem = load_field_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::FIELD],
|
||||
shmem_info.field_sizes,
|
||||
kinput_to_field,
|
||||
wrapped_fields_e,
|
||||
e);
|
||||
|
||||
// These methods don't copy, they simply create a `DeviceTensor` object
|
||||
// that points to correct chunks of the shared memory pool.
|
||||
auto input_shmem = load_input_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::INPUT],
|
||||
shmem_info.input_sizes,
|
||||
num_qp);
|
||||
|
||||
auto residual_shmem = load_residual_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::OUTPUT],
|
||||
shmem_info.residual_size,
|
||||
num_qp);
|
||||
|
||||
auto scratch_mem = load_scratch_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::TEMP],
|
||||
shmem_info.temp_sizes);
|
||||
|
||||
MFEM_SYNC_THREAD;
|
||||
// printf("shmem load elapsed: %.1fus\n", toc() * 1e6);
|
||||
|
||||
// tic();
|
||||
map_fields_to_quadrature_data<TensorProduct>(
|
||||
input_shmem, fields_shmem, input_dtq_shmem, input_fops, ir_weights, scratch_mem,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
// printf("interpolate elapsed: %.1fus\n", toc() * 1e6);
|
||||
|
||||
// tic();
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
const int q = qx + q1d * (qy + q1d * qz);
|
||||
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
auto r = Reshape(&residual_shmem(0, q), residual_size_on_qp);
|
||||
apply_kernel(r, kernel.func, kernel_args, input_shmem, q);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// printf("qf elapsed: %.1fus\n", toc() * 1e6);
|
||||
|
||||
// tic();
|
||||
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
|
||||
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
|
||||
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
|
||||
mfem::get<0>(output_fops),
|
||||
output_dtq_shmem[hardcoded_output_idx],
|
||||
scratch_mem);
|
||||
// printf("integrate elapsed: %.1fus\n", toc() * 1e6);
|
||||
|
||||
}, num_entities, q1d, q1d, q1d, shmem_info.total_size, shmem_cache.ReadWrite());
|
||||
|
||||
if constexpr (std::is_same_v<decltype(output_fop), None>)
|
||||
{
|
||||
residual_l = ye_mem;
|
||||
}
|
||||
else
|
||||
{
|
||||
R->MultTranspose(ye_mem, residual_l);
|
||||
}
|
||||
};
|
||||
|
||||
if constexpr (std::is_same_v<decltype(output_fop), None>)
|
||||
{
|
||||
prolongation_transpose = [&](Vector &r_local, Vector &y)
|
||||
{
|
||||
y = r_local;
|
||||
};
|
||||
}
|
||||
else if constexpr (std::is_same_v<decltype(output_fop), One>)
|
||||
{
|
||||
prolongation_transpose = [&](Vector &r_local, Vector &y)
|
||||
{
|
||||
double local_sum = r_local.Sum();
|
||||
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
|
||||
op.mesh.GetComm());
|
||||
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
|
||||
};
|
||||
}
|
||||
else
|
||||
{
|
||||
auto P = get_prolongation(op.fields[test_space_field_idx]);
|
||||
prolongation_transpose = [P](const Vector &r_local, Vector &y)
|
||||
{
|
||||
P->MultTranspose(r_local, y);
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,308 @@
|
||||
template <
|
||||
typename kernels_tuple,
|
||||
size_t num_solutions,
|
||||
size_t num_parameters,
|
||||
size_t num_fields,
|
||||
size_t num_kernels
|
||||
>
|
||||
template <
|
||||
size_t derivative_idx
|
||||
>
|
||||
template <
|
||||
typename kernel_t
|
||||
>
|
||||
void DifferentiableOperator<kernels_tuple,
|
||||
num_solutions,
|
||||
num_parameters,
|
||||
num_fields,
|
||||
num_kernels>::Derivative<derivative_idx>::assemble_hypreparmatrix_impl(
|
||||
kernel_t kernel, HypreParMatrix &A)
|
||||
{
|
||||
using entity_t = typename kernel_t::entity_t;
|
||||
|
||||
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.outputs,
|
||||
std::make_index_sequence<kernel.num_koutputs> {});
|
||||
|
||||
auto output_fop = std::get<0>(kernel.outputs);
|
||||
|
||||
constexpr int hardcoded_output_idx = 0;
|
||||
|
||||
int num_qp = op.integration_rule.GetNPoints();;
|
||||
int num_el = 0;
|
||||
int dimension = 0;
|
||||
if constexpr (std::is_same_v<entity_t, Entity::Element>)
|
||||
{
|
||||
num_el = op.mesh.GetNE();
|
||||
dimension = op.dim;
|
||||
}
|
||||
else if (std::is_same_v<entity_t, Entity::Face>)
|
||||
{
|
||||
num_el = op.mesh.GetNumFacesWithGhost();
|
||||
dimension = op.dim - 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(always_false<entity_t>, "not implemented");
|
||||
}
|
||||
|
||||
std::vector<const DofToQuad*> dtqmaps;
|
||||
for (const auto &field : op.fields)
|
||||
{
|
||||
dtqmaps.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
|
||||
doftoquad_mode));
|
||||
}
|
||||
|
||||
// Allocate memory for fields on quadrature points
|
||||
auto input_qp_mem = create_input_qp_memory(num_qp, kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto directions_qp_mem = create_input_qp_memory(num_qp, kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
for (auto &d_qp_mem : directions_qp_mem)
|
||||
{
|
||||
d_qp_mem = 0.0;
|
||||
}
|
||||
|
||||
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
|
||||
bool no_kinput_is_dependent = true;
|
||||
for (int i = 0; i < kinput_is_dependent.size(); i++)
|
||||
{
|
||||
if (kinput_to_field[i] == derivative_idx)
|
||||
{
|
||||
no_kinput_is_dependent = false;
|
||||
kinput_is_dependent[i] = true;
|
||||
// out << "function input " << i << " is dependent on "
|
||||
// << op.fields[kinput_to_field[i]].field_label << "\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
kinput_is_dependent[i] = false;
|
||||
}
|
||||
}
|
||||
|
||||
if (no_kinput_is_dependent)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
|
||||
DeviceTensor<1, const double> integration_weights(
|
||||
this->op.integration_rule.GetWeights().Read(), num_qp);
|
||||
|
||||
Vector zero;
|
||||
GeometricFactorMaps geometric_factors
|
||||
{
|
||||
DeviceTensor<3, const double>(zero.Read(), 0, 0, 0)
|
||||
};
|
||||
|
||||
// fields interpolated to the quadrature points in the order of
|
||||
// kernel function arguments
|
||||
auto input_qp = map_inputs_to_memory(input_qp_mem, num_qp,
|
||||
kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto directions_qp = map_inputs_to_memory(directions_qp_mem, num_qp,
|
||||
kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto input_dtq_ops = create_dtq_operators<entity_t>(kernel.inputs, dtqmaps,
|
||||
kinput_to_field);
|
||||
auto dependent_input_dtq_ops = create_dtq_operators_conditional<entity_t>(
|
||||
kernel.inputs,
|
||||
dtqmaps,
|
||||
kinput_to_field,
|
||||
kinput_is_dependent, std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto output_dtq_ops = create_dtq_operators<entity_t>(kernel.outputs, dtqmaps,
|
||||
koutput_to_field);
|
||||
|
||||
constexpr int fixed_output_idx = 0;
|
||||
auto Bv = output_dtq_ops[fixed_output_idx];
|
||||
auto [num_test_qp, test_op_dim, num_test_dof] = Bv.GetShape();
|
||||
const int test_vdim = std::get<0>(kernel.outputs).vdim;
|
||||
|
||||
const int num_trial_dof = dependent_input_dtq_ops[0].GetShape()[2];
|
||||
int trial_vdim = 0;
|
||||
for (int i = 0; i < kinput_is_dependent.size(); i++)
|
||||
{
|
||||
if (kinput_is_dependent[i])
|
||||
{
|
||||
trial_vdim = GetVDim(op.fields[kinput_to_field[i]]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// All trial operators dimensions accumulated
|
||||
int total_trial_op_dim = 0;
|
||||
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
|
||||
{
|
||||
total_trial_op_dim += dependent_input_dtq_ops[s].GetShape()[1];
|
||||
}
|
||||
|
||||
Vector a_qp_mem(test_vdim * test_op_dim * trial_vdim * total_trial_op_dim *
|
||||
num_qp *
|
||||
num_el);
|
||||
const auto a_qp = Reshape(a_qp_mem.ReadWrite(), test_vdim, test_op_dim,
|
||||
trial_vdim, total_trial_op_dim, num_qp,
|
||||
num_el);
|
||||
|
||||
Vector Ae_mem(num_test_dof * test_vdim * num_trial_dof * trial_vdim * num_el);
|
||||
Ae_mem = 0.0;
|
||||
|
||||
auto A_e = Reshape(Ae_mem.ReadWrite(), num_test_dof, test_vdim, num_trial_dof,
|
||||
trial_vdim, num_el);
|
||||
|
||||
for (int e = 0; e < num_el; e++)
|
||||
{
|
||||
map_fields_to_quadrature_data(
|
||||
input_qp, e, this->fields_e,
|
||||
kinput_to_field, input_dtq_ops,
|
||||
integration_weights, geometric_factors, kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
for (int q = 0; q < num_qp; q++)
|
||||
{
|
||||
for (int j = 0; j < trial_vdim; j++)
|
||||
{
|
||||
size_t m_offset = 0;
|
||||
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
|
||||
{
|
||||
auto Bu = dependent_input_dtq_ops[s];
|
||||
auto [unused1, trial_op_dim, unused2] = Bu.GetShape();
|
||||
auto d_qp = Reshape(&(directions_qp[Bu.which_input])[0], trial_vdim,
|
||||
trial_op_dim, num_qp);
|
||||
for (int m = 0; m < trial_op_dim; m++)
|
||||
{
|
||||
d_qp(j, m, q) = 1.0;
|
||||
Vector f_qp = apply_kernel_fwddiff_enzyme(
|
||||
kernel.func,
|
||||
kernel_args,
|
||||
input_qp,
|
||||
kernel_shadow_args,
|
||||
directions_qp,
|
||||
q);
|
||||
// Vector f_qp = apply_kernel_fwddiff_dual(
|
||||
// kernel.func,
|
||||
// kernel_args,
|
||||
// input_qp,
|
||||
// directions_qp,
|
||||
// q);
|
||||
d_qp(j, m, q) = 0.0;
|
||||
|
||||
auto f = Reshape(f_qp.Read(), test_vdim, test_op_dim);
|
||||
|
||||
for (int i = 0; i < test_vdim; i++)
|
||||
{
|
||||
for (int k = 0; k < test_op_dim; k++)
|
||||
{
|
||||
a_qp(i, k, j, m + m_offset, q, e) = f(i, k);
|
||||
}
|
||||
}
|
||||
}
|
||||
m_offset += trial_op_dim;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vector fhat_mem(test_op_dim * num_qp * dimension);
|
||||
auto fhat = Reshape(fhat_mem.ReadWrite(), test_vdim, test_op_dim, num_qp);
|
||||
for (int J = 0; J < num_trial_dof; J++)
|
||||
{
|
||||
for (int j = 0; j < trial_vdim; j++)
|
||||
{
|
||||
fhat_mem = 0.0;
|
||||
size_t m_offset = 0;
|
||||
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
|
||||
{
|
||||
auto Bu = dependent_input_dtq_ops[s];
|
||||
int trial_op_dim = dependent_input_dtq_ops[s].GetShape()[1];
|
||||
for (int q = 0; q < num_qp; q++)
|
||||
{
|
||||
for (int i = 0; i < test_vdim; i++)
|
||||
{
|
||||
for (int k = 0; k < test_op_dim; k++)
|
||||
{
|
||||
for (int m = 0; m < trial_op_dim; m++)
|
||||
{
|
||||
fhat(i, k, q) += a_qp(i, k, j, m + m_offset, q, e) * Bu(q, m, J);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
m_offset += trial_op_dim;
|
||||
}
|
||||
|
||||
auto bvtfhat = Reshape(&A_e(0, 0, J, j, e), num_test_dof, test_vdim);
|
||||
map_quadrature_data_to_fields(bvtfhat, fhat, output_fop,
|
||||
output_dtq_ops[hardcoded_output_idx]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool same_test_and_trial = false;
|
||||
if (koutput_to_field[0] ==
|
||||
kinput_to_field[dependent_input_dtq_ops[0].which_input])
|
||||
{
|
||||
same_test_and_trial = true;
|
||||
}
|
||||
|
||||
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&op.fields[kinput_to_field[dependent_input_dtq_ops[0].which_input]].data);
|
||||
|
||||
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&op.fields[koutput_to_field[0]].data);
|
||||
|
||||
SparseMatrix mat(test_fes->GlobalVSize(), trial_fes->GlobalVSize());
|
||||
|
||||
if (test_fes == nullptr)
|
||||
{
|
||||
MFEM_ABORT("error");
|
||||
}
|
||||
|
||||
for (int e = 0; e < num_el; e++)
|
||||
{
|
||||
auto tmp = Reshape(Ae_mem.ReadWrite(), num_test_dof * test_vdim,
|
||||
num_trial_dof * trial_vdim,
|
||||
num_el);
|
||||
DenseMatrix A_e(&tmp(0, 0, e), num_test_dof * test_vdim,
|
||||
num_trial_dof * trial_vdim);
|
||||
Array<int> test_vdofs, trial_vdofs;
|
||||
test_fes->GetElementVDofs(e, test_vdofs);
|
||||
GetElementVDofs(
|
||||
op.fields[kinput_to_field[dependent_input_dtq_ops[0].which_input]], e,
|
||||
trial_vdofs);
|
||||
mat.AddSubMatrix(test_vdofs, trial_vdofs, A_e, 1);
|
||||
}
|
||||
mat.Finalize();
|
||||
|
||||
if (same_test_and_trial)
|
||||
{
|
||||
HypreParMatrix tmp(test_fes->GetComm(),
|
||||
test_fes->GlobalVSize(),
|
||||
test_fes->GetDofOffsets(),
|
||||
&mat);
|
||||
|
||||
A = *RAP(&tmp, test_fes->Dof_TrueDof_Matrix());
|
||||
A.EliminateBC(op.ess_tdof_list, DiagonalPolicy::DIAG_ONE);
|
||||
}
|
||||
else
|
||||
{
|
||||
HypreParMatrix tmp(test_fes->GetComm(),
|
||||
test_fes->GlobalVSize(),
|
||||
trial_fes->GlobalVSize(),
|
||||
test_fes->GetDofOffsets(),
|
||||
trial_fes->GetDofOffsets(),
|
||||
&mat);
|
||||
|
||||
A = *RAP(test_fes->Dof_TrueDof_Matrix(), &tmp, trial_fes->Dof_TrueDof_Matrix());
|
||||
// A.EliminateBC(op.ess_tdof_list, DiagonalPolicy::DIAG_ONE);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,233 @@
|
||||
template <
|
||||
typename kernels_tuple,
|
||||
size_t num_solutions,
|
||||
size_t num_parameters,
|
||||
size_t num_fields,
|
||||
size_t num_kernels
|
||||
>
|
||||
template <
|
||||
size_t derivative_idx
|
||||
>
|
||||
template <
|
||||
typename kernel_t
|
||||
>
|
||||
void DifferentiableOperator<kernels_tuple,
|
||||
num_solutions,
|
||||
num_parameters,
|
||||
num_fields,
|
||||
num_kernels>::Derivative<derivative_idx>::assemble_vector_impl(
|
||||
kernel_t kernel, Vector &v)
|
||||
{
|
||||
using entity_t = typename kernel_t::entity_t;
|
||||
|
||||
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.outputs,
|
||||
std::make_index_sequence<kernel.num_koutputs> {});
|
||||
|
||||
auto output_fop = std::get<0>(kernel.outputs);
|
||||
|
||||
constexpr int hardcoded_output_idx = 0;
|
||||
|
||||
int num_qp = op.integration_rule.GetNPoints();;
|
||||
int num_el = 0;
|
||||
int dimension = 0;
|
||||
if constexpr (std::is_same_v<entity_t, Entity::Element>)
|
||||
{
|
||||
num_el = op.mesh.GetNE();
|
||||
dimension = op.dim;
|
||||
}
|
||||
else if (std::is_same_v<entity_t, Entity::Face>)
|
||||
{
|
||||
num_el = op.mesh.GetNumFacesWithGhost();
|
||||
dimension = op.dim - 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(always_false<entity_t>, "not implemented");
|
||||
}
|
||||
|
||||
std::vector<const DofToQuad*> dtqmaps;
|
||||
for (const auto &field : op.fields)
|
||||
{
|
||||
dtqmaps.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
|
||||
doftoquad_mode));
|
||||
}
|
||||
|
||||
// Allocate memory for fields on quadrature points
|
||||
auto input_qp_mem = create_input_qp_memory(num_qp, kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto directions_qp_mem = create_input_qp_memory(num_qp, kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
for (auto &d_qp_mem : directions_qp_mem)
|
||||
{
|
||||
d_qp_mem = 0.0;
|
||||
}
|
||||
|
||||
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
|
||||
bool no_kinput_is_dependent = true;
|
||||
for (int i = 0; i < kinput_is_dependent.size(); i++)
|
||||
{
|
||||
if (kinput_to_field[i] == derivative_idx)
|
||||
{
|
||||
no_kinput_is_dependent = false;
|
||||
kinput_is_dependent[i] = true;
|
||||
// out << "function input " << i << " is dependent on "
|
||||
// << op.fields[kinput_to_field[i]].field_label << "\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
kinput_is_dependent[i] = false;
|
||||
}
|
||||
}
|
||||
|
||||
if (no_kinput_is_dependent)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
|
||||
DeviceTensor<1, const double> integration_weights(
|
||||
this->op.integration_rule.GetWeights().Read(), num_qp);
|
||||
|
||||
Vector zero;
|
||||
GeometricFactorMaps geometric_factors
|
||||
{
|
||||
DeviceTensor<3, const double>(zero.Read(), 0, 0, 0)
|
||||
};
|
||||
|
||||
// fields interpolated to the quadrature points in the order of
|
||||
// kernel function arguments
|
||||
auto input_qp = map_inputs_to_memory(input_qp_mem, num_qp,
|
||||
kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto directions_qp = map_inputs_to_memory(directions_qp_mem, num_qp,
|
||||
kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto input_dtq_ops = create_dtq_operators<entity_t>(kernel.inputs, dtqmaps,
|
||||
kinput_to_field);
|
||||
auto dependent_input_dtq_ops = create_dtq_operators_conditional<entity_t>(
|
||||
kernel.inputs,
|
||||
dtqmaps,
|
||||
kinput_to_field,
|
||||
kinput_is_dependent, std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto output_dtq_ops = create_dtq_operators<entity_t>(kernel.outputs, dtqmaps,
|
||||
koutput_to_field);
|
||||
|
||||
constexpr int fixed_output_idx = 0;
|
||||
auto Bv = output_dtq_ops[fixed_output_idx];
|
||||
auto [num_test_qp, test_op_dim, num_test_dof] = Bv.GetShape();
|
||||
const int test_vdim = std::get<0>(kernel.outputs).vdim;
|
||||
|
||||
const int num_trial_dof = dependent_input_dtq_ops[0].GetShape()[2];
|
||||
int trial_vdim = 0;
|
||||
int dependent_field_idx = -1;
|
||||
for (int i = 0; i < kinput_is_dependent.size(); i++)
|
||||
{
|
||||
if (kinput_is_dependent[i])
|
||||
{
|
||||
dependent_field_idx = kinput_to_field[i];
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
trial_vdim = GetVDim(op.fields[dependent_field_idx]);
|
||||
|
||||
// All trial operators dimensions accumulated
|
||||
int total_trial_op_dim = 0;
|
||||
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
|
||||
{
|
||||
total_trial_op_dim += dependent_input_dtq_ops[s].GetShape()[1];
|
||||
}
|
||||
|
||||
Vector a_qp_mem(trial_vdim * total_trial_op_dim * num_qp * num_el);
|
||||
const auto a_qp = Reshape(a_qp_mem.ReadWrite(), trial_vdim,
|
||||
total_trial_op_dim, num_qp, num_el);
|
||||
Vector ve_mem(num_trial_dof * trial_vdim * num_el);
|
||||
ve_mem = 0.0;
|
||||
|
||||
for (int e = 0; e < num_el; e++)
|
||||
{
|
||||
map_fields_to_quadrature_data(
|
||||
input_qp, e, this->fields_e,
|
||||
kinput_to_field, input_dtq_ops,
|
||||
integration_weights, geometric_factors, kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
for (int q = 0; q < num_qp; q++)
|
||||
{
|
||||
for (int j = 0; j < trial_vdim; j++)
|
||||
{
|
||||
size_t m_offset = 0;
|
||||
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
|
||||
{
|
||||
auto Bu = dependent_input_dtq_ops[s];
|
||||
auto [unused1, trial_op_dim, unused2] = Bu.GetShape();
|
||||
auto d_qp = Reshape(&(directions_qp[Bu.which_input])[0], trial_vdim,
|
||||
trial_op_dim, num_qp);
|
||||
for (int m = 0; m < trial_op_dim; m++)
|
||||
{
|
||||
d_qp(j, m, q) = 1.0;
|
||||
// Vector f_qp = apply_kernel_fwddiff_dual(
|
||||
// kernel.func,
|
||||
// kernel_args,
|
||||
// input_qp,
|
||||
// directions_qp,
|
||||
// q);
|
||||
Vector f_qp = apply_kernel_fwddiff_enzyme(
|
||||
kernel.func,
|
||||
kernel_args,
|
||||
input_qp,
|
||||
kernel_shadow_args,
|
||||
directions_qp,
|
||||
q);
|
||||
d_qp(j, m, q) = 0.0;
|
||||
|
||||
auto f = Reshape(f_qp.Read(), test_vdim);
|
||||
a_qp(j, m + m_offset, q, e) = f(0);
|
||||
}
|
||||
m_offset += trial_op_dim;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
auto shat = Reshape(ve_mem.ReadWrite(), num_trial_dof, trial_vdim, num_el);
|
||||
for (int J = 0; J < num_trial_dof; J++)
|
||||
{
|
||||
for (int j = 0; j < trial_vdim; j++)
|
||||
{
|
||||
size_t m_offset = 0;
|
||||
for (int s = 0; s < dependent_input_dtq_ops.size(); s++)
|
||||
{
|
||||
auto Bu = dependent_input_dtq_ops[s];
|
||||
int trial_op_dim = dependent_input_dtq_ops[s].GetShape()[1];
|
||||
for (int q = 0; q < num_qp; q++)
|
||||
{
|
||||
for (int m = 0; m < trial_op_dim; m++)
|
||||
{
|
||||
shat(J, j, e) += a_qp(j, m + m_offset, q, e) * Bu(q, m, J);
|
||||
}
|
||||
}
|
||||
m_offset += trial_op_dim;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
auto R = get_element_restriction(op.fields[dependent_field_idx],
|
||||
element_dof_ordering);
|
||||
Vector ve(R->Width());
|
||||
R->MultTranspose(ve_mem, ve);
|
||||
|
||||
get_prolongation(op.fields[dependent_field_idx])->MultTranspose(ve, v);
|
||||
}
|
||||
@@ -0,0 +1,244 @@
|
||||
template <
|
||||
typename kernels_tuple,
|
||||
size_t num_solutions,
|
||||
size_t num_parameters,
|
||||
size_t num_fields,
|
||||
size_t num_kernels
|
||||
>
|
||||
template <
|
||||
size_t derivative_idx
|
||||
>
|
||||
template <
|
||||
typename kernel_t
|
||||
>
|
||||
void DifferentiableOperator<kernels_tuple,
|
||||
num_solutions,
|
||||
num_parameters,
|
||||
num_fields,
|
||||
num_kernels>::Derivative<derivative_idx>::create_callback(kernel_t kernel,
|
||||
mult_func_t &func)
|
||||
{
|
||||
using entity_t = typename kernel_t::entity_t;
|
||||
|
||||
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
|
||||
|
||||
constexpr int hardcoded_output_idx = 0;
|
||||
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
|
||||
|
||||
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
|
||||
element_dof_ordering);
|
||||
|
||||
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
|
||||
|
||||
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
|
||||
const int num_entities = GetNumEntities<entity_t>(op.mesh);
|
||||
const int num_qp = op.integration_rule.GetNPoints();
|
||||
|
||||
// assume only a single element type for now
|
||||
std::vector<const DofToQuad*> dtq;
|
||||
for (const auto &field : op.fields)
|
||||
{
|
||||
dtq.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
|
||||
doftoquad_mode));
|
||||
}
|
||||
const int q1d = dtq[0]->nqpt;
|
||||
|
||||
derivative_action_e.SetSize(R->Height());
|
||||
|
||||
const int da_size_on_qp = GetSizeOnQP<entity_t>(
|
||||
mfem::get<hardcoded_output_idx>(kernel.outputs),
|
||||
op.fields[test_space_field_idx]);
|
||||
|
||||
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
|
||||
kinput_to_field);
|
||||
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
|
||||
koutput_to_field);
|
||||
|
||||
auto input_fops = create_bare_fops(kernel.inputs);
|
||||
auto output_fops = create_bare_fops(kernel.outputs);
|
||||
|
||||
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
|
||||
const int test_op_dim =
|
||||
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
|
||||
mfem::get<hardcoded_output_idx>(output_fops).vdim;
|
||||
const int num_test_dof = R->Height() /
|
||||
mfem::get<hardcoded_output_idx>(output_fops).vdim /
|
||||
num_entities;
|
||||
|
||||
auto ir_weights = Reshape(this->op.integration_rule.GetWeights().Read(),
|
||||
num_qp);
|
||||
|
||||
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
// Check which qf inputs are dependent on the dependent variable
|
||||
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
|
||||
bool no_kinput_is_dependent = true;
|
||||
for (int i = 0; i < kinput_is_dependent.size(); i++)
|
||||
{
|
||||
if (kinput_to_field[i] == derivative_idx)
|
||||
{
|
||||
no_kinput_is_dependent = false;
|
||||
kinput_is_dependent[i] = true;
|
||||
// out << "function input " << i << " is dependent on "
|
||||
// << op.fields[kinput_to_field[i]].field_label << "\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
kinput_is_dependent[i] = false;
|
||||
}
|
||||
}
|
||||
|
||||
bool with_derivatives = true;
|
||||
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
|
||||
output_dtq_maps,
|
||||
op.fields,
|
||||
num_entities,
|
||||
kernel.inputs,
|
||||
num_qp,
|
||||
input_size_on_qp,
|
||||
da_size_on_qp,
|
||||
derivative_idx);
|
||||
|
||||
Vector shmem_cache(shmem_info.total_size);
|
||||
|
||||
print_shared_memory_info(shmem_info);
|
||||
|
||||
func = [=](Vector &ye_mem) mutable
|
||||
{
|
||||
if (no_kinput_is_dependent)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
restriction<entity_t>(direction, direction_l, direction_e,
|
||||
op.element_dof_ordering);
|
||||
|
||||
auto ye = Reshape(ye_mem.ReadWrite(), num_test_dof, test_vdim, num_entities);
|
||||
auto wrapped_fields_e = wrap_fields(this->fields_e, shmem_info.field_sizes, num_entities);
|
||||
auto wrapped_direction_e = Reshape(direction_e.Read(), shmem_info.direction_size, num_entities);
|
||||
|
||||
forall([=] MFEM_HOST_DEVICE (int e, double *shmem)
|
||||
{
|
||||
auto input_dtq_shmem = load_dtq_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
|
||||
shmem_info.input_dtq_sizes,
|
||||
input_dtq_maps);
|
||||
|
||||
auto output_dtq_shmem = load_dtq_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
|
||||
shmem_info.output_dtq_sizes,
|
||||
output_dtq_maps);
|
||||
|
||||
auto fields_shmem = load_field_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::FIELD],
|
||||
shmem_info.field_sizes,
|
||||
kinput_to_field,
|
||||
wrapped_fields_e,
|
||||
e);
|
||||
|
||||
auto direction_shmem = load_direction_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::DIRECTION],
|
||||
shmem_info.direction_size,
|
||||
wrapped_direction_e,
|
||||
e);
|
||||
|
||||
// These methods don't copy, they simply create a `DeviceTensor` object
|
||||
// that points to correct chunks of the shared memory pool.
|
||||
auto input_shmem = load_input_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::INPUT],
|
||||
shmem_info.input_sizes,
|
||||
num_qp);
|
||||
|
||||
auto shadow_shmem = load_input_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::SHADOW],
|
||||
shmem_info.input_sizes,
|
||||
num_qp);
|
||||
|
||||
auto residual_shmem = load_residual_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::OUTPUT],
|
||||
shmem_info.residual_size,
|
||||
num_qp);
|
||||
|
||||
auto scratch_mem = load_scratch_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::TEMP],
|
||||
shmem_info.temp_sizes);
|
||||
|
||||
map_fields_to_quadrature_data<TensorProduct>(
|
||||
input_shmem, fields_shmem, input_dtq_shmem, input_fops, ir_weights, scratch_mem,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
zero_all(shadow_shmem);
|
||||
map_direction_to_quadrature_data_conditional<TensorProduct>(
|
||||
shadow_shmem, direction_shmem, input_dtq_shmem, input_fops, ir_weights,
|
||||
scratch_mem, kinput_is_dependent,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
const int q = qx + q1d * (qy + q1d * qz);
|
||||
|
||||
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
|
||||
auto r = Reshape(&residual_shmem(0, q), da_size_on_qp);
|
||||
apply_kernel_fwddiff_enzyme(
|
||||
r,
|
||||
kernel.func,
|
||||
kernel_args,
|
||||
input_shmem,
|
||||
kernel_shadow_args,
|
||||
shadow_shmem,
|
||||
q);
|
||||
// printf(">>>>> WARNING: AD DISABLED\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
|
||||
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
|
||||
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
|
||||
mfem::get<0>(output_fops),
|
||||
output_dtq_shmem[hardcoded_output_idx],
|
||||
scratch_mem);
|
||||
}, num_entities, q1d, q1d, 1, shmem_info.total_size, shmem_cache.ReadWrite());
|
||||
|
||||
R->MultTranspose(ye_mem, derivative_action_l);
|
||||
};
|
||||
|
||||
if constexpr (std::is_same_v<decltype(output_fop), One>)
|
||||
{
|
||||
prolongation_transpose = [&](Vector &r_local, Vector &y)
|
||||
{
|
||||
double local_sum = r_local.Sum();
|
||||
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
|
||||
op.mesh.GetComm());
|
||||
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
|
||||
};
|
||||
}
|
||||
else
|
||||
{
|
||||
auto P = get_prolongation(op.fields[test_space_field_idx]);
|
||||
prolongation_transpose = [P](const Vector &r_local, Vector &y)
|
||||
{
|
||||
P->MultTranspose(r_local, y);
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,806 @@
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdlib>
|
||||
#include <functional>
|
||||
#include <iostream>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
#include <vector>
|
||||
#include <type_traits>
|
||||
#include <mfem.hpp>
|
||||
#include <type_traits>
|
||||
#include "dfem_fieldoperator.hpp"
|
||||
#include "dfem_parametricspace.hpp"
|
||||
#include "general/tic_toc.hpp"
|
||||
#include "tuple.hpp"
|
||||
#include <linalg/tensor.hpp>
|
||||
#include <enzyme/utils>
|
||||
#include <enzyme/enzyme>
|
||||
#include "dfem_util.hpp"
|
||||
#include "dfem_interpolate.hpp"
|
||||
#include "dfem_qfunction.hpp"
|
||||
#include "dfem_integrate.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
using mult_func_t = std::function<void(Vector &)>;
|
||||
|
||||
template <
|
||||
typename kernels_tuple,
|
||||
size_t num_solutions,
|
||||
size_t num_parameters,
|
||||
size_t num_fields = num_solutions + num_parameters,
|
||||
size_t num_kernels = mfem::tuple_size<kernels_tuple>::value
|
||||
>
|
||||
class DifferentiableOperator : public Operator
|
||||
{
|
||||
public:
|
||||
DifferentiableOperator(DifferentiableOperator&) = delete;
|
||||
DifferentiableOperator(DifferentiableOperator&&) = delete;
|
||||
|
||||
class Action : public Operator
|
||||
{
|
||||
public:
|
||||
template <typename kernel_t>
|
||||
void create_action_callback(kernel_t kernel, mult_func_t &func);
|
||||
|
||||
template<std::size_t... idx>
|
||||
void materialize_callbacks(kernels_tuple &ks,
|
||||
std::array<mult_func_t, num_kernels>,
|
||||
std::index_sequence<idx...> const&)
|
||||
{
|
||||
(create_action_callback(mfem::get<idx>(ks), funcs[idx]), ...);
|
||||
}
|
||||
|
||||
Action(DifferentiableOperator &op, kernels_tuple &ks) : op(op)
|
||||
{
|
||||
materialize_callbacks(ks, funcs,
|
||||
std::make_index_sequence<mfem::tuple_size<kernels_tuple>::value>());
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &y) const
|
||||
{
|
||||
prolongation(op.solutions, x, solutions_l);
|
||||
|
||||
residual_e = 0.0;
|
||||
for (const auto &f : funcs)
|
||||
{
|
||||
f(residual_e);
|
||||
}
|
||||
|
||||
prolongation_transpose(residual_l, y);
|
||||
|
||||
y.SetSubVector(op.ess_tdof_list, 0.0);
|
||||
}
|
||||
|
||||
void SetParameters(std::vector<Vector *> p) const
|
||||
{
|
||||
MFEM_ASSERT(num_parameters == p.size(),
|
||||
"number of parameters doesn't match descriptors");
|
||||
for (int i = 0; i < num_parameters; i++)
|
||||
{
|
||||
p[i]->Read();
|
||||
parameters_l[i] = *p[i];
|
||||
// parameters_l[i].MakeRef(p[i], 0, p[i]->Size());
|
||||
}
|
||||
}
|
||||
|
||||
protected:
|
||||
DifferentiableOperator &op;
|
||||
std::array<mult_func_t, num_kernels> funcs;
|
||||
|
||||
std::function<void(Vector &, Vector &)> prolongation_transpose;
|
||||
|
||||
mutable std::array<Vector, num_solutions> solutions_l;
|
||||
mutable std::array<Vector, num_parameters> parameters_l;
|
||||
mutable Vector residual_l;
|
||||
|
||||
mutable std::array<Vector, num_fields> fields_e;
|
||||
mutable Vector residual_e;
|
||||
};
|
||||
|
||||
template <size_t derivative_idx>
|
||||
class Derivative : public Operator
|
||||
{
|
||||
public:
|
||||
template <typename kernel_t>
|
||||
void create_callback(kernel_t kernel, mult_func_t &func);
|
||||
|
||||
template<std::size_t... idx>
|
||||
void materialize_callbacks(kernels_tuple &ks,
|
||||
std::array<mult_func_t, num_kernels>,
|
||||
std::index_sequence<idx...> const&)
|
||||
{
|
||||
(create_callback(mfem::get<idx>(ks), funcs[idx]), ...);
|
||||
}
|
||||
|
||||
Derivative(
|
||||
DifferentiableOperator &op,
|
||||
std::array<Vector *, num_solutions> &solutions,
|
||||
std::array<Vector *, num_parameters> ¶meters,
|
||||
kernels_tuple &ks) : op(op), ks(ks)
|
||||
{
|
||||
for (int i = 0; i < num_solutions; i++)
|
||||
{
|
||||
solutions_l[i] = *solutions[i];
|
||||
}
|
||||
|
||||
for (int i = 0; i < num_parameters; i++)
|
||||
{
|
||||
parameters_l[i] = *parameters[i];
|
||||
}
|
||||
|
||||
// G
|
||||
// if constexpr (std::is_same_v<OperatesOn, OperatesOnElement>)
|
||||
// {
|
||||
element_restriction(op.solutions, solutions_l, fields_e,
|
||||
op.element_dof_ordering);
|
||||
element_restriction(op.parameters, parameters_l, fields_e,
|
||||
op.element_dof_ordering,
|
||||
op.solutions.size());
|
||||
// }
|
||||
// else
|
||||
// {
|
||||
// MFEM_ABORT("restriction not implemented for OperatesOn");
|
||||
// }
|
||||
direction = op.fields[derivative_idx];
|
||||
|
||||
size_t derivative_action_l_size = 0;
|
||||
for (auto &s : op.solutions)
|
||||
{
|
||||
derivative_action_l_size += GetVSize(s);
|
||||
this->width += GetTrueVSize(s);
|
||||
}
|
||||
this->height = derivative_action_l_size;
|
||||
derivative_action_l.SetSize(derivative_action_l_size);
|
||||
|
||||
materialize_callbacks(ks, funcs,
|
||||
std::make_index_sequence<num_kernels>());
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &y) const override
|
||||
{
|
||||
current_direction_t = x;
|
||||
current_direction_t.SetSubVector(op.ess_tdof_list, 0.0);
|
||||
|
||||
prolongation(direction, current_direction_t, direction_l);
|
||||
|
||||
derivative_action_e = 0.0;
|
||||
for (const auto &f : funcs)
|
||||
{
|
||||
f(derivative_action_e);
|
||||
}
|
||||
|
||||
prolongation_transpose(derivative_action_l, y);
|
||||
|
||||
y.SetSubVector(op.ess_tdof_list, 0.0);
|
||||
}
|
||||
|
||||
template <typename kernel_t>
|
||||
void assemble_vector_impl(kernel_t kernel, Vector &v);
|
||||
|
||||
template<std::size_t... idx>
|
||||
void assemble_vector(
|
||||
kernels_tuple &ks,
|
||||
Vector &v,
|
||||
std::index_sequence<idx...> const&)
|
||||
{
|
||||
(assemble_vector_impl(mfem::get<idx>(ks), v), ...);
|
||||
}
|
||||
|
||||
void Assemble(Vector &v)
|
||||
{
|
||||
assemble_vector(ks, v, std::make_index_sequence<num_kernels>());
|
||||
}
|
||||
|
||||
template <typename kernel_t>
|
||||
void assemble_hypreparmatrix_impl(kernel_t kernel, HypreParMatrix &A);
|
||||
|
||||
template<std::size_t... idx>
|
||||
void assemble_hypreparmatrix(
|
||||
kernels_tuple &ks,
|
||||
HypreParMatrix &A,
|
||||
std::index_sequence<idx...> const&)
|
||||
{
|
||||
(assemble_hypreparmatrix_impl(mfem::get<idx>(ks), A), ...);
|
||||
}
|
||||
|
||||
void Assemble(HypreParMatrix &A)
|
||||
{
|
||||
assemble_hypreparmatrix(ks, A, std::make_index_sequence<num_kernels>());
|
||||
}
|
||||
|
||||
void AssembleDiagonal(Vector &d) const override {}
|
||||
|
||||
protected:
|
||||
DifferentiableOperator &op;
|
||||
kernels_tuple &ks;
|
||||
std::array<mult_func_t, num_kernels> funcs;
|
||||
|
||||
std::function<void(Vector &, Vector &)> prolongation_transpose;
|
||||
|
||||
FieldDescriptor direction;
|
||||
|
||||
std::array<Vector, num_solutions> solutions_l;
|
||||
std::array<Vector, num_parameters> parameters_l;
|
||||
mutable Vector direction_l;
|
||||
mutable Vector derivative_action_l;
|
||||
|
||||
mutable std::array<Vector, num_fields> fields_e;
|
||||
mutable Vector direction_e;
|
||||
mutable Vector derivative_action_e;
|
||||
|
||||
mutable Vector current_direction_t;
|
||||
};
|
||||
|
||||
DifferentiableOperator(std::array<FieldDescriptor, num_solutions> s,
|
||||
std::array<FieldDescriptor, num_parameters> p,
|
||||
kernels_tuple ks,
|
||||
ParMesh &m,
|
||||
const IntegrationRule &integration_rule) :
|
||||
kernels(ks),
|
||||
mesh(m),
|
||||
dim(mesh.Dimension()),
|
||||
integration_rule(integration_rule),
|
||||
solutions(s),
|
||||
parameters(p)
|
||||
{
|
||||
for (int i = 0; i < num_solutions; i++)
|
||||
{
|
||||
fields[i] = solutions[i];
|
||||
}
|
||||
|
||||
for (int i = 0; i < num_parameters; i++)
|
||||
{
|
||||
fields[i + num_solutions] = parameters[i];
|
||||
}
|
||||
|
||||
residual.reset(new Action(*this, kernels));
|
||||
}
|
||||
|
||||
void SetParameters(std::vector<Vector *> p) const
|
||||
{
|
||||
residual->SetParameters(p);
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &y) const override
|
||||
{
|
||||
residual->Mult(x, y);
|
||||
}
|
||||
|
||||
template <int derivative_idx>
|
||||
std::shared_ptr<Derivative<derivative_idx>>
|
||||
GetDerivativeWrt(std::array<Vector *, num_solutions> solutions,
|
||||
std::array<Vector *, num_parameters> parameters)
|
||||
{
|
||||
return std::shared_ptr<Derivative<derivative_idx>>(
|
||||
new Derivative<derivative_idx>(*this, solutions, parameters, kernels));
|
||||
}
|
||||
|
||||
void SetEssentialTrueDofs(const Array<int> &l)
|
||||
{
|
||||
l.Copy(ess_tdof_list);
|
||||
}
|
||||
|
||||
kernels_tuple kernels;
|
||||
ParMesh &mesh;
|
||||
const int dim;
|
||||
const IntegrationRule &integration_rule;
|
||||
|
||||
std::array<FieldDescriptor, num_solutions> solutions;
|
||||
std::array<FieldDescriptor, num_parameters> parameters;
|
||||
// solutions and parameters
|
||||
std::array<FieldDescriptor, num_fields> fields;
|
||||
|
||||
int residual_lsize = 0;
|
||||
|
||||
mutable std::array<Vector, num_solutions> current_state_l;
|
||||
mutable Vector direction_l;
|
||||
|
||||
mutable Vector current_direction_t;
|
||||
|
||||
Array<int> ess_tdof_list;
|
||||
|
||||
static constexpr ElementDofOrdering element_dof_ordering =
|
||||
ElementDofOrdering::LEXICOGRAPHIC;
|
||||
|
||||
static constexpr DofToQuad::Mode doftoquad_mode =
|
||||
DofToQuad::Mode::TENSOR;
|
||||
|
||||
// static constexpr ElementDofOrdering element_dof_ordering =
|
||||
// ElementDofOrdering::NATIVE;
|
||||
|
||||
// static constexpr DofToQuad::Mode doftoquad_mode =
|
||||
// DofToQuad::Mode::FULL;
|
||||
|
||||
std::shared_ptr<Action> residual;
|
||||
};
|
||||
|
||||
template <
|
||||
typename kernels_tuple,
|
||||
size_t num_solutions,
|
||||
size_t num_parameters,
|
||||
size_t num_fields,
|
||||
size_t num_kernels
|
||||
>
|
||||
template <
|
||||
typename kernel_t
|
||||
>
|
||||
void DifferentiableOperator<kernels_tuple,
|
||||
num_solutions,
|
||||
num_parameters,
|
||||
num_fields,
|
||||
num_kernels>::Action::create_action_callback(
|
||||
kernel_t kernel,
|
||||
mult_func_t &func)
|
||||
{
|
||||
using entity_t = typename kernel_t::entity_t;
|
||||
|
||||
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
|
||||
|
||||
constexpr int hardcoded_output_idx = 0;
|
||||
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
|
||||
|
||||
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
|
||||
element_dof_ordering);
|
||||
|
||||
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
|
||||
|
||||
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
|
||||
const int num_entities = GetNumEntities<entity_t>(op.mesh);
|
||||
const int num_qp = op.integration_rule.GetNPoints();
|
||||
|
||||
// All solutions T-vector sizes make up the width of the operator, since
|
||||
// they are explicitly provided in Mult() for example.
|
||||
|
||||
op.width = GetTrueVSize(op.fields[test_space_field_idx]);
|
||||
op.residual_lsize = GetVSize(op.fields[test_space_field_idx]);
|
||||
|
||||
if constexpr (std::is_same_v<decltype(output_fop), One>)
|
||||
{
|
||||
op.height = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
op.height = op.residual_lsize;
|
||||
}
|
||||
|
||||
residual_l.SetSize(op.residual_lsize);
|
||||
|
||||
// assume only a single element type for now
|
||||
std::vector<const DofToQuad*> dtq;
|
||||
for (const auto &field : op.fields)
|
||||
{
|
||||
dtq.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
|
||||
doftoquad_mode));
|
||||
}
|
||||
const int q1d = (int)floor(pow(num_qp, 1.0/op.mesh.Dimension()) + 0.5);
|
||||
|
||||
residual_e.SetSize(R->Height());
|
||||
|
||||
const int residual_size_on_qp = GetSizeOnQP<entity_t>(
|
||||
mfem::get<hardcoded_output_idx>(kernel.outputs),
|
||||
op.fields[test_space_field_idx]);
|
||||
|
||||
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
|
||||
kinput_to_field);
|
||||
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
|
||||
koutput_to_field);
|
||||
|
||||
auto input_fops = create_bare_fops(kernel.inputs);
|
||||
auto output_fops = create_bare_fops(kernel.outputs);
|
||||
|
||||
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
|
||||
const int test_op_dim =
|
||||
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
|
||||
mfem::get<hardcoded_output_idx>(output_fops).vdim;
|
||||
const int num_test_dof = R->Height() /
|
||||
mfem::get<hardcoded_output_idx>(output_fops).vdim /
|
||||
num_entities;
|
||||
|
||||
auto ir_weights = Reshape(this->op.integration_rule.GetWeights().Read(),
|
||||
num_qp);
|
||||
|
||||
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
|
||||
output_dtq_maps,
|
||||
op.fields,
|
||||
num_entities,
|
||||
kernel.inputs,
|
||||
num_qp,
|
||||
input_size_on_qp,
|
||||
residual_size_on_qp);
|
||||
|
||||
Vector shmem_cache(shmem_info.total_size);
|
||||
|
||||
// print_shared_memory_info(shmem_info);
|
||||
|
||||
func = [=](Vector &ye_mem) mutable
|
||||
{
|
||||
restriction<entity_t>(op.solutions, solutions_l, this->fields_e,
|
||||
op.element_dof_ordering);
|
||||
restriction<entity_t>(op.parameters, parameters_l, this->fields_e,
|
||||
op.element_dof_ordering,
|
||||
op.solutions.size());
|
||||
|
||||
auto ye = Reshape(ye_mem.ReadWrite(), test_vdim, num_test_dof, num_entities);
|
||||
auto wrapped_fields_e = wrap_fields(this->fields_e, shmem_info.field_sizes, num_entities);
|
||||
|
||||
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
|
||||
{
|
||||
// printf("\ne: %d\n", e);
|
||||
// tic();
|
||||
auto input_dtq_shmem = load_dtq_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
|
||||
shmem_info.input_dtq_sizes,
|
||||
input_dtq_maps);
|
||||
|
||||
auto output_dtq_shmem = load_dtq_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
|
||||
shmem_info.output_dtq_sizes,
|
||||
output_dtq_maps);
|
||||
|
||||
auto fields_shmem = load_field_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::FIELD],
|
||||
shmem_info.field_sizes,
|
||||
kinput_to_field,
|
||||
input_fops,
|
||||
wrapped_fields_e,
|
||||
e,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
// These functions don't copy, they simply create a `DeviceTensor` object
|
||||
// that points to correct chunks of the shared memory pool.
|
||||
auto input_shmem = load_input_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::INPUT],
|
||||
shmem_info.input_sizes,
|
||||
num_qp);
|
||||
|
||||
auto residual_shmem = load_residual_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::OUTPUT],
|
||||
shmem_info.residual_size,
|
||||
num_qp);
|
||||
|
||||
auto scratch_mem = load_scratch_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::TEMP],
|
||||
shmem_info.temp_sizes);
|
||||
|
||||
MFEM_SYNC_THREAD;
|
||||
// printf("shmem load elapsed: %.1fus\n", toc() * 1e6);
|
||||
|
||||
// tic();
|
||||
map_fields_to_quadrature_data<TensorProduct>(
|
||||
input_shmem, fields_shmem, input_dtq_shmem, input_fops, ir_weights, scratch_mem,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
// printf("interpolate elapsed: %.1fus\n", toc() * 1e6);
|
||||
|
||||
// tic();
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
const int q = qx + q1d * (qy + q1d * qz);
|
||||
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
auto r = Reshape(&residual_shmem(0, q), residual_size_on_qp);
|
||||
apply_kernel(r, kernel.func, kernel_args, input_shmem, q);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
// printf("qf elapsed: %.1fus\n", toc() * 1e6);
|
||||
|
||||
// tic();
|
||||
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
|
||||
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
|
||||
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
|
||||
mfem::get<0>(output_fops),
|
||||
output_dtq_shmem[hardcoded_output_idx],
|
||||
scratch_mem);
|
||||
// printf("integrate elapsed: %.1fus\n", toc() * 1e6);
|
||||
|
||||
}, num_entities, q1d, q1d, q1d, shmem_info.total_size, shmem_cache.ReadWrite());
|
||||
|
||||
if constexpr (std::is_same_v<decltype(output_fop), None>)
|
||||
{
|
||||
residual_l = ye_mem;
|
||||
}
|
||||
else
|
||||
{
|
||||
R->MultTranspose(ye_mem, residual_l);
|
||||
}
|
||||
};
|
||||
|
||||
if constexpr (std::is_same_v<decltype(output_fop), None>)
|
||||
{
|
||||
prolongation_transpose = [&](Vector &r_local, Vector &y)
|
||||
{
|
||||
y = r_local;
|
||||
};
|
||||
}
|
||||
else if constexpr (std::is_same_v<decltype(output_fop), One>)
|
||||
{
|
||||
prolongation_transpose = [&](Vector &r_local, Vector &y)
|
||||
{
|
||||
double local_sum = r_local.Sum();
|
||||
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
|
||||
op.mesh.GetComm());
|
||||
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
|
||||
};
|
||||
}
|
||||
else
|
||||
{
|
||||
auto P = get_prolongation(op.fields[test_space_field_idx]);
|
||||
prolongation_transpose = [P](const Vector &r_local, Vector &y)
|
||||
{
|
||||
P->MultTranspose(r_local, y);
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
template <
|
||||
typename kernels_tuple,
|
||||
size_t num_solutions,
|
||||
size_t num_parameters,
|
||||
size_t num_fields,
|
||||
size_t num_kernels
|
||||
>
|
||||
template <
|
||||
size_t derivative_idx
|
||||
>
|
||||
template <
|
||||
typename kernel_t
|
||||
>
|
||||
void DifferentiableOperator<kernels_tuple,
|
||||
num_solutions,
|
||||
num_parameters,
|
||||
num_fields,
|
||||
num_kernels>::Derivative<derivative_idx>::create_callback(kernel_t kernel,
|
||||
mult_func_t &func)
|
||||
{
|
||||
using entity_t = typename kernel_t::entity_t;
|
||||
|
||||
auto kinput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.inputs, std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto koutput_to_field = create_descriptors_to_fields_map<entity_t>(op.fields,
|
||||
kernel.outputs, std::make_index_sequence<kernel.num_koutputs> {});
|
||||
|
||||
constexpr int hardcoded_output_idx = 0;
|
||||
const int test_space_field_idx = koutput_to_field[hardcoded_output_idx];
|
||||
|
||||
const Operator *R = get_restriction<entity_t>(op.fields[test_space_field_idx],
|
||||
element_dof_ordering);
|
||||
|
||||
auto output_fop = mfem::get<hardcoded_output_idx>(kernel.outputs);
|
||||
|
||||
const int num_elements = GetNumEntities<Entity::Element>(op.mesh);
|
||||
const int num_entities = GetNumEntities<entity_t>(op.mesh);
|
||||
const int num_qp = op.integration_rule.GetNPoints();
|
||||
|
||||
// assume only a single element type for now
|
||||
std::vector<const DofToQuad*> dtq;
|
||||
for (const auto &field : op.fields)
|
||||
{
|
||||
dtq.emplace_back(GetDofToQuad<entity_t>(field, op.integration_rule,
|
||||
doftoquad_mode));
|
||||
}
|
||||
const int q1d = dtq[0]->nqpt;
|
||||
|
||||
derivative_action_e.SetSize(R->Height());
|
||||
|
||||
const int da_size_on_qp = GetSizeOnQP<entity_t>(
|
||||
mfem::get<hardcoded_output_idx>(kernel.outputs),
|
||||
op.fields[test_space_field_idx]);
|
||||
|
||||
auto input_dtq_maps = create_dtq_maps<entity_t>(kernel.inputs, dtq,
|
||||
kinput_to_field);
|
||||
auto output_dtq_maps = create_dtq_maps<entity_t>(kernel.outputs, dtq,
|
||||
koutput_to_field);
|
||||
|
||||
auto input_fops = create_bare_fops(kernel.inputs);
|
||||
auto output_fops = create_bare_fops(kernel.outputs);
|
||||
|
||||
const int test_vdim = mfem::get<hardcoded_output_idx>(output_fops).vdim;
|
||||
const int test_op_dim =
|
||||
mfem::get<hardcoded_output_idx>(output_fops).size_on_qp /
|
||||
mfem::get<hardcoded_output_idx>(output_fops).vdim;
|
||||
const int num_test_dof = R->Height() /
|
||||
mfem::get<hardcoded_output_idx>(output_fops).vdim /
|
||||
num_entities;
|
||||
|
||||
auto ir_weights = Reshape(this->op.integration_rule.GetWeights().Read(),
|
||||
num_qp);
|
||||
|
||||
auto input_size_on_qp = get_input_size_on_qp(kernel.inputs,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
// Check which qf inputs are dependent on the dependent variable
|
||||
std::array<bool, kernel.num_kinputs> kinput_is_dependent;
|
||||
bool no_kinput_is_dependent = true;
|
||||
for (int i = 0; i < kinput_is_dependent.size(); i++)
|
||||
{
|
||||
if (kinput_to_field[i] == derivative_idx)
|
||||
{
|
||||
no_kinput_is_dependent = false;
|
||||
kinput_is_dependent[i] = true;
|
||||
// out << "function input " << i << " is dependent on "
|
||||
// << op.fields[kinput_to_field[i]].field_label << "\n";
|
||||
}
|
||||
else
|
||||
{
|
||||
kinput_is_dependent[i] = false;
|
||||
}
|
||||
}
|
||||
|
||||
bool with_derivatives = true;
|
||||
auto shmem_info = get_shmem_info<entity_t>(input_dtq_maps,
|
||||
output_dtq_maps,
|
||||
op.fields,
|
||||
num_entities,
|
||||
kernel.inputs,
|
||||
num_qp,
|
||||
input_size_on_qp,
|
||||
da_size_on_qp,
|
||||
derivative_idx);
|
||||
|
||||
Vector shmem_cache(shmem_info.total_size);
|
||||
|
||||
// print_shared_memory_info(shmem_info);
|
||||
|
||||
func = [=](Vector &ye_mem) mutable
|
||||
{
|
||||
if (no_kinput_is_dependent)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
restriction<entity_t>(direction, direction_l, direction_e,
|
||||
op.element_dof_ordering);
|
||||
|
||||
auto ye = Reshape(ye_mem.ReadWrite(), num_test_dof, test_vdim, num_entities);
|
||||
auto wrapped_fields_e = wrap_fields(this->fields_e, shmem_info.field_sizes, num_entities);
|
||||
auto wrapped_direction_e = Reshape(direction_e.ReadWrite(), shmem_info.direction_size, num_entities);
|
||||
|
||||
forall([=] MFEM_HOST_DEVICE (int e, double *shmem)
|
||||
{
|
||||
auto input_dtq_shmem = load_dtq_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::INPUT_DTQ],
|
||||
shmem_info.input_dtq_sizes,
|
||||
input_dtq_maps);
|
||||
|
||||
auto output_dtq_shmem = load_dtq_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::OUTPUT_DTQ],
|
||||
shmem_info.output_dtq_sizes,
|
||||
output_dtq_maps);
|
||||
|
||||
auto fields_shmem = load_field_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::FIELD],
|
||||
shmem_info.field_sizes,
|
||||
kinput_to_field,
|
||||
input_fops,
|
||||
wrapped_fields_e,
|
||||
e,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
auto direction_shmem = load_direction_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::DIRECTION],
|
||||
shmem_info.direction_size,
|
||||
wrapped_direction_e,
|
||||
e);
|
||||
|
||||
// These methods don't copy, they simply create a `DeviceTensor` object
|
||||
// that points to correct chunks of the shared memory pool.
|
||||
auto input_shmem = load_input_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::INPUT],
|
||||
shmem_info.input_sizes,
|
||||
num_qp);
|
||||
|
||||
auto shadow_shmem = load_input_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::SHADOW],
|
||||
shmem_info.input_sizes,
|
||||
num_qp);
|
||||
|
||||
auto residual_shmem = load_residual_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::OUTPUT],
|
||||
shmem_info.residual_size,
|
||||
num_qp);
|
||||
|
||||
auto scratch_mem = load_scratch_mem(
|
||||
shmem,
|
||||
shmem_info.offsets[SharedMemory::Index::TEMP],
|
||||
shmem_info.temp_sizes);
|
||||
|
||||
map_fields_to_quadrature_data<TensorProduct>(
|
||||
input_shmem, fields_shmem, input_dtq_shmem, input_fops, ir_weights, scratch_mem,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
zero_all(shadow_shmem);
|
||||
map_direction_to_quadrature_data_conditional<TensorProduct>(
|
||||
shadow_shmem, direction_shmem, input_dtq_shmem, input_fops, ir_weights,
|
||||
scratch_mem, kinput_is_dependent,
|
||||
std::make_index_sequence<kernel.num_kinputs> {});
|
||||
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
const int q = qx + q1d * (qy + q1d * qz);
|
||||
|
||||
auto kernel_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
auto kernel_shadow_args = decay_tuple<typename kernel_t::kf_param_ts> {};
|
||||
|
||||
auto r = Reshape(&residual_shmem(0, q), da_size_on_qp);
|
||||
apply_kernel_fwddiff_enzyme(
|
||||
r,
|
||||
kernel.func,
|
||||
kernel_args,
|
||||
input_shmem,
|
||||
kernel_shadow_args,
|
||||
shadow_shmem,
|
||||
q);
|
||||
// printf(">>>>> WARNING: AD DISABLED\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
auto fhat = Reshape(&residual_shmem(0, 0), test_vdim, test_op_dim, num_qp);
|
||||
auto y = Reshape(&ye(0, 0, e), num_test_dof, test_vdim);
|
||||
map_quadrature_data_to_fields<TensorProduct>(y, fhat,
|
||||
mfem::get<0>(output_fops),
|
||||
output_dtq_shmem[hardcoded_output_idx],
|
||||
scratch_mem);
|
||||
}, num_entities, q1d, q1d, 1, shmem_info.total_size, shmem_cache.ReadWrite());
|
||||
|
||||
R->MultTranspose(ye_mem, derivative_action_l);
|
||||
};
|
||||
|
||||
if constexpr (std::is_same_v<decltype(output_fop), One>)
|
||||
{
|
||||
prolongation_transpose = [&](Vector &r_local, Vector &y)
|
||||
{
|
||||
double local_sum = r_local.Sum();
|
||||
MPI_Allreduce(&local_sum, y.GetData(), 1, MPI_DOUBLE, MPI_SUM,
|
||||
op.mesh.GetComm());
|
||||
MFEM_ASSERT(y.Size() == 1, "output size doesn't match kernel description");
|
||||
};
|
||||
}
|
||||
else
|
||||
{
|
||||
auto P = get_prolongation(op.fields[test_space_field_idx]);
|
||||
prolongation_transpose = [P](const Vector &r_local, Vector &y)
|
||||
{
|
||||
P->MultTranspose(r_local, y);
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// #include "dfem_assemble_vector.icc"
|
||||
// #include "dfem_assemble_hypreparmatrix.icc"
|
||||
|
||||
}
|
||||
@@ -0,0 +1,118 @@
|
||||
#pragma once
|
||||
|
||||
#include <string>
|
||||
|
||||
class FieldOperator
|
||||
{
|
||||
public:
|
||||
FieldOperator(std::string field_label = "", int size_on_qp = 0) :
|
||||
field_label(field_label),
|
||||
size_on_qp(size_on_qp) {};
|
||||
|
||||
std::string field_label;
|
||||
|
||||
int size_on_qp = -1;
|
||||
|
||||
int dim = -1;
|
||||
|
||||
int vdim = -1;
|
||||
};
|
||||
|
||||
class None : public FieldOperator
|
||||
{
|
||||
public:
|
||||
None(std::string field_label) :
|
||||
FieldOperator(field_label) {}
|
||||
};
|
||||
|
||||
class Weight : public FieldOperator
|
||||
{
|
||||
public:
|
||||
Weight() : FieldOperator("quadrature_weights") {};
|
||||
};
|
||||
|
||||
class Value : public FieldOperator
|
||||
{
|
||||
public:
|
||||
Value(std::string field_label) : FieldOperator(field_label) {};
|
||||
};
|
||||
|
||||
class Gradient : public FieldOperator
|
||||
{
|
||||
public:
|
||||
Gradient(std::string field_label) : FieldOperator(field_label) {};
|
||||
};
|
||||
|
||||
class Curl : public FieldOperator
|
||||
{
|
||||
public:
|
||||
Curl(std::string field_label) : FieldOperator(field_label) {};
|
||||
};
|
||||
|
||||
class Div : public FieldOperator
|
||||
{
|
||||
public:
|
||||
Div(std::string field_label) : FieldOperator(field_label) {};
|
||||
};
|
||||
|
||||
class FaceValueLeft : public FieldOperator
|
||||
{
|
||||
public:
|
||||
FaceValueLeft(std::string field_label) : FieldOperator(field_label) {};
|
||||
};
|
||||
|
||||
class FaceValueRight : public FieldOperator
|
||||
{
|
||||
public:
|
||||
FaceValueRight(std::string field_label) : FieldOperator(field_label) {};
|
||||
};
|
||||
|
||||
class FaceNormal : public FieldOperator
|
||||
{
|
||||
public:
|
||||
FaceNormal(std::string field_label) : FieldOperator(field_label) {};
|
||||
};
|
||||
|
||||
class One : public FieldOperator
|
||||
{
|
||||
public:
|
||||
One(std::string field_label) : FieldOperator(field_label) {};
|
||||
};
|
||||
|
||||
namespace BareFieldOperator
|
||||
{
|
||||
|
||||
struct Base
|
||||
{
|
||||
Base(FieldOperator &o)
|
||||
{
|
||||
size_on_qp = o.size_on_qp;
|
||||
dim = o.dim;
|
||||
vdim = o.vdim;
|
||||
};
|
||||
int size_on_qp = -1;
|
||||
int dim = -1;
|
||||
int vdim = -1;
|
||||
};
|
||||
|
||||
struct None : Base
|
||||
{
|
||||
None(FieldOperator &o) : Base(o) {}
|
||||
};
|
||||
|
||||
struct Weight : Base
|
||||
{
|
||||
Weight(FieldOperator &o) : Base(o) {}
|
||||
};
|
||||
|
||||
struct Value : Base
|
||||
{
|
||||
Value(FieldOperator &o) : Base(o) {}
|
||||
};
|
||||
|
||||
struct Gradient : Base
|
||||
{
|
||||
Gradient(FieldOperator &o) : Base(o) {}
|
||||
};
|
||||
|
||||
}
|
||||
@@ -0,0 +1,292 @@
|
||||
#pragma once
|
||||
|
||||
#include "dfem_util.hpp"
|
||||
#include <type_traits>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields_impl(DeviceTensor<2, double> &y,
|
||||
const DeviceTensor<3, double> &f,
|
||||
const output_t &output,
|
||||
const DofToQuadMap &dtq)
|
||||
{
|
||||
auto B = dtq.B;
|
||||
auto G = dtq.G;
|
||||
// assuming the quadrature point residual has to "play nice with
|
||||
// the test function"
|
||||
if constexpr (std::is_same_v<std::decay_t<output_t>, BareFieldOperator::Value>)
|
||||
{
|
||||
const auto [num_qp, cdim, num_dof] = B.GetShape();
|
||||
const int vdim = output.vdim > 0 ? output.vdim : cdim ;
|
||||
for (int dof = 0; dof < num_dof; dof++)
|
||||
{
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
acc += B(qp, 0, dof) * f(vd, 0, qp);
|
||||
}
|
||||
y(dof, vd) += acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<output_t>, BareFieldOperator::Gradient>)
|
||||
{
|
||||
const auto [num_qp, dim, num_dof] = G.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
for (int dof = 0; dof < num_dof; dof++)
|
||||
{
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
acc += G(qp, d, dof) * f(vd, d, qp);
|
||||
}
|
||||
}
|
||||
y(dof, vd) += acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
// else if constexpr (std::is_same_v<std::decay_t<output_t>, One>)
|
||||
// {
|
||||
// // This is the "integral over all quadrature points type" applying
|
||||
// // B = 1 s.t. B^T * C \in R^1.
|
||||
// const auto [a, b, num_qp] = B.GetShape();
|
||||
// auto cc = Reshape(&c(0, 0, 0), num_qp);
|
||||
// for (int i = 0; i < num_qp; i++)
|
||||
// {
|
||||
// y(0, 0) += cc(i);
|
||||
// }
|
||||
// }
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<output_t>, BareFieldOperator::None>)
|
||||
{
|
||||
const auto [vdim, dim, num_qp] = G.GetShape();
|
||||
auto cc = Reshape(&f(0, 0, 0), num_qp * vdim);
|
||||
auto yy = Reshape(&y(0, 0), num_qp * vdim);
|
||||
for (int i = 0; i < num_qp * vdim; i++)
|
||||
{
|
||||
yy(i) = cc(i);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("quadrature data mapping to field is not implemented for"
|
||||
" this field descriptor");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields_tensor_impl(DeviceTensor<2, double> &y,
|
||||
const DeviceTensor<3, double> &f,
|
||||
const output_t &output,
|
||||
const DofToQuadMap &dtq,
|
||||
std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
auto B = dtq.B;
|
||||
auto G = dtq.G;
|
||||
|
||||
if constexpr (std::is_same_v<std::decay_t<output_t>, BareFieldOperator::Value>)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
const int test_dim = output.size_on_qp / vdim;
|
||||
|
||||
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d, q1d, q1d);
|
||||
auto yd = Reshape(&y(0, 0), d1d, d1d, d1d, vdim);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), q1d, q1d, d1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), q1d, d1d, d1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
acc += fqp(vd, 0, qx, qy, qz) * B(qx, 0, dx);
|
||||
}
|
||||
s0(qz, qy, dx) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int qy = 0; qy < q1d; qy++)
|
||||
{
|
||||
acc += s0(qz, qy, dx) * B(qy, 0, dy);
|
||||
}
|
||||
s1(qz, dy, dx) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int qz = 0; qz < q1d; qz++)
|
||||
{
|
||||
acc += s1(qz, dy, dx) * B(qz, 0, dz);
|
||||
}
|
||||
yd(dx, dy, dz, vd) += acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<output_t>, BareFieldOperator::Gradient>)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = G.GetShape();
|
||||
const int vdim = output.vdim;
|
||||
const int test_dim = output.size_on_qp / vdim;
|
||||
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d, q1d, q1d);
|
||||
auto yd = Reshape(&y(0, 0), d1d, d1d, d1d, vdim);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), q1d, q1d, d1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), q1d, q1d, d1d);
|
||||
auto s2 = Reshape(&scratch_mem[2](0), q1d, q1d, d1d);
|
||||
auto s3 = Reshape(&scratch_mem[3](0), q1d, d1d, d1d);
|
||||
auto s4 = Reshape(&scratch_mem[4](0), q1d, d1d, d1d);
|
||||
auto s5 = Reshape(&scratch_mem[5](0), q1d, d1d, d1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int qx = 0; qx < q1d; qx++)
|
||||
{
|
||||
uvw[0] += fqp(vd, 0, qx, qy, qz) * G(qx, 0, dx);
|
||||
uvw[1] += fqp(vd, 1, qx, qy, qz) * B(qx, 0, dx);
|
||||
uvw[2] += fqp(vd, 2, qx, qy, qz) * B(qx, 0, dx);
|
||||
}
|
||||
s0(qz, qy, dx) = uvw[0];
|
||||
s1(qz, qy, dx) = uvw[1];
|
||||
s2(qz, qy, dx) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int qy = 0; qy < q1d; qy++)
|
||||
{
|
||||
uvw[0] += s0(qz, qy, dx) * B(qy, 0, dy);
|
||||
uvw[1] += s1(qz, qy, dx) * G(qy, 0, dy);
|
||||
uvw[2] += s2(qz, qy, dx) * B(qy, 0, dy);
|
||||
}
|
||||
s3(qz, dy, dx) = uvw[0];
|
||||
s4(qz, dy, dx) = uvw[1];
|
||||
s5(qz, dy, dx) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dx, x, d1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int qz = 0; qz < q1d; qz++)
|
||||
{
|
||||
uvw[0] += s3(qz, dy, dx) * B(qz, 0, dz);
|
||||
uvw[1] += s4(qz, dy, dx) * B(qz, 0, dz);
|
||||
uvw[2] += s5(qz, dy, dx) * G(qz, 0, dz);
|
||||
}
|
||||
yd(dx, dy, dz, vd) += uvw[0] + uvw[1] + uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<output_t>, BareFieldOperator::None>)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
auto fqp = Reshape(&f(0, 0, 0), output.size_on_qp, q1d, q1d, q1d);
|
||||
auto yqp = Reshape(&y(0, 0), output.size_on_qp, q1d, q1d, q1d);
|
||||
|
||||
for (int sq = 0; sq < output.size_on_qp; sq++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
yqp(sq, qx, qy, qz) = fqp(sq, qx, qy, qz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("quadrature data mapping to field is not implemented for"
|
||||
" this field descriptor with sum factorization on tensor product elements");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T = NonTensorProduct, typename output_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_quadrature_data_to_fields(DeviceTensor<2, double> &y,
|
||||
const DeviceTensor<3, double> &f,
|
||||
const output_t &output,
|
||||
const DofToQuadMap &dtq,
|
||||
std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
if constexpr (std::is_same_v<T, NonTensorProduct>)
|
||||
{
|
||||
map_quadrature_data_to_fields_impl(y, f, output, dtq);
|
||||
}
|
||||
else if constexpr (std::is_same_v<T, TensorProduct>)
|
||||
{
|
||||
map_quadrature_data_to_fields_tensor_impl(y, f, output, dtq, scratch_mem);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,400 @@
|
||||
#pragma once
|
||||
|
||||
#include "dfem_util.hpp"
|
||||
#include <type_traits>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
template <typename field_operator_t>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void map_field_to_quadrature_data_tensor_product(
|
||||
DeviceTensor<2> &field_qp,
|
||||
const DofToQuadMap &dtq,
|
||||
const DeviceTensor<1> &field_e,
|
||||
const field_operator_t &input,
|
||||
const DeviceTensor<1, const double> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem)
|
||||
{
|
||||
auto B = dtq.B;
|
||||
auto G = dtq.G;
|
||||
|
||||
if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, BareFieldOperator::Value>)
|
||||
{
|
||||
auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, q1d, q1d, q1d);
|
||||
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), d1d, q1d, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
acc += B(qx, 0, dx) * field(dx, dy, dz, vd);
|
||||
}
|
||||
s0(dz, dy, qx) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int dy = 0; dy < d1d; dy++)
|
||||
{
|
||||
acc += s0(dz, dy, qx) * B(qy, 0, dy);
|
||||
}
|
||||
s1(dz, qy, qx) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int dz = 0; dz < d1d; dz++)
|
||||
{
|
||||
acc += s1(dz, qy, qx) * B(qz, 0, dz);
|
||||
}
|
||||
fqp(vd, qx, qy, qz) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, BareFieldOperator::Gradient>)
|
||||
{
|
||||
const auto [q1d, unused, d1d] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const int dim = input.dim;
|
||||
const auto field = Reshape(&field_e[0], d1d, d1d, d1d, vdim);
|
||||
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d, q1d, q1d);
|
||||
|
||||
auto s0 = Reshape(&scratch_mem[0](0), d1d, d1d, q1d);
|
||||
auto s1 = Reshape(&scratch_mem[1](0), d1d, d1d, q1d);
|
||||
auto s2 = Reshape(&scratch_mem[2](0), d1d, q1d, q1d);
|
||||
auto s3 = Reshape(&scratch_mem[3](0), d1d, q1d, q1d);
|
||||
auto s4 = Reshape(&scratch_mem[4](0), d1d, q1d, q1d);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(dy, y, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t uv[2] = {0.0, 0.0};
|
||||
for (int dx = 0; dx < d1d; dx++)
|
||||
{
|
||||
const real_t f = field(dx, dy, dz, vd);
|
||||
uv[0] += f * B(qx, 0, dx);
|
||||
uv[1] += f * G(qx, 0, dx);
|
||||
}
|
||||
s0(dz, dy, qx) = uv[0];
|
||||
s1(dz, dy, qx) = uv[1];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(dz, z, d1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int dy = 0; dy < d1d; dy++)
|
||||
{
|
||||
const real_t s0i = s0(dz, dy, qx);
|
||||
uvw[0] += s1(dz, dy, qx) * B(qy, 0, dy);
|
||||
uvw[1] += s0i * G(qy, 0, dy);
|
||||
uvw[2] += s0i * B(qy, 0, dy);
|
||||
}
|
||||
s2(dz, qy, qx) = uvw[0];
|
||||
s3(dz, qy, qx) = uvw[1];
|
||||
s4(dz, qy, qx) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
real_t uvw[3] = {0.0, 0.0, 0.0};
|
||||
for (int dz = 0; dz < d1d; dz++)
|
||||
{
|
||||
uvw[0] += s2(dz, qy, qx) * B(qz, 0, dz);
|
||||
uvw[1] += s3(dz, qy, qx) * B(qz, 0, dz);
|
||||
uvw[2] += s4(dz, qy, qx) * G(qz, 0, dz);
|
||||
}
|
||||
fqp(vd, 0, qx, qy, qz) = uvw[0];
|
||||
fqp(vd, 1, qx, qy, qz) = uvw[1];
|
||||
fqp(vd, 2, qx, qy, qz) = uvw[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
}
|
||||
// TODO: Create separate function for clarity
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, BareFieldOperator::Weight>)
|
||||
{
|
||||
const int num_qp = integration_weights.GetShape()[0];
|
||||
// TODO: eeek
|
||||
const int q1d = (int)floor(pow(num_qp, 1.0/input.dim) + 0.5);
|
||||
auto w = Reshape(&integration_weights[0], q1d, q1d, q1d);
|
||||
auto f = Reshape(&field_qp[0], q1d, q1d, q1d);
|
||||
MFEM_FOREACH_THREAD(qx, x, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qy, y, q1d)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(qz, z, q1d)
|
||||
{
|
||||
f(qx, qy, qz) = w(qx, qy, qz);
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
else if constexpr (
|
||||
std::is_same_v<std::decay_t<field_operator_t>, BareFieldOperator::None>)
|
||||
{
|
||||
const int q1d = B.GetShape()[0];
|
||||
auto field = Reshape(&field_e[0], input.size_on_qp, q1d * q1d * q1d);
|
||||
field_qp = field;
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(always_false<std::decay_t<field_operator_t>>,
|
||||
"can't map field to quadrature data");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
template <typename field_operator_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_field_to_quadrature_data(
|
||||
DeviceTensor<2> field_qp,
|
||||
const DofToQuadMap &dtq,
|
||||
const DeviceTensor<1, const double> &field_e,
|
||||
field_operator_t &input,
|
||||
DeviceTensor<1, const double> integration_weights)
|
||||
{
|
||||
auto B = dtq.B;
|
||||
auto G = dtq.G;
|
||||
if constexpr (std::is_same_v<field_operator_t, BareFieldOperator::Value>)
|
||||
{
|
||||
auto [num_qp, dim, num_dof] = B.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e(0), num_dof, vdim);
|
||||
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int dof = 0; dof < num_dof; dof++)
|
||||
{
|
||||
acc += B(qp, 0, dof) * field(dof, vd);
|
||||
}
|
||||
field_qp(vd, qp) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if constexpr (
|
||||
std::is_same_v<field_operator_t, BareFieldOperator::Gradient>)
|
||||
{
|
||||
const auto [num_qp, dim, num_dof] = G.GetShape();
|
||||
const int vdim = input.vdim;
|
||||
const auto field = Reshape(&field_e(0), num_dof, vdim);
|
||||
|
||||
auto f = Reshape(&field_qp[0], vdim, dim, num_qp);
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
for (int vd = 0; vd < vdim; vd++)
|
||||
{
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
double acc = 0.0;
|
||||
for (int dof = 0; dof < num_dof; dof++)
|
||||
{
|
||||
acc += G(qp, d, dof) * field(dof, vd);
|
||||
}
|
||||
f(vd, d, qp) = acc;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// else if constexpr (std::is_same_v<field_operator_t, FaceNormal>)
|
||||
// {
|
||||
// auto normal = geometric_factors.normal;
|
||||
// auto [num_qp, dim, num_entities] = normal.GetShape();
|
||||
// auto f = Reshape(&field_qp[0], dim, num_qp);
|
||||
// for (int qp = 0; qp < num_qp; qp++)
|
||||
// {
|
||||
// for (int d = 0; d < dim; d++)
|
||||
// {
|
||||
// f(d, qp) = normal(qp, d, entity_idx);
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
// TODO: Create separate function for clarity
|
||||
else if constexpr (std::is_same_v<field_operator_t, BareFieldOperator::Weight>)
|
||||
{
|
||||
const int num_qp = integration_weights.GetShape()[0];
|
||||
auto f = Reshape(&field_qp[0], num_qp);
|
||||
for (int qp = 0; qp < num_qp; qp++)
|
||||
{
|
||||
f(qp) = integration_weights(qp);
|
||||
}
|
||||
}
|
||||
else if constexpr (std::is_same_v<field_operator_t, BareFieldOperator::None>)
|
||||
{
|
||||
auto [num_qp, unused, num_dof] = B.GetShape();
|
||||
const int size_on_qp = input.size_on_qp;
|
||||
const auto field = Reshape(&field_e[0], size_on_qp * num_qp);
|
||||
auto f = Reshape(&field_qp[0], size_on_qp * num_qp);
|
||||
for (int i = 0; i < size_on_qp * num_qp; i++)
|
||||
{
|
||||
f(i) = field(i);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(always_false<field_operator_t>,
|
||||
"can't map field to quadrature data");
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T = NonTensorProduct, size_t num_kinputs, typename field_operator_ts, std::size_t... i>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void map_fields_to_quadrature_data(
|
||||
std::array<DeviceTensor<2>, num_kinputs> &fields_qp,
|
||||
const std::array<DeviceTensor<1>, num_kinputs> &fields_e,
|
||||
const std::array<DofToQuadMap, num_kinputs> &dtqmaps,
|
||||
const field_operator_ts &fops,
|
||||
const DeviceTensor<1, const double> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
std::index_sequence<i...>)
|
||||
{
|
||||
if constexpr (std::is_same_v<T, TensorProduct>)
|
||||
{
|
||||
|
||||
(map_field_to_quadrature_data_tensor_product(fields_qp[i],
|
||||
dtqmaps[i], fields_e[i],
|
||||
mfem::get<i>(fops), integration_weights,
|
||||
scratch_mem),
|
||||
...);
|
||||
}
|
||||
else
|
||||
{
|
||||
(map_field_to_quadrature_data(fields_qp[i],
|
||||
dtqmaps[i], fields_e[i],
|
||||
mfem::get<i>(fops), integration_weights),
|
||||
...);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename field_operator_t>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_field_to_quadrature_data_conditional(
|
||||
DeviceTensor<2> &field_qp,
|
||||
const DeviceTensor<1> &field_e,
|
||||
const DofToQuadMap &dtqmap,
|
||||
field_operator_t &fop,
|
||||
const DeviceTensor<1, const double> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
const bool &condition)
|
||||
{
|
||||
if (condition)
|
||||
{
|
||||
if constexpr (std::is_same_v<T, TensorProduct>)
|
||||
{
|
||||
map_field_to_quadrature_data_tensor_product(field_qp, dtqmap,
|
||||
field_e, fop,
|
||||
integration_weights,
|
||||
scratch_mem);
|
||||
}
|
||||
else
|
||||
{
|
||||
map_field_to_quadrature_data(field_qp, dtqmap, field_e, fop,
|
||||
integration_weights);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T = NonTensorProduct, size_t num_fields, size_t num_kinputs, typename field_operator_ts, std::size_t... i>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_fields_to_quadrature_data_conditional(
|
||||
std::array<DeviceTensor<2>, num_kinputs> &fields_qp,
|
||||
const std::array<DeviceTensor<1, const double>, num_fields> &fields_e,
|
||||
const std::array<DofToQuadMap, num_kinputs> &dtqmaps,
|
||||
field_operator_ts fops,
|
||||
const DeviceTensor<1, const double> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
const std::array<bool, num_kinputs> &conditions,
|
||||
std::index_sequence<i...>)
|
||||
{
|
||||
(map_field_to_quadrature_data_conditional<T>(fields_qp[i],
|
||||
fields_e[i],
|
||||
dtqmaps[i],
|
||||
mfem::get<i>(fops),
|
||||
integration_weights,
|
||||
scratch_mem,
|
||||
conditions[i]),
|
||||
...);
|
||||
}
|
||||
|
||||
template <typename T = NonTensorProduct, size_t num_kinputs, typename field_operator_ts, std::size_t... i>
|
||||
MFEM_HOST_DEVICE
|
||||
void map_direction_to_quadrature_data_conditional(
|
||||
std::array<DeviceTensor<2>, num_kinputs> &directions_qp,
|
||||
const DeviceTensor<1> &direction_e,
|
||||
const std::array<DofToQuadMap, num_kinputs> &dtqmaps,
|
||||
field_operator_ts fops,
|
||||
const DeviceTensor<1, const double> &integration_weights,
|
||||
const std::array<DeviceTensor<1>, 6> &scratch_mem,
|
||||
const std::array<bool, num_kinputs> &conditions,
|
||||
std::index_sequence<i...>)
|
||||
{
|
||||
(map_field_to_quadrature_data_conditional<T>(directions_qp[i],
|
||||
direction_e,
|
||||
dtqmaps[i],
|
||||
mfem::get<i>(fops),
|
||||
integration_weights,
|
||||
scratch_mem,
|
||||
conditions[i]),
|
||||
...);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
#pragma once
|
||||
|
||||
#include <mfem.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class ParametricSpace
|
||||
{
|
||||
|
||||
public:
|
||||
ParametricSpace(int spatial_dim, int local_size, int element_size,
|
||||
int total_size) :
|
||||
spatial_dim(spatial_dim),
|
||||
local_size(local_size),
|
||||
element_size(element_size),
|
||||
total_size(total_size),
|
||||
identity(total_size)
|
||||
{
|
||||
dtq.ndof = (int)floor(pow(element_size, 1.0/spatial_dim) + 0.5);
|
||||
dtq.nqpt = dtq.ndof;
|
||||
}
|
||||
|
||||
ParametricSpace(int local_size) :
|
||||
local_size(local_size),
|
||||
element_size(local_size),
|
||||
total_size(local_size),
|
||||
identity(local_size)
|
||||
{
|
||||
dtq.ndof = (int)floor(pow(element_size, 1.0/spatial_dim) + 0.5);
|
||||
dtq.nqpt = dtq.ndof;
|
||||
}
|
||||
|
||||
int Dimension() const
|
||||
{
|
||||
return spatial_dim;
|
||||
}
|
||||
|
||||
int GetLocalSize() const
|
||||
{
|
||||
return local_size;
|
||||
}
|
||||
|
||||
int GetElementSize() const
|
||||
{
|
||||
return element_size;
|
||||
}
|
||||
|
||||
int GetTotalSize() const
|
||||
{
|
||||
return total_size;
|
||||
}
|
||||
|
||||
const DofToQuad &GetDofToQuad() const
|
||||
{
|
||||
return dtq;
|
||||
}
|
||||
|
||||
const Operator *GetProlongation() const
|
||||
{
|
||||
return &identity;
|
||||
}
|
||||
|
||||
const Operator *GetRestriction() const
|
||||
{
|
||||
return &identity;
|
||||
}
|
||||
|
||||
private:
|
||||
int spatial_dim;
|
||||
|
||||
// Hint for the local dimension. E.g. the size on the quadrature point or vdim.
|
||||
int local_size;
|
||||
|
||||
// Size of the data on an element
|
||||
int element_size;
|
||||
|
||||
int total_size;
|
||||
|
||||
IdentityOperator identity;
|
||||
|
||||
DofToQuad dtq;
|
||||
};
|
||||
|
||||
class ParametricFunction : public Vector
|
||||
{
|
||||
public:
|
||||
ParametricFunction(ParametricSpace &space) :
|
||||
Vector(space.GetTotalSize()),
|
||||
space(space)
|
||||
{}
|
||||
|
||||
ParametricSpace &space;
|
||||
|
||||
using Vector::operator=;
|
||||
|
||||
};
|
||||
|
||||
}
|
||||
@@ -0,0 +1,268 @@
|
||||
#pragma once
|
||||
#include "dfem_util.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_kf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
double &arg)
|
||||
{
|
||||
arg = u(0);
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_kf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
internal::tensor<double> &arg)
|
||||
{
|
||||
arg(0) = u(0);
|
||||
}
|
||||
|
||||
template <typename T, int n>
|
||||
MFEM_HOST_DEVICE
|
||||
void process_kf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
internal::tensor<T, n> &arg)
|
||||
{
|
||||
for (int i = 0; i < n; i++)
|
||||
{
|
||||
arg(i) = u(i);
|
||||
}
|
||||
}
|
||||
|
||||
template <int n, int m>
|
||||
MFEM_HOST_DEVICE
|
||||
void process_kf_arg(
|
||||
const DeviceTensor<1> &u,
|
||||
internal::tensor<double, n, m> &arg)
|
||||
{
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i) = u((i * m) + j);
|
||||
}
|
||||
}
|
||||
// assuming col major layout. translating to row major.
|
||||
// i + N_i*j
|
||||
// arg(0, 0) = u(0);
|
||||
// arg(0, 1) = u(0 + 2 * 1);
|
||||
// arg(1, 0) = u(1 + 2 * 0);
|
||||
// arg(1, 1) = u(1 + 2 * 1);
|
||||
}
|
||||
|
||||
template <typename arg_type>
|
||||
MFEM_HOST_DEVICE
|
||||
void process_kf_arg(const DeviceTensor<2> &u, arg_type &arg, int qp)
|
||||
{
|
||||
// out << "qp: " << qp << "\n";
|
||||
// for (int i = 0; i < u.GetShape()[0] * u.GetShape()[1]; i++)
|
||||
// {
|
||||
// out << (&u(0, 0))[i] << " ";
|
||||
// }
|
||||
// out << "\n";
|
||||
|
||||
const auto u_qp = Reshape(&u(0, qp), u.GetShape()[0]);
|
||||
// for (int i = 0; i < u_qp.GetShape()[0]; i++)
|
||||
// {
|
||||
// out << (&u_qp(0))[i] << " ";
|
||||
// }
|
||||
// out << "\n";
|
||||
|
||||
process_kf_arg(u_qp, arg);
|
||||
}
|
||||
|
||||
template <size_t num_fields, typename kf_args, std::size_t... i>
|
||||
MFEM_HOST_DEVICE
|
||||
void process_kf_args(const std::array<DeviceTensor<2>, num_fields> &u,
|
||||
kf_args &args, int qp, std::index_sequence<i...>)
|
||||
{
|
||||
(process_kf_arg(u[i], mfem::get<i>(args), qp), ...);
|
||||
}
|
||||
|
||||
template <typename T0, typename T1> inline
|
||||
Vector process_kf_result(T0, T1)
|
||||
{
|
||||
static_assert(always_false<T0, T1>,
|
||||
"process_kf_result not implemented for result type");
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_kf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const double &x)
|
||||
{
|
||||
r(0) = x;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_kf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const internal::tensor<T> &x)
|
||||
{
|
||||
r(0) = x(0);
|
||||
}
|
||||
|
||||
template <typename T, int n>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_kf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const internal::tensor<T, n> &x)
|
||||
{
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
r(i) = x(i);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int n, int m>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void process_kf_result(
|
||||
DeviceTensor<1, T> &r,
|
||||
const internal::tensor<T, n, m> &x)
|
||||
{
|
||||
// out << "x: " << x << "\n";
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
for (size_t j = 0; j < m; j++)
|
||||
{
|
||||
r(i + n * j) = x(i, j);
|
||||
}
|
||||
}
|
||||
|
||||
// out << "r: ";
|
||||
// for (int i = 0; i < r.GetShape()[0]; i++)
|
||||
// {
|
||||
// out << r(i) << " ";
|
||||
// }
|
||||
// out << "\n\n";
|
||||
}
|
||||
|
||||
template <typename T> inline
|
||||
void process_kf_arg(const DeviceTensor<1> &u, const DeviceTensor<1> &v,
|
||||
double &arg)
|
||||
{
|
||||
arg = u(0);
|
||||
}
|
||||
|
||||
template <int n, int m> inline
|
||||
void process_kf_arg(const DeviceTensor<1> &u, const DeviceTensor<1> &v,
|
||||
internal::tensor<double, n, m> &arg)
|
||||
{
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
for (int j = 0; j < n; j++)
|
||||
{
|
||||
arg(j, i) = u((i * m) + j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename arg_type> inline
|
||||
void process_kf_arg(const DeviceTensor<2> &u, const DeviceTensor<2> &v,
|
||||
arg_type &arg, int qp)
|
||||
{
|
||||
const auto u_qp = Reshape(&u(0, qp), u.GetShape()[0]);
|
||||
const auto v_qp = Reshape(&v(0, qp), v.GetShape()[0]);
|
||||
process_kf_arg(u_qp, v_qp, arg);
|
||||
}
|
||||
|
||||
template <size_t num_fields, typename kf_args, std::size_t... i> inline
|
||||
void process_kf_args(std::array<DeviceTensor<2>, num_fields> &u,
|
||||
std::array<DeviceTensor<2>, num_fields> &v,
|
||||
kf_args &args, int qp, std::index_sequence<i...>)
|
||||
{
|
||||
(process_kf_arg(u[i], v[i], mfem::get<i>(args), qp), ...);
|
||||
}
|
||||
|
||||
template <typename kernel_func_t, typename kernel_args_ts, size_t num_args>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void apply_kernel(
|
||||
DeviceTensor<1, double> &f_qp,
|
||||
const kernel_func_t &kf,
|
||||
kernel_args_ts &args,
|
||||
const std::array<DeviceTensor<2>, num_args> &u,
|
||||
int qp)
|
||||
{
|
||||
process_kf_args(u, args, qp,
|
||||
std::make_index_sequence<mfem::tuple_size<kernel_args_ts>::value> {});
|
||||
|
||||
process_kf_result(f_qp, mfem::get<0>(mfem::apply(kf, args)));
|
||||
}
|
||||
|
||||
// Version for active function arguments only
|
||||
//
|
||||
// This is an Enzyme regression and can be removed in later versions.
|
||||
template <typename kernel_t, typename arg_ts, std::size_t... Is,
|
||||
typename inactive_arg_ts>
|
||||
inline auto fwddiff_apply_enzyme_indexed(kernel_t kernel, arg_ts &&args,
|
||||
arg_ts &&shadow_args,
|
||||
std::index_sequence<Is...>,
|
||||
inactive_arg_ts &&inactive_args,
|
||||
std::index_sequence<>)
|
||||
{
|
||||
using kf_return_t = typename create_function_signature<
|
||||
decltype(&kernel_t::operator())>::type::return_t;
|
||||
return __enzyme_fwddiff<kf_return_t>(
|
||||
+kernel, enzyme_dup, &mfem::get<Is>(args)..., enzyme_interleave,
|
||||
&mfem::get<Is>(shadow_args)...);
|
||||
}
|
||||
|
||||
// Interleave function arguments for enzyme
|
||||
template <typename kernel_t, typename arg_ts, std::size_t... Is,
|
||||
typename inactive_arg_ts, std::size_t... Js>
|
||||
inline auto fwddiff_apply_enzyme_indexed(kernel_t kernel, arg_ts &&args,
|
||||
arg_ts &&shadow_args,
|
||||
std::index_sequence<Is...>,
|
||||
inactive_arg_ts &&inactive_args,
|
||||
std::index_sequence<Js...>)
|
||||
{
|
||||
using kf_return_t = typename create_function_signature<
|
||||
decltype(&kernel_t::operator())>::type::return_t;
|
||||
return __enzyme_fwddiff<kf_return_t>(
|
||||
+kernel, enzyme_dup, &std::get<Is>(args)..., enzyme_const,
|
||||
&mfem::get<Js>(inactive_args)..., enzyme_interleave,
|
||||
&mfem::get<Is>(shadow_args)...);
|
||||
}
|
||||
|
||||
template <typename kernel_t, typename arg_ts, typename inactive_arg_ts>
|
||||
inline auto fwddiff_apply_enzyme(kernel_t kernel, arg_ts &&args,
|
||||
arg_ts &&shadow_args,
|
||||
inactive_arg_ts &&inactive_args)
|
||||
{
|
||||
auto arg_indices = std::make_index_sequence<
|
||||
mfem::tuple_size<std::remove_reference_t<arg_ts>>::value> {};
|
||||
|
||||
auto inactive_arg_indices = std::make_index_sequence<
|
||||
mfem::tuple_size<std::remove_reference_t<inactive_arg_ts>>::value> {};
|
||||
|
||||
return fwddiff_apply_enzyme_indexed(kernel, args, shadow_args, arg_indices,
|
||||
inactive_args, inactive_arg_indices);
|
||||
}
|
||||
|
||||
template <typename kf_t, typename kernel_arg_ts, size_t num_args>
|
||||
MFEM_HOST_DEVICE inline
|
||||
void apply_kernel_fwddiff_enzyme(
|
||||
DeviceTensor<1, double> &f_qp,
|
||||
const kf_t &kf,
|
||||
kernel_arg_ts &args,
|
||||
const std::array<DeviceTensor<2>, num_args> &u,
|
||||
kernel_arg_ts &shadow_args,
|
||||
const std::array<DeviceTensor<2>, num_args> &v,
|
||||
int qp_idx)
|
||||
{
|
||||
process_kf_args(u, args, qp_idx,
|
||||
std::make_index_sequence<mfem::tuple_size<kernel_arg_ts>::value> {});
|
||||
|
||||
process_kf_args(v, shadow_args, qp_idx,
|
||||
std::make_index_sequence<mfem::tuple_size<kernel_arg_ts>::value> {});
|
||||
|
||||
process_kf_result(f_qp,
|
||||
mfem::get<0>(fwddiff_apply_enzyme(kf, args, shadow_args, mfem::tuple<> {})));
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,116 @@
|
||||
#pragma once
|
||||
|
||||
#include <mfem.hpp>
|
||||
|
||||
class SharedMemoryManager
|
||||
{
|
||||
private:
|
||||
struct MemoryBlock
|
||||
{
|
||||
char* ptr;
|
||||
int size;
|
||||
bool used;
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE static const int MAX_BLOCKS = 16;
|
||||
MFEM_HOST_DEVICE static MemoryBlock blocks[MAX_BLOCKS];
|
||||
MFEM_HOST_DEVICE static int num_blocks;
|
||||
MFEM_HOST_DEVICE static char* base_ptr;
|
||||
|
||||
public:
|
||||
MFEM_HOST_DEVICE static void init(void* shmem, int total_size)
|
||||
{
|
||||
base_ptr = static_cast<char*>(shmem);
|
||||
num_blocks = 1;
|
||||
blocks[0] = {base_ptr, total_size, false};
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
MFEM_HOST_DEVICE static T* reserve(int n)
|
||||
{
|
||||
int size_bytes = n * sizeof(T);
|
||||
for (int i = 0; i < num_blocks; ++i)
|
||||
{
|
||||
if (!blocks[i].used && blocks[i].size >= size_bytes)
|
||||
{
|
||||
blocks[i].used = true;
|
||||
if (blocks[i].size > size_bytes)
|
||||
{
|
||||
// Split block
|
||||
if (num_blocks < MAX_BLOCKS)
|
||||
{
|
||||
blocks[num_blocks] = {blocks[i].ptr + size_bytes, blocks[i].size - size_bytes, false};
|
||||
++num_blocks;
|
||||
blocks[i].size = size_bytes;
|
||||
}
|
||||
}
|
||||
return reinterpret_cast<T*>(blocks[i].ptr);
|
||||
}
|
||||
}
|
||||
return nullptr; // Allocation failed
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE static void release(void* ptr)
|
||||
{
|
||||
for (int i = 0; i < num_blocks; ++i)
|
||||
{
|
||||
if (blocks[i].ptr == ptr)
|
||||
{
|
||||
blocks[i].used = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE static void release_and_try_merge(void* ptr)
|
||||
{
|
||||
for (int i = 0; i < num_blocks; ++i)
|
||||
{
|
||||
if (blocks[i].ptr == ptr)
|
||||
{
|
||||
blocks[i].used = false;
|
||||
merge_adjacent_free_blocks();
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
MFEM_HOST_DEVICE static void merge_adjacent_free_blocks()
|
||||
{
|
||||
// Simple bubble sort for simplicity (can be optimized)
|
||||
for (int i = 0; i < num_blocks - 1; ++i)
|
||||
{
|
||||
for (int j = 0; j < num_blocks - i - 1; ++j)
|
||||
{
|
||||
if (blocks[j].ptr > blocks[j + 1].ptr)
|
||||
{
|
||||
MemoryBlock temp = blocks[j];
|
||||
blocks[j] = blocks[j + 1];
|
||||
blocks[j + 1] = temp;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int i = 0; i < num_blocks - 1; ++i)
|
||||
{
|
||||
if (!blocks[i].used && !blocks[i + 1].used)
|
||||
{
|
||||
blocks[i].size += blocks[i + 1].size;
|
||||
for (int j = i + 1; j < num_blocks - 1; ++j)
|
||||
{
|
||||
blocks[j] = blocks[j + 1];
|
||||
}
|
||||
--num_blocks;
|
||||
--i;
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
MFEM_HOST_DEVICE SharedMemoryManager::MemoryBlock
|
||||
SharedMemoryManager::blocks[SharedMemoryManager::MAX_BLOCKS];
|
||||
|
||||
MFEM_HOST_DEVICE int SharedMemoryManager::num_blocks;
|
||||
|
||||
MFEM_HOST_DEVICE char* SharedMemoryManager::base_ptr;
|
||||
@@ -0,0 +1,39 @@
|
||||
#pragma once
|
||||
#include "dfem.hpp"
|
||||
|
||||
#define DFEM_TEST_MAIN(function) \
|
||||
int main(int argc, char* argv[]) \
|
||||
{ \
|
||||
Mpi::Init(); \
|
||||
\
|
||||
const char* device_config = "cpu"; \
|
||||
const char* mesh_file = "../data/ref-square.mesh"; \
|
||||
int polynomial_order = 1; \
|
||||
int ir_order = 2; \
|
||||
int refinements = 0; \
|
||||
\
|
||||
OptionsParser args(argc, argv); \
|
||||
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use."); \
|
||||
args.AddOption(&polynomial_order, "-o", "--order", ""); \
|
||||
args.AddOption(&refinements, "-r", "--r", ""); \
|
||||
args.AddOption(&ir_order, "-iro", "--iro", ""); \
|
||||
args.AddOption(&device_config, "-d", "--device", \
|
||||
"Device configuration string, see Device::Configure()."); \
|
||||
args.ParseCheck(); \
|
||||
\
|
||||
Device device(device_config); \
|
||||
if (Mpi::Root() == 0) \
|
||||
{ \
|
||||
device.Print(); \
|
||||
} \
|
||||
\
|
||||
out << std::setprecision(12); \
|
||||
\
|
||||
int ret; \
|
||||
\
|
||||
ret = function(mesh_file, refinements, polynomial_order); \
|
||||
out << #function; \
|
||||
ret ? out << " FAILURE\n" : out << " OK\n"; \
|
||||
\
|
||||
return ret; \
|
||||
}\
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,130 @@
|
||||
// SPDX-ArtifactOfProjectName: noisy
|
||||
// SPDX-ArtifactOfProjectHomePage: https://github.com/VincentZalzal/noisy
|
||||
// SPDX-FileCopyrightText: Copyright 2024 Vincent Zalzal
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
|
||||
namespace vz {
|
||||
|
||||
struct Counters {
|
||||
unsigned m_def_ctor = 0;
|
||||
unsigned m_copy_ctor = 0;
|
||||
unsigned m_move_ctor = 0;
|
||||
unsigned m_copy_assign = 0;
|
||||
unsigned m_move_assign = 0;
|
||||
unsigned m_dtor = 0;
|
||||
|
||||
void reset() {
|
||||
*this = {};
|
||||
}
|
||||
|
||||
bool leaks() const {
|
||||
return m_def_ctor + m_copy_ctor + m_move_ctor != m_dtor;
|
||||
}
|
||||
|
||||
friend std::ostream& operator<<(std::ostream& os, const Counters& c) {
|
||||
stream_counter(os, "Default constructor count: ", c.m_def_ctor );
|
||||
stream_counter(os, "Copy constructor count: ", c.m_copy_ctor );
|
||||
stream_counter(os, "Move constructor count: ", c.m_move_ctor );
|
||||
stream_counter(os, "Copy assignment count: ", c.m_copy_assign);
|
||||
stream_counter(os, "Move assignment count: ", c.m_move_assign);
|
||||
stream_counter(os, "Destructor count: ", c.m_dtor );
|
||||
return os;
|
||||
}
|
||||
|
||||
friend bool operator==(const Counters& lhs, const Counters& rhs) {
|
||||
return
|
||||
lhs.m_def_ctor == rhs.m_def_ctor &&
|
||||
lhs.m_copy_ctor == rhs.m_copy_ctor &&
|
||||
lhs.m_move_ctor == rhs.m_move_ctor &&
|
||||
lhs.m_copy_assign == rhs.m_copy_assign &&
|
||||
lhs.m_move_assign == rhs.m_move_assign &&
|
||||
lhs.m_dtor == rhs.m_dtor ;
|
||||
}
|
||||
|
||||
friend bool operator!=(const Counters& lhs, const Counters& rhs) { return !(lhs == rhs); }
|
||||
|
||||
private:
|
||||
static void stream_counter(std::ostream& os, const char* msg, unsigned value) {
|
||||
if (value != 0)
|
||||
os << msg << std::setw(2) << value << '\n';
|
||||
}
|
||||
};
|
||||
|
||||
namespace detail {
|
||||
|
||||
struct Globals {
|
||||
~Globals() {
|
||||
if (m_verbose)
|
||||
std::cout << "\n===== Noisy counters =====\n" << m_counters;
|
||||
}
|
||||
|
||||
Counters m_counters;
|
||||
unsigned m_next_id = 0;
|
||||
bool m_verbose = true;
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
class Noisy {
|
||||
private:
|
||||
static detail::Globals& globals() {
|
||||
static detail::Globals s_globals;
|
||||
return s_globals;
|
||||
}
|
||||
|
||||
public:
|
||||
static Counters& counters() { return globals().m_counters; }
|
||||
static void set_verbose(bool verbose) { globals().m_verbose = verbose; }
|
||||
|
||||
Noisy() {
|
||||
if (globals().m_verbose)
|
||||
std::cout << *this << ": default constructor\n";
|
||||
globals().m_counters.m_def_ctor++;
|
||||
}
|
||||
|
||||
Noisy(const Noisy& other) {
|
||||
if (globals().m_verbose)
|
||||
std::cout << *this << ": copy constructor from " << other << '\n';
|
||||
globals().m_counters.m_copy_ctor++;
|
||||
}
|
||||
|
||||
Noisy(Noisy&& other) noexcept {
|
||||
if (globals().m_verbose)
|
||||
std::cout << *this << ": move constructor from " << other << '\n';
|
||||
globals().m_counters.m_move_ctor++;
|
||||
}
|
||||
|
||||
~Noisy() {
|
||||
if (globals().m_verbose)
|
||||
std::cout << *this << ": destructor\n";
|
||||
globals().m_counters.m_dtor++;
|
||||
}
|
||||
|
||||
Noisy& operator=(const Noisy& other) {
|
||||
if (globals().m_verbose)
|
||||
std::cout << *this << ": copy assignment from " << other << '\n';
|
||||
globals().m_counters.m_copy_assign++;
|
||||
return *this;
|
||||
}
|
||||
|
||||
Noisy& operator=(Noisy&& other) noexcept {
|
||||
if (globals().m_verbose)
|
||||
std::cout << *this << ": move assignment from " << other << '\n';
|
||||
globals().m_counters.m_move_assign++;
|
||||
return *this;
|
||||
}
|
||||
|
||||
unsigned id() const { return m_id; }
|
||||
|
||||
friend std::ostream& operator<<(std::ostream& os, const Noisy& noisy) { return os << "Noisy(" << std::setw(2) << noisy.m_id << ')'; }
|
||||
|
||||
private:
|
||||
unsigned m_id = globals().m_next_id++;
|
||||
};
|
||||
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
* Calculate shared memory requirements
|
||||
* Interpolation and integration
|
||||
---
|
||||
* If grad involved, need B and G
|
||||
* Fit largest field, depends on polynomial order (#dofs)
|
||||
-> vdim is irrelevant
|
||||
* Temporaries for each sum
|
||||
- DDQ (d1d x d1d x q1d) x 2 -> DDQ0, DDQ1
|
||||
- DQQ (d1d x q1d x q1d) x 3 -> DQQ0, DQQ1, DQQ2
|
||||
- QQQ (q1d x q1d x q1d) x 3 -> QQQ0, QQQ1, QQQ2
|
||||
|
||||
We need the following combinations at the same time
|
||||
(1) DDQ0 + DDQ1 + DQQ0 + DQQ1 + DQQ2
|
||||
(2) DQQ0 + DQQ1 + DQQ2 + QQQ0 + QQQ1 + QQQ2
|
||||
(3) QQQ0 + QQQ1 + QQQ2 + QQD0 + QQD1 + QQD2
|
||||
(4) QQD0 + QQD1 + QQD2 + QDD0 + QDD1 + QDD2
|
||||
|
||||
Allocate largest memory footprint from 2, 3 or 4 and
|
||||
add memory footprint of fields and B/G.
|
||||
|
||||
Annotations with NR and R mean "not reusable" and
|
||||
"reusable", respectively. This means the memory location is
|
||||
reused for _all_ e.g. interpolation of a value etc.
|
||||
|
||||
----
|
||||
For the action of nonlinear diffusion in 2D we have
|
||||
(rho * |u|^2 \nabla u, \nabla v)
|
||||
|
||||
* Load
|
||||
RHO (D x D) | R (after interpolation)
|
||||
U (D x D x VDIM) | R (after interpolation)
|
||||
B (Q x D) | NR
|
||||
G (Q x D) | NR
|
||||
|
||||
* Interpolate Value
|
||||
Temporary (Q x D) | R
|
||||
R (Q x Q) | NR
|
||||
U (Q x Q x VDIM) | NR
|
||||
|
||||
* Interpolate Grad
|
||||
Temporaries (Q x D) + (Q x D) | R
|
||||
U (Q x Q x DIM x VDIM) | NR
|
||||
|
||||
Quadrature point function
|
||||
-> purely thread local
|
||||
|
||||
* Integrate Grad
|
||||
R | temp from Interpolation
|
||||
R | U from Load
|
||||
@@ -0,0 +1,845 @@
|
||||
// This is serac's tuple implementation
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "general/backends.hpp"
|
||||
#include <utility>
|
||||
#include <mfem.hpp>
|
||||
#include <tuple>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @brief This is a class that mimics most of std::tuple's interface,
|
||||
* except that it is usable in CUDA kernels and admits some arithmetic operator overloads.
|
||||
*
|
||||
* see https://en.cppreference.com/w/cpp/utility/tuple for more information about std::tuple
|
||||
*/
|
||||
template <typename... T>
|
||||
struct tuple
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
*/
|
||||
template <typename T0>
|
||||
struct tuple<T0>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1>
|
||||
struct tuple<T0, T1>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2>
|
||||
struct tuple<T0, T1, T2>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3>
|
||||
struct tuple<T0, T1, T2, T3>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
* @tparam T4 The fifth type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4>
|
||||
struct tuple<T0, T1, T2, T3, T4>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
* @tparam T4 The fifth type stored in the tuple
|
||||
* @tparam T5 The sixth type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5>
|
||||
struct tuple<T0, T1, T2, T3, T4, T5>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
T5 v5; ///< The sixth member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
* @tparam T4 The fifth type stored in the tuple
|
||||
* @tparam T5 The sixth type stored in the tuple
|
||||
* @tparam T6 The seventh type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6>
|
||||
struct tuple<T0, T1, T2, T3, T4, T5, T6>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
T5 v5; ///< The sixth member of the tuple
|
||||
T6 v6; ///< The seventh member of the tuple
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type that mimics std::tuple
|
||||
*
|
||||
* @tparam T0 The first type stored in the tuple
|
||||
* @tparam T1 The second type stored in the tuple
|
||||
* @tparam T2 The third type stored in the tuple
|
||||
* @tparam T3 The fourth type stored in the tuple
|
||||
* @tparam T4 The fifth type stored in the tuple
|
||||
* @tparam T5 The sixth type stored in the tuple
|
||||
* @tparam T6 The seventh type stored in the tuple
|
||||
* @tparam T7 The eighth type stored in the tuple
|
||||
*/
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6, typename T7>
|
||||
struct tuple<T0, T1, T2, T3, T4, T5, T6, T7>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
T5 v5; ///< The sixth member of the tuple
|
||||
T6 v6; ///< The seventh member of the tuple
|
||||
T7 v7; ///< The eighth member of the tuple
|
||||
};
|
||||
|
||||
template <typename T0, typename T1, typename T2, typename T3, typename T4, typename T5, typename T6, typename T7, typename T8>
|
||||
struct tuple<T0, T1, T2, T3, T4, T5, T6, T7, T8>
|
||||
{
|
||||
T0 v0; ///< The first member of the tuple
|
||||
T1 v1; ///< The second member of the tuple
|
||||
T2 v2; ///< The third member of the tuple
|
||||
T3 v3; ///< The fourth member of the tuple
|
||||
T4 v4; ///< The fifth member of the tuple
|
||||
T5 v5; ///< The sixth member of the tuple
|
||||
T6 v6; ///< The seventh member of the tuple
|
||||
T7 v7; ///< The eighth member of the tuple
|
||||
T8 v8;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Class template argument deduction rule for tuples
|
||||
* @tparam T The variadic template parameter for tuple types
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE
|
||||
tuple(T...) -> tuple<T...>;
|
||||
|
||||
/**
|
||||
* @brief helper function for combining a list of values into a tuple
|
||||
* @tparam T types of the values to be tuple-d
|
||||
* @param args the actual values to be put into a tuple
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE tuple<T...> make_tuple(const T&... args)
|
||||
{
|
||||
return tuple<T...> {args...};
|
||||
}
|
||||
|
||||
template <class... Types>
|
||||
struct tuple_size
|
||||
{
|
||||
};
|
||||
|
||||
template <class... Types>
|
||||
struct tuple_size<mfem::tuple<Types...>> :
|
||||
std::integral_constant<std::size_t, sizeof...(Types)>
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* @tparam i the tuple index to access
|
||||
* @tparam T the types stored in the tuple
|
||||
* @brief return a reference to the ith tuple entry
|
||||
*/
|
||||
template <int i, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto& get(tuple<T...>& values)
|
||||
{
|
||||
static_assert(i < sizeof...(T), "");
|
||||
if constexpr (i == 0)
|
||||
{
|
||||
return values.v0;
|
||||
}
|
||||
if constexpr (i == 1)
|
||||
{
|
||||
return values.v1;
|
||||
}
|
||||
if constexpr (i == 2)
|
||||
{
|
||||
return values.v2;
|
||||
}
|
||||
if constexpr (i == 3)
|
||||
{
|
||||
return values.v3;
|
||||
}
|
||||
if constexpr (i == 4)
|
||||
{
|
||||
return values.v4;
|
||||
}
|
||||
if constexpr (i == 5)
|
||||
{
|
||||
return values.v5;
|
||||
}
|
||||
if constexpr (i == 6)
|
||||
{
|
||||
return values.v6;
|
||||
}
|
||||
if constexpr (i == 7)
|
||||
{
|
||||
return values.v7;
|
||||
}
|
||||
if constexpr (i == 8)
|
||||
{
|
||||
return values.v8;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam i the tuple index to access
|
||||
* @tparam T the types stored in the tuple
|
||||
* @brief return a copy of the ith tuple entry
|
||||
*/
|
||||
template <int i, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr const auto& get(const tuple<T...>& values)
|
||||
{
|
||||
static_assert(i < sizeof...(T), "");
|
||||
if constexpr (i == 0)
|
||||
{
|
||||
return values.v0;
|
||||
}
|
||||
if constexpr (i == 1)
|
||||
{
|
||||
return values.v1;
|
||||
}
|
||||
if constexpr (i == 2)
|
||||
{
|
||||
return values.v2;
|
||||
}
|
||||
if constexpr (i == 3)
|
||||
{
|
||||
return values.v3;
|
||||
}
|
||||
if constexpr (i == 4)
|
||||
{
|
||||
return values.v4;
|
||||
}
|
||||
if constexpr (i == 5)
|
||||
{
|
||||
return values.v5;
|
||||
}
|
||||
if constexpr (i == 6)
|
||||
{
|
||||
return values.v6;
|
||||
}
|
||||
if constexpr (i == 7)
|
||||
{
|
||||
return values.v7;
|
||||
}
|
||||
if constexpr (i == 8)
|
||||
{
|
||||
return values.v8;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief a function intended to be used for extracting the ith type from a tuple.
|
||||
*
|
||||
* @note type<i>(my_tuple) returns a value, whereas get<i>(my_tuple) returns a reference
|
||||
*
|
||||
* @tparam i the index of the tuple to query
|
||||
* @tparam T the types stored in the tuple
|
||||
* @param values the tuple of values
|
||||
* @return a copy of the ith entry of the input
|
||||
*/
|
||||
template <int i, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto type(const tuple<T...>& values)
|
||||
{
|
||||
static_assert(i < sizeof...(T), "");
|
||||
if constexpr (i == 0)
|
||||
{
|
||||
return values.v0;
|
||||
}
|
||||
if constexpr (i == 1)
|
||||
{
|
||||
return values.v1;
|
||||
}
|
||||
if constexpr (i == 2)
|
||||
{
|
||||
return values.v2;
|
||||
}
|
||||
if constexpr (i == 3)
|
||||
{
|
||||
return values.v3;
|
||||
}
|
||||
if constexpr (i == 4)
|
||||
{
|
||||
return values.v4;
|
||||
}
|
||||
if constexpr (i == 5)
|
||||
{
|
||||
return values.v5;
|
||||
}
|
||||
if constexpr (i == 6)
|
||||
{
|
||||
return values.v6;
|
||||
}
|
||||
if constexpr (i == 7)
|
||||
{
|
||||
return values.v7;
|
||||
}
|
||||
if constexpr (i == 8)
|
||||
{
|
||||
return values.v8;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the + operator of tuples
|
||||
*
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param y tuple of values
|
||||
* @return the returned tuple sum
|
||||
*/
|
||||
template <typename... S, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto plus_helper(const tuple<S...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) + get<i>(y)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief return a tuple of values defined by elementwise sum of x and y
|
||||
*/
|
||||
template <typename... S, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator+(const tuple<S...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
static_assert(sizeof...(S) == sizeof...(T));
|
||||
return plus_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the += operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @tparam i integer sequence used to index the tuples
|
||||
* @param x tuple of values to be incremented
|
||||
* @param y tuple of increment values
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr void plus_equals_helper(tuple<T...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
((get<i>(x) += get<i>(y)), ...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief add values contained in y, to the tuple x
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator+=(tuple<T...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
return plus_equals_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the -= operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @tparam i integer sequence used to index the tuples
|
||||
* @param x tuple of values to be subracted from
|
||||
* @param y tuple of values to subtract from x
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr void minus_equals_helper(tuple<T...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
((get<i>(x) -= get<i>(y)), ...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuples x and y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief add values contained in y, to the tuple x
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator-=(tuple<T...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
return minus_equals_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the - operator of tuples
|
||||
*
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param y tuple of values
|
||||
* @return the returned tuple difference
|
||||
*/
|
||||
template <typename... S, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto minus_helper(const tuple<S...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) - get<i>(y)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief return a tuple of values defined by elementwise difference of x and y
|
||||
*/
|
||||
template <typename... S, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator-(const tuple<S...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
static_assert(sizeof...(S) == sizeof...(T));
|
||||
return minus_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the - operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @return the returned tuple difference
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto unary_minus_helper(const tuple<T...>& x,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{-get<i>(x)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @brief return a tuple of values defined by applying the unary minus operator to each element of x
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator-(const tuple<T...>& x)
|
||||
{
|
||||
return unary_minus_helper(x,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the / operator of tuples
|
||||
*
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param y tuple of values
|
||||
* @return the returned tuple ratio
|
||||
*/
|
||||
template <typename... S, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto div_helper(const tuple<S...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) / get<i>(y)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief return a tuple of values defined by elementwise division of x by y
|
||||
*/
|
||||
template <typename... S, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator/(const tuple<S...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
static_assert(sizeof...(S) == sizeof...(T));
|
||||
return div_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the / operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a the constant numerator
|
||||
* @return the returned tuple ratio
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto div_helper(const double a,
|
||||
const tuple<T...>& x, std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{a / get<i>(x)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the / operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a the constant denomenator
|
||||
* @return the returned tuple ratio
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto div_helper(const tuple<T...>& x,
|
||||
const double a, std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) / a...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple x
|
||||
* @param a the numerator
|
||||
* @param x a tuple of denominator values
|
||||
* @brief return a tuple of values defined by division of a by the elements of x
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator/(const double a, const tuple<T...>& x)
|
||||
{
|
||||
return div_helper(a, x,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of numerator values
|
||||
* @param a a denominator
|
||||
* @brief return a tuple of values defined by elementwise division of x by a
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator/(const tuple<T...>& x, const double a)
|
||||
{
|
||||
return div_helper(x, a,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the * operator of tuples
|
||||
*
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param y tuple of values
|
||||
* @return the returned tuple product
|
||||
*/
|
||||
template <typename... S, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto mult_helper(const tuple<S...>& x,
|
||||
const tuple<T...>& y,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) * get<i>(y)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam S the types stored in the tuple x
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @param x a tuple of values
|
||||
* @param y a tuple of values
|
||||
* @brief return a tuple of values defined by elementwise multiplication of x and y
|
||||
*/
|
||||
template <typename... S, typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator*(const tuple<S...>& x,
|
||||
const tuple<T...>& y)
|
||||
{
|
||||
static_assert(sizeof...(S) == sizeof...(T));
|
||||
return mult_helper(x, y,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(S))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the * operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a a constant multiplier
|
||||
* @return the returned tuple product
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto mult_helper(const double a,
|
||||
const tuple<T...>& x, std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{a * get<i>(x)...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper function for the * operator of tuples
|
||||
*
|
||||
* @tparam T the types stored in the tuple y
|
||||
* @tparam i The integer sequence to i
|
||||
* @param x tuple of values
|
||||
* @param a a constant multiplier
|
||||
* @return the returned tuple product
|
||||
*/
|
||||
template <typename... T, int... i>
|
||||
MFEM_HOST_DEVICE constexpr auto mult_helper(const tuple<T...>& x,
|
||||
const double a, std::integer_sequence<int, i...>)
|
||||
{
|
||||
return tuple{get<i>(x) * a...};
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @param a a scaling factor
|
||||
* @param x the tuple object
|
||||
* @brief multiply each component of x by the value a on the left
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator*(const double a, const tuple<T...>& x)
|
||||
{
|
||||
return mult_helper(a, x,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @param x the tuple object
|
||||
* @param a a scaling factor
|
||||
* @brief multiply each component of x by the value a on the right
|
||||
*/
|
||||
template <typename... T>
|
||||
MFEM_HOST_DEVICE constexpr auto operator*(const tuple<T...>& x, const double a)
|
||||
{
|
||||
return mult_helper(x, a,
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @tparam i a list of indices used to acces each element of the tuple
|
||||
* @param out the ostream to write the output to
|
||||
* @param A the tuple of values
|
||||
* @brief helper used to implement printing a tuple of values
|
||||
*/
|
||||
template <typename... T, std::size_t... i>
|
||||
auto& print_helper(std::ostream& out, const mfem::tuple<T...>& A,
|
||||
std::integer_sequence<size_t, i...>)
|
||||
{
|
||||
out << "tuple{";
|
||||
(..., (out << (i == 0 ? "" : ", ") << mfem::get<i>(A)));
|
||||
out << "}";
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam T the types stored in the tuple
|
||||
* @param out the ostream to write the output to
|
||||
* @param A the tuple of values
|
||||
* @brief print a tuple of values
|
||||
*/
|
||||
template <typename... T>
|
||||
auto& operator<<(std::ostream& out, const mfem::tuple<T...>& A)
|
||||
{
|
||||
return print_helper(out, A, std::make_integer_sequence<size_t, sizeof...(T)>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief A helper to apply a lambda to a tuple
|
||||
*
|
||||
* @tparam lambda The functor type
|
||||
* @tparam T The tuple types
|
||||
* @tparam i The integer sequence to i
|
||||
* @param f The functor to apply to the tuple
|
||||
* @param args The input tuple
|
||||
* @return The functor output
|
||||
*/
|
||||
template <typename lambda, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE auto apply_helper(lambda f, tuple<T...>& args,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return f(get<i>(args)...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam lambda a callable type
|
||||
* @tparam T the types of arguments to be passed in to f
|
||||
* @param f the callable object
|
||||
* @param args a tuple of arguments
|
||||
* @brief a way of passing an n-tuple to a function that expects n separate arguments
|
||||
*
|
||||
* e.g. foo(bar, baz) is equivalent to apply(foo, mfem::tuple(bar,baz));
|
||||
*/
|
||||
template <typename lambda, typename... T>
|
||||
MFEM_HOST_DEVICE auto apply(lambda f, tuple<T...>& args)
|
||||
{
|
||||
return apply_helper(f, std::move(args),
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @overload
|
||||
*/
|
||||
template <typename lambda, typename... T, int... i>
|
||||
MFEM_HOST_DEVICE auto apply_helper(lambda f, const tuple<T...>& args,
|
||||
std::integer_sequence<int, i...>)
|
||||
{
|
||||
return f(get<i>(args)...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @tparam lambda a callable type
|
||||
* @tparam T the types of arguments to be passed in to f
|
||||
* @param f the callable object
|
||||
* @param args a tuple of arguments
|
||||
* @brief a way of passing an n-tuple to a function that expects n separate arguments
|
||||
*
|
||||
* e.g. foo(bar, baz) is equivalent to apply(foo, mfem::tuple(bar,baz));
|
||||
*/
|
||||
template <typename lambda, typename... T>
|
||||
MFEM_HOST_DEVICE auto apply(lambda f, const tuple<T...>& args)
|
||||
{
|
||||
return apply_helper(f, std::move(args),
|
||||
std::make_integer_sequence<int, static_cast<int>(sizeof...(T))>());
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief a struct used to determine the type at index I of a tuple
|
||||
*
|
||||
* @note see: https://en.cppreference.com/w/cpp/utility/tuple/tuple_element
|
||||
*
|
||||
* @tparam I the index of the desired type
|
||||
* @tparam T a tuple of different types
|
||||
*/
|
||||
template <size_t I, class T>
|
||||
struct tuple_element;
|
||||
|
||||
// recursive case
|
||||
/// @overload
|
||||
template <size_t I, class Head, class... Tail>
|
||||
struct tuple_element<I, tuple<Head, Tail...>> : tuple_element<I - 1,
|
||||
tuple<Tail...>>
|
||||
{
|
||||
};
|
||||
|
||||
// base case
|
||||
/// @overload
|
||||
template <class Head, class... Tail>
|
||||
struct tuple_element<0, tuple<Head, Tail...>>
|
||||
{
|
||||
using type = Head; ///< the type at the specified index
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Trait for checking if a type is a @p mfem::tuple
|
||||
*/
|
||||
template <typename T>
|
||||
struct is_tuple : std::false_type
|
||||
{
|
||||
};
|
||||
|
||||
/// @overload
|
||||
template <typename... T>
|
||||
struct is_tuple<mfem::tuple<T...>> : std::true_type
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Trait for checking if a type if a @p mfem::tuple containing only @p mfem::tuple
|
||||
*/
|
||||
template <typename T>
|
||||
struct is_tuple_of_tuples : std::false_type
|
||||
{
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Trait for checking if a type if a @p mfem::tuple containing only @p mfem::tuple
|
||||
*/
|
||||
template <typename... T>
|
||||
struct is_tuple_of_tuples<mfem::tuple<T...>>
|
||||
{
|
||||
static constexpr bool value = (is_tuple<T>::value &&
|
||||
...); ///< true/false result of type check
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
@@ -0,0 +1,123 @@
|
||||
#include "dfem/dfem_refactor.hpp"
|
||||
#include "fem/bilininteg.hpp"
|
||||
#include "fem/coefficient.hpp"
|
||||
#include "linalg/auxiliary.hpp"
|
||||
#include "linalg/hypre.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
Mpi::Init();
|
||||
int num_procs = Mpi::WorldSize();
|
||||
int myid = Mpi::WorldRank();
|
||||
Hypre::Init();
|
||||
|
||||
const char *mesh_file = "../data/ref-square.mesh";
|
||||
int polynomial_order = 1;
|
||||
int ir_order = 2;
|
||||
int refinements = 1;
|
||||
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
L2_FECollection fec(polynomial_order, dim, BasisType::GaussLobatto);
|
||||
ParFiniteElementSpace fes(&mesh, &fec);
|
||||
|
||||
const IntegrationRule &ir = IntRules.Get(fes.GetFE(0)->GetGeomType(),
|
||||
ir_order * fec.GetOrder());
|
||||
const IntegrationRule &ir_face = IntRules.Get(
|
||||
fes.GetTraceElement(0, fes.GetMesh()->GetFaceGeometry(0))->GetGeomType(),
|
||||
ir_order * fec.GetOrder());
|
||||
ParGridFunction u(&fes);
|
||||
|
||||
// // -\nabla \cdot (\nabla u + p * I) -> (\nabla u + p * I, \nabla v)
|
||||
// auto advection_kernel = [](const tensor<double, 2> &dudxi,
|
||||
// const tensor<double, 2, 2> &J,
|
||||
// const double &w)
|
||||
// {
|
||||
// constexpr tensor<double, 2> b{1.0, 1.0};
|
||||
// return std::tuple{dot(b, dudxi * inv(J)) * det(J) * w};
|
||||
// };
|
||||
|
||||
// std::tuple argument_operators_0{Gradient{"quantity"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
|
||||
// std::tuple output_operator_0{Value{"quantity"}};
|
||||
// ElementOperator op_0{advection_kernel, argument_operators_0, output_operator_0};
|
||||
|
||||
// std::array solutions{FieldDescriptor{&fes, "quantity"}};
|
||||
// std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
// DifferentiableOperator advection_op{solutions, parameters, std::tuple{op_0}, mesh, ir};
|
||||
|
||||
// auto adv_du = advection_op.template GetDerivativeWrt<0>({&u}, {mesh_nodes});
|
||||
// HypreParMatrix A;
|
||||
// adv_du->Assemble(A);
|
||||
|
||||
// std::ofstream mmatofs("dfem_mat.dat");
|
||||
// A.PrintMatlab(mmatofs);
|
||||
// mmatofs.close();
|
||||
|
||||
auto trace_kernel = [](const double &uL, const double &uR, const double &J,
|
||||
const double &w)
|
||||
{
|
||||
return std::tuple{1.0 / J * w};
|
||||
};
|
||||
|
||||
std::tuple argument_operators_0
|
||||
{
|
||||
FaceValueLeft{"quantity"},
|
||||
FaceValueRight{"quantity"},
|
||||
Gradient{"coordinates"},
|
||||
Weight{"integration_weights"}
|
||||
};
|
||||
std::tuple output_operator_0{Value{"quantity"}};
|
||||
FaceElementOperator op_0{trace_kernel, argument_operators_0, output_operator_0};
|
||||
|
||||
std::array solutions{FieldDescriptor{&fes, "quantity"}};
|
||||
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator trace_op{solutions, parameters, std::tuple{op_0}, mesh, ir_face};
|
||||
|
||||
auto vector_func = [](const Vector &, Vector &u)
|
||||
{
|
||||
u = 1.0;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient vel_coeff(dim, vector_func);
|
||||
|
||||
ParBilinearForm adv_form(&fes);
|
||||
constexpr double alpha = 1.0;
|
||||
auto integ = new ConvectionIntegrator(vel_coeff, alpha);
|
||||
integ->SetIntRule(&ir);
|
||||
adv_form.AddInteriorFaceIntegrator(
|
||||
new NonconservativeDGTraceIntegrator(vel_coeff, alpha));
|
||||
// adv_form.AddDomainIntegrator(integ);
|
||||
adv_form.Assemble();
|
||||
adv_form.Finalize();
|
||||
|
||||
auto K = adv_form.ParallelAssemble();
|
||||
std::ofstream kmatofs("mfem_mat.dat");
|
||||
K->PrintMatlab(kmatofs);
|
||||
kmatofs.close();
|
||||
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << u << std::flush;
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
#include "dfem.hpp"
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
Mpi::Init();
|
||||
|
||||
std::cout << std::setprecision(9);
|
||||
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
int polynomial_order = 1;
|
||||
int refinements = 0;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
|
||||
args.AddOption(&polynomial_order, "-o", "--order", "");
|
||||
args.AddOption(&refinements, "-r", "--r", "");
|
||||
args.ParseCheck();
|
||||
|
||||
Mesh mesh_serial(mesh_file, 1, 1);
|
||||
mesh_serial.SetCurvature(1);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
const int dim = mesh_serial.Dimension();
|
||||
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
mesh_serial.Clear();
|
||||
|
||||
constexpr int vdim = 2;
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr(mesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
std::cout << "nqpts = " << ir.GetNPoints() << std::endl;
|
||||
std::cout << "ndofs = " << h1fes.GlobalTrueVSize() << std::endl;
|
||||
|
||||
ParGridFunction u(&h1fes);
|
||||
|
||||
auto exact_solution = [](const Vector &coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
u(0) = x*x + y;
|
||||
u(1) = x + 0.5*y*y;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient exact_solution_coeff(dim, exact_solution);
|
||||
|
||||
auto elasticity_kernel = [](tensor<double, 2, 2> &dudxi,
|
||||
tensor<double, 2, 2> &J,
|
||||
double &w)
|
||||
{
|
||||
using mfem::internal::tensor;
|
||||
using mfem::internal::IsotropicIdentity;
|
||||
|
||||
double lambda, mu;
|
||||
{
|
||||
lambda = 1.0;
|
||||
mu = 1.0;
|
||||
}
|
||||
static constexpr auto I = IsotropicIdentity<2>();
|
||||
auto eps = sym(dudxi * inv(J));
|
||||
auto JxW = transpose(inv(J)) * det(J) * w;
|
||||
auto r = (lambda * tr(eps) * I + 2.0 * mu * eps) * JxW;
|
||||
return r;
|
||||
};
|
||||
|
||||
tensor<double, 2, 2> dudxi, s_dudxi, J;
|
||||
double w = 1.0;
|
||||
|
||||
enzyme::get<0>
|
||||
(enzyme::autodiff<enzyme::Forward,
|
||||
enzyme::DuplicatedNoNeed<tensor<double, 2, 2>>>
|
||||
(+elasticity_kernel,
|
||||
enzyme::Duplicated<tensor<double, 2, 2> *>(&dudxi, &s_dudxi),
|
||||
enzyme::Const<tensor<double, 2, 2>*>(&J),
|
||||
enzyme::Const<double*>(&w)));
|
||||
|
||||
// std::tuple input_descriptors = {Gradient{"displacement"}, Gradient{"coordinates"}, Weight{"integration_weight"}};
|
||||
// std::tuple output_descriptors = {Gradient{"displacement"}};
|
||||
// ElementOperator qf {elasticity_kernel, input_descriptors, output_descriptors};
|
||||
|
||||
// ElementOperator forcing_qf
|
||||
// {
|
||||
// [](tensor<double, 2> x, tensor<double, 2, 2> J, double w)
|
||||
// {
|
||||
// double lambda, mu;
|
||||
// {
|
||||
// lambda = 1.0;
|
||||
// mu = 1.0;
|
||||
// }
|
||||
// auto f = x;
|
||||
// f(0) = 4.0*mu + 2.0*lambda;
|
||||
// f(1) = 2.0*mu + lambda;
|
||||
// return f * det(J) * w;
|
||||
// },
|
||||
// // inputs
|
||||
// std::tuple{
|
||||
// Value{"coordinates"},
|
||||
// Gradient{"coordinates"},
|
||||
// Weight{"integration_weight"}},
|
||||
// // outputs
|
||||
// std::tuple{
|
||||
// Value{"displacement"}}
|
||||
// };
|
||||
|
||||
// std::vector<Field> solutions{{&u, "displacement"}};
|
||||
// std::vector<Field> parameters{{mesh.GetNodes(), "coordinates"}};
|
||||
// std::vector<Field> dependent_fields{{&u, "displacement"}};
|
||||
// DifferentiableForm dop(solutions, parameters, dependent_fields, mesh);
|
||||
|
||||
// dop.AddElementOperator<AD::Enzyme>(qf, ir);
|
||||
// dop.AddElementOperator<AD::None>(forcing_qf, ir);
|
||||
// dop.SetEssentialTrueDofs(ess_tdof_list);
|
||||
|
||||
// GMRESSolver gmres(MPI_COMM_WORLD);
|
||||
// gmres.SetRelTol(1e-12);
|
||||
// gmres.SetMaxIter(5000);
|
||||
// gmres.SetPrintLevel(IterativeSolver::PrintLevel().Summary());
|
||||
|
||||
// NewtonSolver newton(MPI_COMM_WORLD);
|
||||
// newton.SetSolver(gmres);
|
||||
// newton.SetOperator(dop);
|
||||
// newton.SetRelTol(1e-12);
|
||||
// newton.SetMaxIter(100);
|
||||
// newton.SetPrintLevel(1);
|
||||
|
||||
// u = 1e-6;
|
||||
// u.ProjectBdrCoefficient(exact_solution_coeff, ess_bdr);
|
||||
// Vector x;
|
||||
// u.GetTrueDofs(x);
|
||||
|
||||
// Vector zero;
|
||||
// newton.Mult(zero, x);
|
||||
|
||||
// u.Distribute(x);
|
||||
|
||||
// std::cout << "|u-u_ex|_L2 = " << u.ComputeL2Error(exact_solution_coeff) << "\n";
|
||||
|
||||
return 0;
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,115 @@
|
||||
#include <tuple>
|
||||
#include <type_traits>
|
||||
#include <iostream>
|
||||
#include <enzyme/enzyme>
|
||||
|
||||
template <typename T>
|
||||
constexpr auto get_type_name() -> std::string_view
|
||||
{
|
||||
#if defined(__clang__)
|
||||
constexpr auto prefix = std::string_view {"[T = "};
|
||||
constexpr auto suffix = "]";
|
||||
constexpr auto function = std::string_view{__PRETTY_FUNCTION__};
|
||||
#elif defined(__GNUC__)
|
||||
constexpr auto prefix = std::string_view {"with T = "};
|
||||
constexpr auto suffix = "; ";
|
||||
constexpr auto function = std::string_view{__PRETTY_FUNCTION__};
|
||||
#elif defined(_MSC_VER)
|
||||
constexpr auto prefix = std::string_view {"get_type_name<"};
|
||||
constexpr auto suffix = ">(void)";
|
||||
constexpr auto function = std::string_view{__FUNCSIG__};
|
||||
#else
|
||||
#error Unsupported compiler
|
||||
#endif
|
||||
|
||||
const auto start = function.find(prefix) + prefix.size();
|
||||
const auto end = function.find(suffix);
|
||||
const auto size = end - start;
|
||||
|
||||
return function.substr(start, size);
|
||||
}
|
||||
|
||||
template <typename ... Ts>
|
||||
constexpr auto decay_types(std::tuple<Ts...> const &)
|
||||
-> std::tuple<std::remove_cv_t<std::remove_reference_t<Ts>>...>;
|
||||
|
||||
template <typename T>
|
||||
using decay_tuple = decltype(decay_types(std::declval<T>()));
|
||||
|
||||
template <class F> struct FunctionSignature;
|
||||
|
||||
template <typename output_t, typename... input_ts>
|
||||
struct FunctionSignature<output_t(input_ts...)>
|
||||
{
|
||||
using return_t = output_t;
|
||||
using parameter_ts = std::tuple<input_ts...>;
|
||||
};
|
||||
|
||||
template <class T> struct create_function_signature;
|
||||
|
||||
template <typename output_t, typename T, typename... input_ts>
|
||||
struct create_function_signature<output_t (T::*)(input_ts...) const>
|
||||
{
|
||||
using type = FunctionSignature<output_t(input_ts...)>;
|
||||
};
|
||||
|
||||
template <typename arg_ts, std::size_t... Is>
|
||||
auto create_enzyme_args(arg_ts &args,
|
||||
arg_ts &shadow_args,
|
||||
std::index_sequence<Is...>)
|
||||
{
|
||||
((std::cout << std::get<Is>(shadow_args) << "\n"), ...);
|
||||
return std::tuple<enzyme::Duplicated<decltype(std::get<Is>(args))>...>
|
||||
{
|
||||
{ std::get<Is>(args), std::get<Is>(shadow_args) }...
|
||||
};
|
||||
}
|
||||
|
||||
template <typename kernel_t, typename arg_ts>
|
||||
auto fwddiff_apply_enzyme(kernel_t kernel, arg_ts &&args, arg_ts &&shadow_args)
|
||||
{
|
||||
auto arg_indices =
|
||||
std::make_index_sequence<std::tuple_size_v<std::remove_reference_t<arg_ts>>> {};
|
||||
|
||||
auto enzyme_args = create_enzyme_args(args, shadow_args, arg_indices);
|
||||
|
||||
using kf_return_t = typename create_function_signature<
|
||||
decltype(&kernel_t::operator())>::type::return_t;
|
||||
|
||||
std::cout << "args is " << get_type_name<decltype(args)>() << "\n\n";
|
||||
std::cout << "enzyme_args type is " << get_type_name<decltype(enzyme_args)>() <<
|
||||
"\n\n";
|
||||
std::cout << "return type is " << get_type_name<decltype(kf_return_t{})>() <<
|
||||
"\n\n";
|
||||
|
||||
return std::apply([&](auto &&...args)
|
||||
{
|
||||
return enzyme::get<0>(
|
||||
enzyme::autodiff<enzyme::Forward>
|
||||
(+kernel, args...));
|
||||
},
|
||||
enzyme_args);
|
||||
}
|
||||
|
||||
int main()
|
||||
{
|
||||
|
||||
auto func = [](const double &x)
|
||||
{
|
||||
return x*x;
|
||||
};
|
||||
|
||||
using kf_param_ts = typename create_function_signature<
|
||||
decltype(&decltype(func)::operator())>::type::parameter_ts;
|
||||
using kf_output_t = typename create_function_signature<
|
||||
decltype(&decltype(func)::operator())>::type::return_t;
|
||||
auto kernel_args = decay_tuple<kf_param_ts> {};
|
||||
auto kernel_shadow_args = decay_tuple<kf_param_ts> {};
|
||||
|
||||
std::get<0>(kernel_args) = 3;
|
||||
std::get<0>(kernel_shadow_args) = 1;
|
||||
const auto res = fwddiff_apply_enzyme(func, kernel_args, kernel_shadow_args);
|
||||
std::cout << res << " == 6\n";
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,114 @@
|
||||
#include "dfem.hpp"
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
Mpi::Init();
|
||||
|
||||
std::cout << std::setprecision(9);
|
||||
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
int polynomial_order = 1;
|
||||
int refinements = 0;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
|
||||
args.AddOption(&polynomial_order, "-o", "--order", "");
|
||||
args.AddOption(&refinements, "-r", "--r", "");
|
||||
args.ParseCheck();
|
||||
|
||||
Mesh mesh_serial(mesh_file, 1, 1);
|
||||
mesh_serial.SetCurvature(1);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
const int dim = mesh_serial.Dimension();
|
||||
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
mesh_serial.Clear();
|
||||
|
||||
constexpr int vdim = 2;
|
||||
|
||||
// test_partial_assembly_setup_qf(mesh, 1, polynomial_order);
|
||||
// exit(0);
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr(mesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
std::cout << "nqpts = " << ir.GetNPoints() << std::endl;
|
||||
std::cout << "ndofs = " << h1fes.GlobalTrueVSize() << std::endl;
|
||||
|
||||
ParGridFunction u(&h1fes);
|
||||
ParGridFunction g(&h1fes);
|
||||
ParGridFunction rho(&h1fes);
|
||||
|
||||
auto exact_solution = [](const Vector &coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
u(0) = x*x + y;
|
||||
u(1) = x + 0.5*y*y;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient exact_solution_coeff(dim, exact_solution);
|
||||
|
||||
auto objective = [](tensor<double, 2> u, double rho,
|
||||
tensor<double, 2, 2> J,
|
||||
double w)
|
||||
{
|
||||
return sqnorm(u) * det(J) * w;
|
||||
};
|
||||
|
||||
std::tuple inputs{Value{"displacement"}, Value{"density"}, Gradient{"coordinates"}, Weight{"integration_weight"}};
|
||||
std::tuple outputs{ One{"integral"} };
|
||||
ElementOperator objective_eop { objective, inputs, outputs };
|
||||
|
||||
std::vector<Field> solution_fields{{&u, "displacement"}};
|
||||
std::vector<Field> parameter_fields{{mesh.GetNodes(), "coordinates"}, {&rho, "density"}};
|
||||
std::vector<Field> dependent_variables{{&u, "displacement"}};
|
||||
DifferentiableForm dop(solution_fields, parameter_fields, dependent_variables,
|
||||
mesh);
|
||||
|
||||
dop.AddElementOperator(objective_eop, ir);
|
||||
|
||||
u.ProjectCoefficient(exact_solution_coeff);
|
||||
Vector zero;
|
||||
|
||||
Vector y(1);
|
||||
Vector utdof;
|
||||
u.GetTrueDofs(utdof);
|
||||
dop.Mult(utdof, y);
|
||||
|
||||
// finite difference test
|
||||
Vector dgdu(u.Size());
|
||||
Vector fx(y);
|
||||
out << "g: ";
|
||||
print_vector(fx);
|
||||
out << "\n";
|
||||
|
||||
for (int i = 0; i < u.Size(); i++)
|
||||
{
|
||||
double h = 1e-6;
|
||||
u(i) += h;
|
||||
dop.Mult(u, y);
|
||||
u(i) -= h;
|
||||
y -= fx;
|
||||
y /= h;
|
||||
dgdu(i) = y(0);
|
||||
}
|
||||
|
||||
out << "dgdu: ";
|
||||
print_vector(dgdu);
|
||||
|
||||
// Vector dgdu = dop.GetGradientWrt({&u, "displacement"});
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,138 @@
|
||||
#include "dfem.hpp"
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
Mpi::Init();
|
||||
|
||||
std::cout << std::setprecision(9);
|
||||
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
int polynomial_order = 1;
|
||||
int refinements = 0;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
|
||||
args.AddOption(&polynomial_order, "-o", "--order", "");
|
||||
args.AddOption(&refinements, "-r", "--r", "");
|
||||
args.ParseCheck();
|
||||
|
||||
Mesh mesh_serial(mesh_file, 1, 1);
|
||||
mesh_serial.SetCurvature(1);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
const int dim = mesh_serial.Dimension();
|
||||
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
mesh_serial.Clear();
|
||||
|
||||
constexpr int vdim = 1;
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr(mesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
std::cout << "nqpts = " << ir.GetNPoints() << std::endl;
|
||||
std::cout << "ndofs = " << h1fes.GlobalTrueVSize() << std::endl;
|
||||
|
||||
ParGridFunction u(&h1fes);
|
||||
|
||||
auto exact_solution = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
// PRESENT
|
||||
return pow(x,2) + 0.5*x*pow(y,2);
|
||||
};
|
||||
|
||||
FunctionCoefficient exact_solution_coeff(exact_solution);
|
||||
|
||||
auto plaplacian = [](double u,
|
||||
tensor<double, 2> dudxi,
|
||||
tensor<double, 2, 2> J,
|
||||
double w)
|
||||
{
|
||||
using mfem::internal::tensor;
|
||||
auto dudx = dudxi * inv(J);
|
||||
auto JxW = transpose(inv(J)) * det(J) * w;
|
||||
// PRESENT: Implement (1+u^2) * ∇u
|
||||
return (1.0 + u*u) * dudx * JxW;
|
||||
};
|
||||
|
||||
// PRESENT: Implement descriptors
|
||||
std::tuple input_descriptors = {Value{"potential"}, Gradient{"potential"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
|
||||
// PRESENT: Implement descriptors
|
||||
std::tuple output_descriptors = {Gradient{"potential"}};
|
||||
|
||||
ElementOperator qf {plaplacian, input_descriptors, output_descriptors};
|
||||
|
||||
ElementOperator forcing_qf
|
||||
{
|
||||
[](tensor<double, 2> coords, tensor<double, 2, 2> J, double w)
|
||||
{
|
||||
int p = 2;
|
||||
double x = coords(0);
|
||||
double y = coords(1);
|
||||
// *INDENT-OFF*
|
||||
double mathematica_please_help_me = 2.*pow(x,2)*pow(y,2)*(pow(x,2) + 0.5*x*pow(y,2)) + 2*pow(2*x + 0.5*pow(y,2),2)*(pow(x,2) + 0.5*x*pow(y,2)) + 2*(1 + pow(pow(x,2) + 0.5*x*pow(y,2),2)) + 1.*x*(1 + pow(pow(x,2) + 0.5*x*pow(y,2),2));
|
||||
return mathematica_please_help_me * det(J) * w;
|
||||
// *INDENT-ON*
|
||||
},
|
||||
// inputs
|
||||
std::tuple{
|
||||
Value{"coordinates"},
|
||||
Gradient{"coordinates"},
|
||||
Weight{"integration_weight"}},
|
||||
// outputs
|
||||
std::tuple{
|
||||
Value{"potential"}}
|
||||
};
|
||||
|
||||
std::tuple list_of_qfs{qf_1, qf_2, qf_n};
|
||||
|
||||
std::vector<Field> solutions{{&u, "potential"}};
|
||||
std::vector<Field> parameters{{mesh.GetNodes(), "coordinates"}};
|
||||
DifferentiableForm dop(solutions, parameters, mesh);
|
||||
dop.SetEssentialTrueDofs(ess_tdof_list);
|
||||
|
||||
auto R = dop.GetResidual(list_of_qfs, ir);
|
||||
auto Jacobian_aka_dRdu = dop.GetDerivative<0>(list_of_qfs, ir);
|
||||
|
||||
// R(u) = (\grad u, \grad v) + (f, v)
|
||||
// dop.AddElementOperator<AD::Enzyme>(qf, ir);
|
||||
// dop.AddElementOperator<AD::None>(forcing_qf, ir);
|
||||
|
||||
GMRESSolver gmres(MPI_COMM_WORLD);
|
||||
gmres.SetRelTol(1e-12);
|
||||
gmres.SetMaxIter(5000);
|
||||
gmres.SetPrintLevel(IterativeSolver::PrintLevel().Summary());
|
||||
|
||||
NewtonSolver newton(MPI_COMM_WORLD);
|
||||
newton.SetSolver(gmres);
|
||||
newton.SetOperator(dop);
|
||||
newton.SetRelTol(1e-12);
|
||||
newton.SetMaxIter(100);
|
||||
newton.SetPrintLevel(1);
|
||||
|
||||
u = 1e-6;
|
||||
u.ProjectBdrCoefficient(exact_solution_coeff, ess_bdr);
|
||||
Vector x;
|
||||
u.GetTrueDofs(x);
|
||||
|
||||
Vector zero;
|
||||
newton.Mult(zero, x);
|
||||
|
||||
u.Distribute(x);
|
||||
|
||||
std::cout << "|u-u_ex|_L2 = " << u.ComputeL2Error(exact_solution_coeff) << "\n";
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,192 @@
|
||||
#include "dfem/dfem_refactor.hpp"
|
||||
#include "linalg/hypre.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
template <typename diffusion_t, typename force_t>
|
||||
class DiffusionOperator : public Operator
|
||||
{
|
||||
template <typename diffusion_du_t>
|
||||
class DiffusionJacobianOperator : public Operator
|
||||
{
|
||||
public:
|
||||
DiffusionJacobianOperator(const DiffusionOperator *diffusion,
|
||||
std::shared_ptr<diffusion_du_t> diff_du) :
|
||||
Operator(diffusion->Height()), s(diffusion)
|
||||
{
|
||||
diff_du->Assemble(A);
|
||||
A.EliminateBC(s->ess_tdofs, Operator::DiagonalPolicy::DIAG_ONE);
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &y) const override
|
||||
{
|
||||
A.Mult(x, y);
|
||||
}
|
||||
|
||||
const DiffusionOperator *s;
|
||||
HypreParMatrix A;
|
||||
};
|
||||
|
||||
public:
|
||||
DiffusionOperator(diffusion_t &diffusion, force_t &force,
|
||||
Array<int> &ess_tdofs) :
|
||||
Operator(diffusion.Height()), diffusion(diffusion),
|
||||
force(force), ess_tdofs(ess_tdofs), f(force.Height()) {}
|
||||
|
||||
void SetParameters(ParGridFunction &mesh_nodes)
|
||||
{
|
||||
diffusion.SetParameters({&mesh_nodes});
|
||||
force.SetParameters({&mesh_nodes});
|
||||
|
||||
Vector zero;
|
||||
|
||||
this->mesh_nodes.SetSpace(mesh_nodes.ParFESpace());
|
||||
this->mesh_nodes = mesh_nodes;
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &r) const override
|
||||
{
|
||||
diffusion.Mult(x, r);
|
||||
force.Mult(x, f);
|
||||
r -= f;
|
||||
r.SetSubVector(ess_tdofs, 0.0);
|
||||
}
|
||||
|
||||
Operator &GetGradient(const Vector &x) const override
|
||||
{
|
||||
ParGridFunction u(const_cast<ParFiniteElementSpace *>
|
||||
(*std::get_if<const ParFiniteElementSpace *>
|
||||
(&diffusion.solutions[0].data)));
|
||||
u.SetFromTrueDofs(x);
|
||||
|
||||
auto dfdu = diffusion.template GetDerivativeWrt<0>({&u}, {&mesh_nodes});
|
||||
dfdu->Assemble(A);
|
||||
A.EliminateBC(ess_tdofs, DiagonalPolicy::DIAG_ONE);
|
||||
return A;
|
||||
// delete jacobian_operator;
|
||||
// jacobian_operator = new
|
||||
// DiffusionJacobianOperator<typename std::remove_pointer<decltype(dfdu.get())>::type>
|
||||
// (this, dfdu);
|
||||
// return *jacobian_operator;
|
||||
}
|
||||
|
||||
diffusion_t &diffusion;
|
||||
force_t &force;
|
||||
|
||||
const Array<int> ess_tdofs;
|
||||
mutable Vector f;
|
||||
|
||||
mutable ParGridFunction mesh_nodes;
|
||||
|
||||
mutable Operator *jacobian_operator = nullptr;
|
||||
mutable HypreParMatrix A;
|
||||
};
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
Mpi::Init();
|
||||
int num_procs = Mpi::WorldSize();
|
||||
int myid = Mpi::WorldRank();
|
||||
Hypre::Init();
|
||||
|
||||
const char *mesh_file = "../data/ref-square.mesh";
|
||||
int polynomial_order = 2;
|
||||
int ir_order = 2;
|
||||
int refinements = 4;
|
||||
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection potential_fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace potential_fes(&mesh, &potential_fec);
|
||||
|
||||
const IntegrationRule &potential_ir =
|
||||
IntRules.Get(potential_fes.GetFE(0)->GetGeomType(),
|
||||
ir_order * potential_fec.GetOrder());
|
||||
|
||||
Array<int> bdr_attr_is_ess(mesh.bdr_attributes.Max());
|
||||
bdr_attr_is_ess = 1;
|
||||
Array<int> ess_tdofs;
|
||||
potential_fes.GetEssentialTrueDofs(bdr_attr_is_ess, ess_tdofs);
|
||||
|
||||
ParGridFunction u(&potential_fes);
|
||||
u = 0.0;
|
||||
|
||||
auto diffusion_kernel = [](const internal::dual<double, double> &u,
|
||||
const tensor<internal::dual<double, double>, 2> &dudxi,
|
||||
const tensor<double, 2, 2> &J,
|
||||
const double &w)
|
||||
{
|
||||
auto invJ = inv(J);
|
||||
auto dudx = dudxi * invJ;
|
||||
return std::tuple{(1.0 + u * u) * dudx * det(J) * w * transpose(invJ)};
|
||||
};
|
||||
|
||||
std::tuple argument_operators_0{Value{"potential"}, Gradient{"potential"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
|
||||
std::tuple output_operator_0{Gradient{"potential"}};
|
||||
ElementOperator op_0{diffusion_kernel, argument_operators_0, output_operator_0};
|
||||
|
||||
auto force_kernel = [](const tensor<double, 2, 2> &J,
|
||||
const double &w)
|
||||
{
|
||||
return std::tuple{1.0 * det(J) * w};
|
||||
};
|
||||
std::tuple argument_operators_1{Gradient{"coordinates"}, Weight{"integration_weights"}};
|
||||
std::tuple output_operator_1{Value{"potential"}};
|
||||
ElementOperator op_1{force_kernel, argument_operators_1, output_operator_1};
|
||||
|
||||
std::array solutions{FieldDescriptor{&potential_fes, "potential"}};
|
||||
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator diffusion_op{solutions, parameters, std::tuple{op_0}, mesh, potential_ir};
|
||||
DifferentiableOperator force_op{solutions, parameters, std::tuple{op_1}, mesh, potential_ir};
|
||||
|
||||
DiffusionOperator diffusion(diffusion_op, force_op, ess_tdofs);
|
||||
|
||||
diffusion.SetParameters({*mesh_nodes});
|
||||
|
||||
HypreBoomerAMG amg;
|
||||
amg.SetPrintLevel(0);
|
||||
|
||||
CGSolver solver(MPI_COMM_WORLD);
|
||||
solver.SetAbsTol(1e-12);
|
||||
solver.SetRelTol(1e-12);
|
||||
solver.SetMaxIter(500);
|
||||
solver.SetPrintLevel(2);
|
||||
solver.SetPreconditioner(amg);
|
||||
|
||||
NewtonSolver newton(MPI_COMM_WORLD);
|
||||
newton.SetOperator(diffusion);
|
||||
newton.SetSolver(solver);
|
||||
newton.SetRelTol(1e-8);
|
||||
newton.SetMaxIter(10);
|
||||
newton.SetPrintLevel(1);
|
||||
|
||||
Vector zero;
|
||||
Vector x(potential_fes.GetTrueVSize());
|
||||
u.ParallelProject(x);
|
||||
newton.Mult(zero, x);
|
||||
|
||||
u.SetFromTrueDofs(x);
|
||||
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << u << std::flush;
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,102 @@
|
||||
#include "mfem.hpp"
|
||||
#include "dfem/dfem_refactor.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
|
||||
auto main(int argc, char *argv[]) -> int
|
||||
{
|
||||
Mpi::Init();
|
||||
|
||||
std::cout << std::setprecision(9);
|
||||
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
int polynomial_order = 1;
|
||||
int refinements = 0;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
|
||||
args.AddOption(&polynomial_order, "-o", "--order", "");
|
||||
args.AddOption(&refinements, "-r", "--r", "");
|
||||
args.ParseCheck();
|
||||
|
||||
Mesh mesh_serial(mesh_file, 1, 1);
|
||||
mesh_serial.SetCurvature(1);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
const int dim = mesh_serial.Dimension();
|
||||
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
constexpr int vdim = 1;
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
Array<int> ess_tdof_list;
|
||||
Array<int> ess_bdr(mesh.bdr_attributes.Max());
|
||||
ess_bdr = 1;
|
||||
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
std::cout << "nqpts = " << ir.GetNPoints() << std::endl;
|
||||
std::cout << "ndofs = " << h1fes.GlobalTrueVSize() << std::endl;
|
||||
|
||||
ParGridFunction u(&h1fes);
|
||||
|
||||
auto exact_solution = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
return 2.345 + x + y;
|
||||
};
|
||||
|
||||
FunctionCoefficient exact_solution_coeff(exact_solution);
|
||||
|
||||
u.ProjectCoefficient(exact_solution_coeff);
|
||||
|
||||
auto domain_qf = [](const double &u,
|
||||
const tensor<double, 2, 2> &J,
|
||||
const double &w)
|
||||
{
|
||||
out << u << "\n" << J << "\n" << w << "\n\n";
|
||||
return std::tuple{u * det(J) * w};
|
||||
};
|
||||
|
||||
std::tuple input_descriptors = {Value{"potential"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
|
||||
std::tuple output_descriptors = {Value{"potential"}};
|
||||
ElementOperator eop{domain_qf, input_descriptors, output_descriptors};
|
||||
|
||||
auto ops = std::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
DifferentiableOperator dop{solutions, parameters, ops, mesh, ir};
|
||||
|
||||
Vector x(h1fes.GetTrueVSize()), y(h1fes.GetTrueVSize());
|
||||
|
||||
u.GetTrueDofs(x);
|
||||
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
// Derivative wrt "potential", indicated by the index 0 of the set {solutions} \cup {parameters}
|
||||
auto dFd0 = dop.GetDerivativeWrt<0>({&u}, {mesh_nodes});
|
||||
dFd0->Mult(x, y);
|
||||
|
||||
Vector dFd0_vec;
|
||||
dFd0->Assemble(dFd0_vec);
|
||||
|
||||
// Derivative wrt "coordinates", indicated by the index 1 of the set {solutions} \cup {parameters}
|
||||
auto dFd1 = dop.GetDerivativeWrt<1>({&u}, {mesh_nodes});
|
||||
dFd1->Mult(x, y);
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,302 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
template <typename momentum_t, typename mass_conservation_t>
|
||||
class NavierStokesOperator : public Operator
|
||||
{
|
||||
template <typename momentum_du_t, typename momentum_dp_t>
|
||||
class NavierStokesJacobianOperator : public Operator
|
||||
{
|
||||
public:
|
||||
NavierStokesJacobianOperator(const NavierStokesOperator *ns,
|
||||
std::shared_ptr<momentum_du_t> mom_du,
|
||||
std::shared_ptr<momentum_dp_t> mom_dp) :
|
||||
Operator(ns->Height()), ns(ns), block_op(ns->block_offsets)
|
||||
{
|
||||
mom_du->Assemble(A);
|
||||
A.EliminateBC(ns->vel_ess_tdofs, Operator::DiagonalPolicy::DIAG_ONE);
|
||||
|
||||
mom_dp->Assemble(D);
|
||||
D.EliminateRows(ns->vel_ess_tdofs);
|
||||
|
||||
Dt = new TransposeOperator(D);
|
||||
|
||||
block_op.SetBlock(0, 0, &A);
|
||||
block_op.SetBlock(0, 1, &D);
|
||||
block_op.SetBlock(1, 0, Dt);
|
||||
// std::ofstream amatofs("dfem_mat.dat");
|
||||
// block_op.PrintMatlab(amatofs);
|
||||
// amatofs.close();
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &y) const override
|
||||
{
|
||||
block_op.Mult(x, y);
|
||||
}
|
||||
|
||||
~NavierStokesJacobianOperator()
|
||||
{
|
||||
delete Dt;
|
||||
}
|
||||
|
||||
const NavierStokesOperator *ns = nullptr;
|
||||
HypreParMatrix A, D;
|
||||
TransposeOperator *Dt = nullptr;
|
||||
BlockOperator block_op;
|
||||
};
|
||||
|
||||
public:
|
||||
NavierStokesOperator(momentum_t &momentum,
|
||||
mass_conservation_t &mass_conservation,
|
||||
Array<int> &offsets, Array<int> &vel_ess_tdofs) :
|
||||
Operator(offsets.Last()), momentum(momentum),
|
||||
mass_conservation(mass_conservation),
|
||||
block_offsets(offsets), vel_ess_tdofs(vel_ess_tdofs) {}
|
||||
|
||||
void SetParameters(ParGridFunction &mesh_nodes)
|
||||
{
|
||||
momentum.SetParameters({&mesh_nodes});
|
||||
mass_conservation.SetParameters({&mesh_nodes});
|
||||
this->mesh_nodes.SetSpace(mesh_nodes.ParFESpace());
|
||||
this->mesh_nodes = mesh_nodes;
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &r) const override
|
||||
{
|
||||
Vector ru(r.ReadWrite() + block_offsets[0],
|
||||
block_offsets[1] - block_offsets[0]);
|
||||
Vector rp(r.ReadWrite() + block_offsets[1],
|
||||
block_offsets[2] - block_offsets[1]);
|
||||
|
||||
momentum.Mult(x, ru);
|
||||
|
||||
mass_conservation.Mult(x, rp);
|
||||
|
||||
ru.SetSubVector(vel_ess_tdofs, 0.0);
|
||||
}
|
||||
|
||||
Operator &GetGradient(const Vector &x) const override
|
||||
{
|
||||
xtmp = x;
|
||||
BlockVector xb(xtmp.ReadWrite(), block_offsets);
|
||||
|
||||
ParGridFunction u(const_cast<ParFiniteElementSpace *>
|
||||
(*std::get_if<const ParFiniteElementSpace *>
|
||||
(&momentum.solutions[0].data)));
|
||||
ParGridFunction p(const_cast<ParFiniteElementSpace *>
|
||||
(*std::get_if<const ParFiniteElementSpace *>
|
||||
(&momentum.solutions[1].data)));
|
||||
u.SetFromTrueDofs(xb.GetBlock(0));
|
||||
p.SetFromTrueDofs(xb.GetBlock(1));
|
||||
auto mom_du = momentum.template GetDerivativeWrt<0>({&u, &p}, {&mesh_nodes});
|
||||
auto mom_dp = momentum.template GetDerivativeWrt<1>({&u, &p}, {&mesh_nodes});
|
||||
delete jacobian_operator;
|
||||
jacobian_operator = new NavierStokesJacobianOperator<
|
||||
typename std::remove_pointer<decltype(mom_du.get())>::type,
|
||||
typename std::remove_pointer<decltype(mom_dp.get())>::type>(this, mom_du,
|
||||
mom_dp);
|
||||
return *jacobian_operator;
|
||||
}
|
||||
|
||||
momentum_t &momentum;
|
||||
mass_conservation_t &mass_conservation;
|
||||
|
||||
const Array<int> block_offsets;
|
||||
const Array<int> vel_ess_tdofs;
|
||||
mutable Vector xtmp;
|
||||
|
||||
mutable ParGridFunction mesh_nodes;
|
||||
|
||||
mutable Operator *jacobian_operator = nullptr;
|
||||
};
|
||||
|
||||
double reynolds = 10.0;
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
constexpr int dim = 3;
|
||||
constexpr int vdim = dim;
|
||||
|
||||
Mpi::Init();
|
||||
int num_procs = Mpi::WorldSize();
|
||||
int myid = Mpi::WorldRank();
|
||||
Hypre::Init();
|
||||
|
||||
const char *mesh_file = "../data/ref-cube.mesh";
|
||||
int polynomial_order = 2;
|
||||
int ir_order = 2;
|
||||
int refinements = 2;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&refinements, "-r", "--refinements", "");
|
||||
args.AddOption(&reynolds, "-rey", "--reynolds", "");
|
||||
args.ParseCheck();
|
||||
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection velocity_fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace velocity_fes(&mesh, &velocity_fec, dim);
|
||||
|
||||
H1_FECollection pressure_fec(polynomial_order - 1, dim);
|
||||
ParFiniteElementSpace pressure_fes(&mesh, &pressure_fec);
|
||||
|
||||
const IntegrationRule &velocity_ir =
|
||||
IntRules.Get(velocity_fes.GetFE(0)->GetGeomType(),
|
||||
ir_order * velocity_fec.GetOrder());
|
||||
|
||||
const IntegrationRule &pressure_ir =
|
||||
IntRules.Get(pressure_fes.GetFE(0)->GetGeomType(),
|
||||
ir_order * pressure_fec.GetOrder());
|
||||
|
||||
Array<int> bdr_attr_is_ess(mesh.bdr_attributes.Max());
|
||||
bdr_attr_is_ess = 1;
|
||||
Array<int> vel_ess_tdofs;
|
||||
velocity_fes.GetEssentialTrueDofs(bdr_attr_is_ess, vel_ess_tdofs);
|
||||
|
||||
ParGridFunction u(&velocity_fes);
|
||||
ParGridFunction p(&pressure_fes);
|
||||
|
||||
auto u_f = [](const Vector &coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double z = coords(2);
|
||||
if (z >= 1.0)
|
||||
{
|
||||
u(0) = 1.0;
|
||||
}
|
||||
else
|
||||
{
|
||||
u(0) = 0.0;
|
||||
}
|
||||
u(1) = 0.0;
|
||||
u(2) = 0.0;
|
||||
};
|
||||
auto u_coef = VectorFunctionCoefficient(dim, u_f);
|
||||
|
||||
u.ProjectCoefficient(u_coef);
|
||||
p = 0.0;
|
||||
|
||||
// -\nabla \cdot (\nabla u + p * I) -> (\nabla u + p * I, \nabla v)
|
||||
auto momentum_kernel = [](const tensor<double, dim> &u,
|
||||
const tensor<double, dim, dim> &dudxi,
|
||||
const double &p,
|
||||
const tensor<double, dim, dim> &J,
|
||||
const double &w)
|
||||
{
|
||||
static constexpr auto I = mfem::internal::IsotropicIdentity<dim>();
|
||||
auto invJ = inv(J);
|
||||
auto dudx = dudxi * invJ;
|
||||
double Re = reynolds;
|
||||
return mfem::tuple{(outer(u, u) - 1.0 / Re * dudx + p * I) * det(J) * w * transpose(invJ)};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators_0{Value{"velocity"}, Gradient{"velocity"}, Value{"pressure"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator_0{Gradient{"velocity"}};
|
||||
ElementOperator op_0{momentum_kernel, argument_operators_0, output_operator_0};
|
||||
|
||||
// (\nabla \cdot u, q)
|
||||
auto mass_conservation_kernel = [](const tensor<double, dim, dim> &dudxi,
|
||||
const tensor<double, dim, dim> &J,
|
||||
const double &w)
|
||||
{
|
||||
return mfem::tuple{tr(dudxi * inv(J)) * det(J) * w};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators_1{Gradient{"velocity"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator_1{Value{"pressure"}};
|
||||
ElementOperator op_1{mass_conservation_kernel, argument_operators_1, output_operator_1};
|
||||
|
||||
std::array solutions{FieldDescriptor{&velocity_fes, "velocity"}, FieldDescriptor{&pressure_fes, "pressure"}};
|
||||
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator momentum_op{solutions, parameters, mfem::tuple{op_0}, mesh, velocity_ir};
|
||||
DifferentiableOperator mass_conservation_op{solutions, parameters, mfem::tuple{op_1}, mesh, pressure_ir};
|
||||
|
||||
// Preconditioner form
|
||||
auto pressure_mass_kernel = [](const double &p,
|
||||
const tensor<double, dim, dim> &J,
|
||||
const double &w)
|
||||
{
|
||||
return mfem::tuple{p * det(J) * w};
|
||||
};
|
||||
|
||||
mfem::tuple pms_args{Value{"pressure"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple pms_outs{Value{"pressure"}};
|
||||
ElementOperator pressure_mass{pressure_mass_kernel, pms_args, pms_outs};
|
||||
std::array pms_sols{FieldDescriptor{&pressure_fes, "pressure"}};
|
||||
std::array pms_params{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
DifferentiableOperator pressure_mass_op{pms_sols, pms_params, mfem::tuple{pressure_mass}, mesh, pressure_ir};
|
||||
|
||||
Array<int> block_offsets(3);
|
||||
block_offsets[0] = 0;
|
||||
block_offsets[1] = velocity_fes.GetTrueVSize();
|
||||
block_offsets[2] = pressure_fes.GetTrueVSize();
|
||||
block_offsets.PartialSum();
|
||||
|
||||
NavierStokesOperator navierstokes(momentum_op, mass_conservation_op,
|
||||
block_offsets,
|
||||
vel_ess_tdofs);
|
||||
|
||||
BlockVector x(block_offsets), y(block_offsets);
|
||||
u.ParallelProject(x.GetBlock(0));
|
||||
// p.ParallelProject(x.GetBlock(1));
|
||||
navierstokes.SetParameters(*mesh_nodes);
|
||||
|
||||
HypreParMatrix A;
|
||||
momentum_op.template GetDerivativeWrt<0>({&u, &p}, {mesh_nodes})->Assemble(A);
|
||||
A.EliminateBC(vel_ess_tdofs, Operator::DiagonalPolicy::DIAG_ONE);
|
||||
HypreBoomerAMG amg(A);
|
||||
amg.SetMaxLevels(50);
|
||||
amg.SetPrintLevel(0);
|
||||
|
||||
HypreParMatrix Mp;
|
||||
pressure_mass_op.template GetDerivativeWrt<0>({&p}, {mesh_nodes})->Assemble(Mp);
|
||||
|
||||
HypreDiagScale Mp_inv(Mp);
|
||||
|
||||
BlockDiagonalPreconditioner prec(block_offsets);
|
||||
prec.SetDiagonalBlock(0, &amg);
|
||||
prec.SetDiagonalBlock(1, &Mp_inv);
|
||||
|
||||
GMRESSolver solver(MPI_COMM_WORLD);
|
||||
solver.SetAbsTol(0.0);
|
||||
solver.SetRelTol(1e-8);
|
||||
solver.SetKDim(100);
|
||||
solver.SetMaxIter(500);
|
||||
solver.SetPrintLevel(2);
|
||||
solver.SetPreconditioner(prec);
|
||||
|
||||
NewtonSolver newton(MPI_COMM_WORLD);
|
||||
newton.SetOperator(navierstokes);
|
||||
newton.SetSolver(solver);
|
||||
newton.SetRelTol(1e-8);
|
||||
newton.SetMaxIter(50);
|
||||
newton.SetPrintLevel(1);
|
||||
|
||||
Vector zero;
|
||||
newton.Mult(zero, x);
|
||||
|
||||
u.SetFromTrueDofs(x.GetBlock(0));
|
||||
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock << "parallel " << num_procs << " " << myid << "\n";
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << u << std::flush;
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,174 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_diffusion(
|
||||
std::string mesh_file, int refinements, int polynomial_order)
|
||||
{
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
MFEM_ASSERT(mesh_serial.Dimension() == 2, "incorrect mesh dimension");
|
||||
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
mesh_serial.Clear();
|
||||
|
||||
out << "#el: " << mesh.GetNE() << "\n";
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
|
||||
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
|
||||
|
||||
const IntegrationRule& ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder());
|
||||
|
||||
out << "#qp: " << ir.GetNPoints() << "\n";
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
ParGridFunction rho_g(&h1fes);
|
||||
|
||||
auto kernel = [] MFEM_HOST_DEVICE(const tensor<double, 2, 2>& J,
|
||||
const double& w, const tensor<double, 2>& dudxi)
|
||||
{
|
||||
auto invJ = inv(J);
|
||||
return mfem::tuple{dudxi * invJ * transpose(invJ) * det(J) * w};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators =
|
||||
{
|
||||
Gradient{"coordinates"}, Weight{}, Gradient{"potential"}
|
||||
};
|
||||
mfem::tuple output_operator = {Gradient{"potential"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector& coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
return 2.345 + 0.25 * x * x * y + y * y * x;
|
||||
};
|
||||
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(f1_g), y(h1fes.TrueVSize());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
y.HostRead();
|
||||
|
||||
ParBilinearForm a(&h1fes);
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator);
|
||||
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
a.Assemble();
|
||||
a.Finalize();
|
||||
|
||||
Vector y2(h1fes.TrueVSize());
|
||||
a.Mult(x, y2);
|
||||
y2.HostRead();
|
||||
Vector diff(y2);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
print_vector(diff);
|
||||
print_vector(y2);
|
||||
print_vector(y);
|
||||
return 1;
|
||||
}
|
||||
|
||||
// // Test linearization here as well
|
||||
// auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
|
||||
|
||||
// if (dFdu->Height() != h1fes.GetTrueVSize())
|
||||
// {
|
||||
// out << "dFdu unexpected height of " << dFdu->Height() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
|
||||
// dFdu->Mult(x, y);
|
||||
// y.HostRead();
|
||||
// a.Mult(x, y2);
|
||||
// y2.HostRead();
|
||||
|
||||
// diff = y2;
|
||||
// diff -= y;
|
||||
// if (diff.Norml2() > 1e-10)
|
||||
// {
|
||||
// print_vector(diff);
|
||||
// print_vector(y2);
|
||||
// print_vector(y);
|
||||
// return 1;
|
||||
// }
|
||||
|
||||
// // fd jacobian test
|
||||
// {
|
||||
// double eps = 1.0e-6;
|
||||
// Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
|
||||
// v *= eps;
|
||||
// xpv += v;
|
||||
// xmv -= v;
|
||||
// dop.Mult(xpv, fxpv);
|
||||
// dop.Mult(xmv, fxmv);
|
||||
// fxpv -= fxmv;
|
||||
// fxpv /= (2.0*eps);
|
||||
|
||||
// fxpv -= y;
|
||||
// if (fxpv.Norml2() > eps)
|
||||
// {
|
||||
// out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
// }
|
||||
|
||||
// f1_g.ProjectCoefficient(f1_c);
|
||||
// rho_g.ProjectCoefficient(rho_c);
|
||||
// auto dFdrho = dop.GetDerivativeWrt<1>({&f1_g}, {&rho_g, mesh_nodes});
|
||||
// if (dFdrho->Height() != h1fes.GetTrueVSize())
|
||||
// {
|
||||
// out << "dFdrho unexpected height of " << dFdrho->Height() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
|
||||
// dFdrho->Mult(rho_g, y);
|
||||
|
||||
// // fd test
|
||||
// {
|
||||
// double eps = 1.0e-6;
|
||||
// Vector v(rho_g), rhopv(rho_g), rhomv(rho_g), frhopv(x.Size()),
|
||||
// frhomv(x.Size()); v *= eps; rhopv += v; rhomv -= v;
|
||||
// dop.SetParameters({&rhopv, mesh_nodes});
|
||||
// dop.Mult(x, frhopv);
|
||||
// dop.SetParameters({&rhomv, mesh_nodes});
|
||||
// dop.Mult(x, frhomv);
|
||||
// frhopv -= frhomv;
|
||||
// frhopv /= (2.0*eps);
|
||||
|
||||
// frhopv -= y;
|
||||
// if (frhopv.Norml2() > eps)
|
||||
// {
|
||||
// out << "||dFdu_FD u^* - ex||_l2 = " << frhopv.Norml2() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
// }
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_diffusion);
|
||||
@@ -0,0 +1,296 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
#include "examples/dfem/dfem_parametricspace.hpp"
|
||||
#include "fem/bilininteg.hpp"
|
||||
#include "general/tic_toc.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
|
||||
int test_diffusion_3d(
|
||||
std::string mesh_file, int refinements, int polynomial_order)
|
||||
{
|
||||
constexpr int num_samples = 10;
|
||||
constexpr int dim = 3;
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
MFEM_ASSERT(mesh_serial.Dimension() == dim, "incorrect mesh dimension");
|
||||
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(polynomial_order);
|
||||
mesh_serial.Clear();
|
||||
|
||||
out << "#el: " << mesh.GetNE() << "\n";
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
|
||||
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
|
||||
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
|
||||
|
||||
const IntegrationRule& ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
|
||||
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
|
||||
0)->GetDim() - 1);
|
||||
|
||||
printf("#ndof per el = %d\n", h1fes.GetFE(0)->GetDof());
|
||||
printf("#nqp = %d\n", ir.GetNPoints());
|
||||
printf("#q1d = %d\n", (int)floor(pow(ir.GetNPoints(), 1.0/dim) + 0.5));
|
||||
|
||||
ParametricSpace qdata_space(dim, dim * dim, ir.GetNPoints(),
|
||||
dim * dim * ir.GetNPoints() * mesh.GetNE());
|
||||
ParametricFunction qdata(qdata_space);
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
ParGridFunction rho_g(&h1fes);
|
||||
|
||||
auto f1 = [](const Vector& coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
const double z = coords(2);
|
||||
return 2.345 + x + x*y + 1.25 * z*x;
|
||||
};
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(f1_g), y(h1fes.GetTrueVSize());
|
||||
{
|
||||
auto diffusion_mf_kernel =
|
||||
[] MFEM_HOST_DEVICE (
|
||||
const tensor<double, dim>& dudxi,
|
||||
const tensor<double, dim, dim>& J,
|
||||
const double& w)
|
||||
{
|
||||
auto invJ = inv(J);
|
||||
return mfem::tuple{dudxi * invJ * transpose(invJ) * det(J) * w};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Gradient{"potential"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator = {Gradient{"potential"}};
|
||||
|
||||
ElementOperator eop = {diffusion_mf_kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
dop.SetParameters({mesh_nodes});
|
||||
StopWatch sw;
|
||||
sw.Start();
|
||||
for (int i = 0; i < num_samples; i++)
|
||||
{
|
||||
dop.Mult(x, y);
|
||||
}
|
||||
sw.Stop();
|
||||
printf("dfem mf: %fs\n", sw.RealTime() / num_samples);
|
||||
y.HostRead();
|
||||
}
|
||||
|
||||
{
|
||||
auto diffusion_setup_kernel =
|
||||
[] MFEM_HOST_DEVICE (
|
||||
const tensor<double, dim, dim>& J,
|
||||
const double& w)
|
||||
{
|
||||
auto invJ = inv(J);
|
||||
return mfem::tuple{invJ * transpose(invJ) * det(J) * w};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator = {None{"qdata"}};
|
||||
|
||||
ElementOperator eop = {diffusion_setup_kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array
|
||||
{
|
||||
FieldDescriptor{&mesh_fes, "coordinates"},
|
||||
FieldDescriptor{&qdata_space, "qdata"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
dop.SetParameters({mesh_nodes, &qdata});
|
||||
StopWatch sw;
|
||||
sw.Start();
|
||||
for (int i = 0; i < num_samples; i++)
|
||||
{
|
||||
dop.Mult(x, qdata);
|
||||
}
|
||||
sw.Stop();
|
||||
printf("dfem pa setup: %fs\n", sw.RealTime() / num_samples);
|
||||
qdata.HostRead();
|
||||
}
|
||||
|
||||
// printf("qdata: ");
|
||||
// print_vector(qdata);
|
||||
|
||||
{
|
||||
auto diffusion_apply_kernel =
|
||||
[] MFEM_HOST_DEVICE (
|
||||
const tensor<double, dim>& dudxi,
|
||||
const tensor<double, dim, dim>& qdata)
|
||||
{
|
||||
return mfem::tuple{dudxi * qdata};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Gradient{"potential"}, None{"qdata"}};
|
||||
mfem::tuple output_operator = {Gradient{"potential"}};
|
||||
|
||||
ElementOperator eop = {diffusion_apply_kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&qdata_space, "qdata"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
dop.SetParameters({&qdata});
|
||||
StopWatch sw;
|
||||
sw.Start();
|
||||
for (int i = 0; i < num_samples; i++)
|
||||
{
|
||||
dop.Mult(x, y);
|
||||
}
|
||||
sw.Stop();
|
||||
printf("dfem pa apply: %fs\n", sw.RealTime() / num_samples);
|
||||
y.HostRead();
|
||||
}
|
||||
|
||||
// printf("y: ");
|
||||
// print_vector(y);
|
||||
|
||||
Vector y2(h1fes.TrueVSize());
|
||||
{
|
||||
ParBilinearForm a(&h1fes);
|
||||
auto diff_integ = new DiffusionIntegrator;
|
||||
diff_integ->SetIntRule(&ir);
|
||||
a.AddDomainIntegrator(diff_integ);
|
||||
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
|
||||
OperatorPtr A;
|
||||
StopWatch sw;
|
||||
sw.Start();
|
||||
a.Assemble();
|
||||
a.Finalize();
|
||||
Array<int> empty;
|
||||
a.FormSystemMatrix(empty, A);
|
||||
sw.Stop();
|
||||
printf("mfem pa setup: %fs\n", sw.RealTime());
|
||||
|
||||
sw.Clear();
|
||||
sw.Start();
|
||||
for (int i = 0; i < num_samples; i++)
|
||||
{
|
||||
A->Mult(x, y2);
|
||||
}
|
||||
sw.Stop();
|
||||
printf("mfem pa apply: %fs\n", sw.RealTime() / num_samples);
|
||||
y2.HostRead();
|
||||
}
|
||||
// printf("y2: ");
|
||||
// print_vector(y2);
|
||||
|
||||
Vector diff(y2);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-15)
|
||||
{
|
||||
// printf("y ");
|
||||
// print_vector(y);
|
||||
// printf("y2: ");
|
||||
// print_vector(y2);
|
||||
// printf("diff: ");
|
||||
// print_vector(diff);
|
||||
return 1;
|
||||
}
|
||||
|
||||
// Test linearization here as well
|
||||
// auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
|
||||
|
||||
// if (dFdu->Height() != h1fes.GetTrueVSize())
|
||||
// {
|
||||
// out << "dFdu unexpected height of " << dFdu->Height() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
|
||||
// dFdu->Mult(x, y);
|
||||
// y.HostRead();
|
||||
// a.Mult(x, y2);
|
||||
// y2.HostRead();
|
||||
|
||||
// diff = y2;
|
||||
// diff -= y;
|
||||
// if (diff.Norml2() > 1e-10)
|
||||
// {
|
||||
// print_vector(diff);
|
||||
// print_vector(y2);
|
||||
// print_vector(y);
|
||||
// return 1;
|
||||
// }
|
||||
|
||||
// // fd jacobian test
|
||||
// {
|
||||
// double eps = 1.0e-6;
|
||||
// Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
|
||||
// v *= eps;
|
||||
// xpv += v;
|
||||
// xmv -= v;
|
||||
// dop.Mult(xpv, fxpv);
|
||||
// dop.Mult(xmv, fxmv);
|
||||
// fxpv -= fxmv;
|
||||
// fxpv /= (2.0*eps);
|
||||
|
||||
// fxpv -= y;
|
||||
// if (fxpv.Norml2() > eps)
|
||||
// {
|
||||
// out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
// }
|
||||
|
||||
// f1_g.ProjectCoefficient(f1_c);
|
||||
// rho_g.ProjectCoefficient(rho_c);
|
||||
// auto dFdrho = dop.GetDerivativeWrt<1>({&f1_g}, {&rho_g, mesh_nodes});
|
||||
// if (dFdrho->Height() != h1fes.GetTrueVSize())
|
||||
// {
|
||||
// out << "dFdrho unexpected height of " << dFdrho->Height() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
|
||||
// dFdrho->Mult(rho_g, y);
|
||||
|
||||
// // fd test
|
||||
// {
|
||||
// double eps = 1.0e-6;
|
||||
// Vector v(rho_g), rhopv(rho_g), rhomv(rho_g), frhopv(x.Size()),
|
||||
// frhomv(x.Size()); v *= eps; rhopv += v; rhomv -= v;
|
||||
// dop.SetParameters({&rhopv, mesh_nodes});
|
||||
// dop.Mult(x, frhopv);
|
||||
// dop.SetParameters({&rhomv, mesh_nodes});
|
||||
// dop.Mult(x, frhomv);
|
||||
// frhopv -= frhomv;
|
||||
// frhopv /= (2.0*eps);
|
||||
|
||||
// frhopv -= y;
|
||||
// if (frhopv.Norml2() > eps)
|
||||
// {
|
||||
// out << "||dFdu_FD u^* - ex||_l2 = " << frhopv.Norml2() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
// }
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_diffusion_3d);
|
||||
@@ -0,0 +1,109 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_elasticity(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
const int vdim = dim;
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
Array<int> ess_bdr(mesh.bdr_attributes.Max());
|
||||
Array<int> ess_tdof;
|
||||
ess_bdr = 1;
|
||||
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 6 * h1fec.GetOrder());
|
||||
|
||||
out << "#qp: " << ir.GetNPoints() << "\n";
|
||||
out << "#dof_el: " << h1fes.GetRestrictionMatrix()->Height() / mesh.GetNE() <<
|
||||
"\n";
|
||||
|
||||
ParGridFunction u(&h1fes);
|
||||
|
||||
auto f1 = [](const Vector& coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
u(0) = 2.345 + 0.25 * x * x * y + y * y * x;
|
||||
u(1) = 2.345 - 0.25 * x * y * y + y * x * x;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient u_c(dim, f1);
|
||||
u.ProjectCoefficient(u_c);
|
||||
|
||||
ConstantCoefficient l_coeff(0.5), m_coeff(0.25);
|
||||
|
||||
ParBilinearForm A_form(&h1fes);
|
||||
auto A_integ = new ElasticityIntegrator(l_coeff, m_coeff);
|
||||
A_integ->SetIntegrationRule(ir);
|
||||
A_form.AddDomainIntegrator(A_integ);
|
||||
A_form.Assemble();
|
||||
A_form.Finalize();
|
||||
|
||||
auto elasticity_kernel = [](const tensor<double, 2, 2> &dudxi,
|
||||
const tensor<double, 2, 2> &J,
|
||||
const double &w)
|
||||
{
|
||||
constexpr double lambda = 0.5;
|
||||
constexpr double mu = 0.25;
|
||||
static constexpr auto I = mfem::internal::IsotropicIdentity<2>();
|
||||
auto invJ = inv(J);
|
||||
auto eps = sym(dudxi * invJ);
|
||||
return mfem::tuple{transpose(lambda * tr(eps) * I + 2.0 * mu * eps) * det(J) * w * transpose(invJ)};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators{Gradient{"displacement"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator{Gradient{"displacement"}};
|
||||
|
||||
ElementOperator op{elasticity_kernel, argument_operators, output_operator};
|
||||
|
||||
std::array solutions{FieldDescriptor{&h1fes, "displacement"}};
|
||||
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop{solutions, parameters, mfem::tuple{op}, mesh, ir};
|
||||
|
||||
Vector x(u), y1(h1fes.GetTrueVSize()),
|
||||
y2(h1fes.GetTrueVSize());
|
||||
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y1);
|
||||
y1.HostRead();
|
||||
|
||||
A_form.Mult(x, y2);
|
||||
y2.HostRead();
|
||||
|
||||
Vector diff(y2);
|
||||
diff -= y1;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
out << "||F(u) - ex||_l2 = " << diff.Norml2() << "\n";
|
||||
print_vector(diff);
|
||||
print_vector(y1);
|
||||
print_vector(y2);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_elasticity);
|
||||
@@ -0,0 +1,115 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
#include "examples/dfem/dfem_parametricspace.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_interpolate_gradient_linear_scalar_3d(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
constexpr int dim = 3;
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
|
||||
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction *mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
// const IntegrationRule &ir =
|
||||
// IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
IntegrationRules gll_rules(0, Quadrature1D::GaussLobatto);
|
||||
const IntegrationRule &ir = gll_rules.Get(h1fes.GetFE(0)->GetGeomType(),
|
||||
2 * polynomial_order - 1);
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
|
||||
ParametricSpace pspace(dim, dim, ir.GetNPoints(),
|
||||
dim * ir.GetNPoints() * mesh.GetNE());
|
||||
ParametricFunction qdata(pspace);
|
||||
|
||||
auto kernel = [](const tensor<double, dim> &dudxi,
|
||||
const tensor<double, dim, dim> &J)
|
||||
{
|
||||
return mfem::tuple{dudxi * inv(J)};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Gradient{"potential"}, Gradient{"coordinates"}};
|
||||
mfem::tuple output_operator = {None{"qdata"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array
|
||||
{
|
||||
FieldDescriptor{&mesh_fes, "coordinates"},
|
||||
FieldDescriptor{&pspace, "qdata"}
|
||||
};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
const double z = coords(2);
|
||||
return 2.345 + x * y * z + y * z;
|
||||
};
|
||||
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize() * dim);
|
||||
dop.SetParameters({mesh_nodes, &qdata});
|
||||
dop.Mult(x, y);
|
||||
|
||||
Vector f_test(h1fes.GetElementRestriction(
|
||||
ElementDofOrdering::LEXICOGRAPHIC)->Height() * dim);
|
||||
for (int e = 0; e < mesh.GetNE(); e++)
|
||||
{
|
||||
ElementTransformation *T = mesh.GetElementTransformation(e);
|
||||
|
||||
for (int qp = 0; qp < ir.GetNPoints(); qp++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(qp);
|
||||
T->SetIntPoint(&ip);
|
||||
|
||||
Vector g(dim);
|
||||
f1_g.GetGradient(*T, g);
|
||||
// printf("(%f, %f, %f): (%f, %f, %f)\n", ip.x, ip.y, ip.z, g(0), g(1), g(2));
|
||||
for (int d = 0; d < dim; d++)
|
||||
{
|
||||
int qpo = qp * dim;
|
||||
int eo = e * (ir.GetNPoints() * dim);
|
||||
f_test(d + qpo + eo) = g(d);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vector diff(f_test);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
print_vector(diff);
|
||||
print_vector(f_test);
|
||||
print_vector(y);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_interpolate_gradient_linear_scalar_3d);
|
||||
@@ -0,0 +1,91 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_interpolate_linear_scalar(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
|
||||
auto kernel = [](const double &u, const tensor<double, 2, 2> &J,
|
||||
const double &w)
|
||||
{
|
||||
return mfem::tuple{u};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Value{"potential"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator = {None{"potential"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
return 2.345 + x + y;
|
||||
};
|
||||
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
Vector f_test(h1fes.GetElementRestriction(
|
||||
ElementDofOrdering::LEXICOGRAPHIC)->Height());
|
||||
for (int e = 0; e < mesh.GetNE(); e++)
|
||||
{
|
||||
ElementTransformation *T = mesh.GetElementTransformation(e);
|
||||
for (int qp = 0; qp < ir.GetNPoints(); qp++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(qp);
|
||||
T->SetIntPoint(&ip);
|
||||
|
||||
f_test((e * ir.GetNPoints()) + qp) = f1_c.Eval(*T, ip);
|
||||
}
|
||||
}
|
||||
|
||||
Vector diff(f_test);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
print_vector(diff);
|
||||
print_vector(f_test);
|
||||
print_vector(y);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_interpolate_linear_scalar);
|
||||
@@ -0,0 +1,93 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_interpolate_linear_scalar_3d(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
constexpr int dim = 3;
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
|
||||
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction *mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
|
||||
auto kernel = [](const double &u)
|
||||
{
|
||||
return mfem::tuple{u};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Value{"potential"}};
|
||||
mfem::tuple output_operator = {None{"potential"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
const double z = coords(2);
|
||||
return 2.345 + x + y + 1.25 * z;
|
||||
};
|
||||
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
Vector f_test(h1fes.GetElementRestriction(
|
||||
ElementDofOrdering::LEXICOGRAPHIC)->Height());
|
||||
for (int e = 0; e < mesh.GetNE(); e++)
|
||||
{
|
||||
ElementTransformation *T = mesh.GetElementTransformation(e);
|
||||
for (int qp = 0; qp < ir.GetNPoints(); qp++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(qp);
|
||||
T->SetIntPoint(&ip);
|
||||
|
||||
f_test((e * ir.GetNPoints()) + qp) = f1_c.Eval(*T, ip);
|
||||
}
|
||||
}
|
||||
|
||||
Vector diff(f_test);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
print_vector(diff);
|
||||
print_vector(f_test);
|
||||
print_vector(y);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_interpolate_linear_scalar_3d);
|
||||
@@ -0,0 +1,100 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_interpolate_linear_vector(std::string mesh_file, int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
constexpr int vdim = 2;
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
QuadratureSpace qspace(mesh, ir);
|
||||
QuadratureFunction qf(&qspace, vdim);
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
|
||||
auto kernel = [](const tensor<double, 2> &u)
|
||||
{
|
||||
return mfem::tuple{u};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Value{"potential"}};
|
||||
mfem::tuple output_operator = {None{"potential"}};
|
||||
|
||||
ElementOperator eop{kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector &coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
u(0) = 2.345 + x + y;
|
||||
u(1) = 12.345 + x + y;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient f1_c(vdim, f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(f1_g), y(f1_g.Size());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
Vector f_test(qf.Size());
|
||||
for (int e = 0; e < mesh.GetNE(); e++)
|
||||
{
|
||||
ElementTransformation *T = mesh.GetElementTransformation(e);
|
||||
for (int qp = 0; qp < ir.GetNPoints(); qp++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(qp);
|
||||
T->SetIntPoint(&ip);
|
||||
|
||||
Vector f(vdim);
|
||||
f1_g.GetVectorValue(*T, ip, f);
|
||||
for (int d = 0; d < vdim; d++)
|
||||
{
|
||||
int qpo = qp * vdim;
|
||||
int eo = e * (ir.GetNPoints() * vdim);
|
||||
f_test(d + qpo + eo) = f(d);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vector diff(f_test);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
print_vector(diff);
|
||||
print_vector(f_test);
|
||||
print_vector(y);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_interpolate_linear_vector);
|
||||
@@ -0,0 +1,105 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_interpolate_linear_vector_3d(std::string mesh_file, int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
|
||||
constexpr int dim = 3;
|
||||
constexpr int vdim = 3;
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
|
||||
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
mesh.SetCurvature(1);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
QuadratureSpace qspace(mesh, ir);
|
||||
QuadratureFunction qf(&qspace, vdim);
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
|
||||
auto kernel = [](const tensor<double, vdim> &u)
|
||||
{
|
||||
return mfem::tuple{u};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Value{"potential"}};
|
||||
mfem::tuple output_operator = {None{"potential"}};
|
||||
|
||||
ElementOperator eop{kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector &coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
const double z = coords(2);
|
||||
u(0) = 2.345 + x + y + 3.0 * z;
|
||||
u(1) = 12.345 + x + y + 2.0 * z;
|
||||
u(2) = 5.345 + x + y + 1.0 * z;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient f1_c(vdim, f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(f1_g), y(f1_g.Size());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
Vector f_test(qf.Size());
|
||||
for (int e = 0; e < mesh.GetNE(); e++)
|
||||
{
|
||||
ElementTransformation *T = mesh.GetElementTransformation(e);
|
||||
for (int qp = 0; qp < ir.GetNPoints(); qp++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(qp);
|
||||
T->SetIntPoint(&ip);
|
||||
|
||||
Vector f(vdim);
|
||||
f1_g.GetVectorValue(*T, ip, f);
|
||||
for (int d = 0; d < vdim; d++)
|
||||
{
|
||||
int qpo = qp * vdim;
|
||||
int eo = e * (ir.GetNPoints() * vdim);
|
||||
f_test(d + qpo + eo) = f(d);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vector diff(f_test);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
print_vector(diff);
|
||||
print_vector(f_test);
|
||||
print_vector(y);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_interpolate_linear_vector_3d);
|
||||
@@ -0,0 +1,113 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
#include "fem/bilininteg.hpp"
|
||||
#include "fem/normal_deriv_restriction.hpp"
|
||||
#include <fstream>
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int dfem_test_mass_scalar_2d(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
constexpr int dim = 2;
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
|
||||
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction *mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() + 1);
|
||||
|
||||
// IntegrationRules gll_rules(0, Quadrature1D::GaussLobatto);
|
||||
// const IntegrationRule &ir = gll_rules.Get(h1fes.GetFE(0)->GetGeomType(),
|
||||
// 2 * polynomial_order - 1);
|
||||
|
||||
printf("#nqp = %d\n", ir.GetNPoints());
|
||||
printf("#q1d = %d\n", (int)floor(pow(ir.GetNPoints(), 1.0/dim) + 0.5));
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
|
||||
auto kernel = [](const double& u,
|
||||
const tensor<double, dim> x,
|
||||
const tensor<double, dim, dim> J,
|
||||
const double& w)
|
||||
{
|
||||
out << x << ": " << u << "\n";
|
||||
return mfem::tuple{u * w * det(J)};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Value{"potential"}, Value{"coordinates"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator = {Value{"potential"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
return 2.345 + x + x*y + 1.25 * x;
|
||||
};
|
||||
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector f1_g_e(f1_g.Size());
|
||||
auto R = h1fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC);
|
||||
// R->Mult(f1_g, f1_g_e);
|
||||
auto r_out = std::ofstream("r_mat.mtx");
|
||||
R->PrintMatlab(r_out);
|
||||
r_out.close();
|
||||
print_vector(f1_g);
|
||||
// print_vector(f1_g_e);
|
||||
|
||||
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
ParBilinearForm a(&h1fes);
|
||||
auto mass_integ = new MassIntegrator;
|
||||
mass_integ->SetIntRule(&ir);
|
||||
a.AddDomainIntegrator(mass_integ);
|
||||
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
a.Assemble();
|
||||
a.Finalize();
|
||||
|
||||
Vector y2(h1fes.TrueVSize());
|
||||
a.Mult(x, y2);
|
||||
y2.HostRead();
|
||||
|
||||
Vector diff(y2);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
print_vector(diff);
|
||||
print_vector(y2);
|
||||
print_vector(y);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(dfem_test_mass_scalar_2d);
|
||||
@@ -0,0 +1,147 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
#include "fem/bilininteg.hpp"
|
||||
#include "fem/fe/fe_base.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int dfem_test_mass_scalar_3d(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
constexpr int dim = 3;
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
MFEM_ASSERT(mesh_serial.Dimension() == dim, "wrong mesh dimension");
|
||||
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(polynomial_order);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction *mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
const IntegrationRule& ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
|
||||
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
|
||||
0)->GetDim() - 1);
|
||||
|
||||
// IntegrationRules gll_rules(0, Quadrature1D::GaussLobatto);
|
||||
// const IntegrationRule &ir = gll_rules.Get(h1fes.GetFE(0)->GetGeomType(),
|
||||
// 2 * polynomial_order - 1);
|
||||
|
||||
auto dtq = h1fes.GetFE(0)->GetDofToQuad(ir, DofToQuad::TENSOR);
|
||||
// printf("\n B: ");
|
||||
// dtq.B.Print(out, dtq.B.Size());
|
||||
// printf("\n G: ");
|
||||
// dtq.G.Print(out, dtq.G.Size());
|
||||
// printf("\n w: ");
|
||||
// ir.GetWeights().Print(out, ir.GetWeights().Size());
|
||||
|
||||
// printf("#ndof per el = %d\n", h1fes.GetFE(0)->GetDof());
|
||||
// printf("#nqp = %d\n", ir.GetNPoints());
|
||||
// printf("#q1d = %d\n", (int)floor(pow(ir.GetNPoints(), 1.0/dim) + 0.5));
|
||||
|
||||
// printf("nodes: ");
|
||||
// print_vector(*mesh_nodes);
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
|
||||
auto kernel = [](const double &u,
|
||||
const tensor<double, dim, dim> &J,
|
||||
const double &w)
|
||||
{
|
||||
return mfem::tuple{u * det(J) * w};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Value{"potential"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator = {Value{"potential"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
const double z = coords(2);
|
||||
return 2.345 + x + x*y + 1.25 * z*x;
|
||||
};
|
||||
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
// printf("\nf1_g: ");
|
||||
// print_vector(f1_g);
|
||||
|
||||
auto R = h1fes.GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC);
|
||||
// Vector f1_g_e(R->Height());
|
||||
// R->Mult(f1_g, f1_g_e);
|
||||
// printf("\nf1_g_e: ");
|
||||
// print_vector(f1_g_e);
|
||||
// auto r_out = std::ofstream("r_mat.mtx");
|
||||
// R->PrintMatlab(r_out);
|
||||
// r_out.close();
|
||||
|
||||
Vector x(*f1_g.GetTrueDofs()), y(h1fes.TrueVSize());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
ParBilinearForm a(&h1fes);
|
||||
auto mass_integ = new MassIntegrator;
|
||||
mass_integ->SetIntRule(&ir);
|
||||
a.AddDomainIntegrator(mass_integ);
|
||||
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
a.Assemble();
|
||||
a.Finalize();
|
||||
|
||||
Vector y2(h1fes.TrueVSize());
|
||||
a.Mult(x, y2);
|
||||
y2.HostRead();
|
||||
|
||||
Vector diff(y2);
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-15)
|
||||
{
|
||||
printf("y ");
|
||||
print_vector(y);
|
||||
printf("y2: ");
|
||||
print_vector(y2);
|
||||
printf("diff: ");
|
||||
print_vector(diff);
|
||||
return 1;
|
||||
}
|
||||
|
||||
Vector y3(h1fes.TrueVSize());
|
||||
auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
|
||||
|
||||
dFdu->Mult(x, y3);
|
||||
diff = y2;
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-15)
|
||||
{
|
||||
printf("y2 ");
|
||||
print_vector(y2);
|
||||
printf("y3: ");
|
||||
print_vector(y3);
|
||||
printf("diff: ");
|
||||
print_vector(diff);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(dfem_test_mass_scalar_3d);
|
||||
@@ -0,0 +1,114 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_neo_hookean_elasticity_2d(
|
||||
std::string mesh_file, int refinements, int polynomial_order)
|
||||
{
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
MFEM_ASSERT(dim == 2, "This test is for 2D meshes only");
|
||||
mesh_serial.Clear();
|
||||
|
||||
out << "#el: " << mesh.GetNE() << "\n";
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
|
||||
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, dim);
|
||||
|
||||
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
|
||||
|
||||
const IntegrationRule& ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder());
|
||||
|
||||
out << "#qp: " << ir.GetNPoints() << "\n";
|
||||
|
||||
ParGridFunction u_g(&h1fes);
|
||||
|
||||
auto kernel = [] MFEM_HOST_DEVICE(const tensor<double, 2, 2>& J,
|
||||
const double& w,
|
||||
const tensor<double, 2, 2>& dudxi)
|
||||
{
|
||||
// Neo-Hookean parameters
|
||||
const double lambda = 1.0;
|
||||
const double mu = 0.5;
|
||||
|
||||
static constexpr auto I = mfem::internal::IsotropicIdentity<2>();
|
||||
auto F = I + (dudxi * inv(J));
|
||||
auto E = 0.5 * (transpose(F) * F - I);
|
||||
auto invF = inv(F);
|
||||
|
||||
// 2D plane strain formulation
|
||||
auto P = mu * (F - transpose(invF)) + lambda * log(det(F)) * transpose(invF);
|
||||
|
||||
return mfem::tuple{P * det(J) * w};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators = {Gradient{"coordinates"}, Weight{},
|
||||
Gradient{"displacement"}
|
||||
};
|
||||
mfem::tuple output_operator = {Gradient{"displacement"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "displacement"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto displacement = [](const Vector& coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
u(0) = 0.1 * x * y;
|
||||
u(1) = 0.1 * y * x;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient disp_coeff(2, displacement);
|
||||
u_g.ProjectCoefficient(disp_coeff);
|
||||
|
||||
Vector x(u_g), y(h1fes.TrueVSize());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
y.HostRead();
|
||||
|
||||
// Test linearization
|
||||
auto dFdu = dop.GetDerivativeWrt<0>({&u_g}, {mesh_nodes});
|
||||
dFdu->Mult(x, y);
|
||||
|
||||
// Finite difference Jacobian test
|
||||
{
|
||||
double eps = 1.0e-6;
|
||||
Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
|
||||
v *= eps;
|
||||
xpv += v;
|
||||
xmv -= v;
|
||||
dop.Mult(xpv, fxpv);
|
||||
dop.Mult(xmv, fxmv);
|
||||
fxpv -= fxmv;
|
||||
fxpv /= (2.0*eps);
|
||||
|
||||
fxpv -= y;
|
||||
if (fxpv.Norml2() > eps)
|
||||
{
|
||||
out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_neo_hookean_elasticity_2d);
|
||||
@@ -0,0 +1,169 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_nonlinear_diffusion(
|
||||
std::string mesh_file, int refinements, int polynomial_order)
|
||||
{
|
||||
constexpr int dim = 3;
|
||||
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
mesh_serial.Clear();
|
||||
|
||||
out << "#el: " << mesh.GetNE() << "\n";
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction*>(mesh.GetNodes());
|
||||
ParFiniteElementSpace& mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
out << "#dofs " << h1fes.GetTrueVSize() << "\n";
|
||||
|
||||
const IntegrationRule& ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
|
||||
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
|
||||
0)->GetDim() - 1);
|
||||
|
||||
out << "#qp: " << ir.GetNPoints() << "\n";
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
|
||||
bool inactive_derivative = false;
|
||||
|
||||
auto kernel = [] MFEM_HOST_DEVICE(
|
||||
const tensor<double, dim, dim>& J,
|
||||
const double& w,
|
||||
const tensor<double, dim>& dudxi,
|
||||
const double& u)
|
||||
{
|
||||
auto invJ = inv(J);
|
||||
return mfem::tuple{(u * u) * dudxi * invJ * transpose(invJ) * det(J) * w};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators =
|
||||
{
|
||||
Gradient{"coordinates"},
|
||||
Weight{},
|
||||
Gradient{"potential"},
|
||||
Value{"potential"}
|
||||
};
|
||||
|
||||
mfem::tuple output_operator =
|
||||
{
|
||||
Gradient{"potential"}
|
||||
};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = mfem::tuple{eop};
|
||||
|
||||
auto solutions = std::array
|
||||
{
|
||||
FieldDescriptor{&h1fes, "potential"}
|
||||
};
|
||||
auto parameters = std::array
|
||||
{
|
||||
FieldDescriptor{&mesh_fes, "coordinates"}
|
||||
};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector& coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
const double z = coords(2);
|
||||
return 2.345 + 0.25 * x * x * y + y * y * x + z;
|
||||
};
|
||||
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(f1_g), y(h1fes.TrueVSize());
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
y.HostRead();
|
||||
|
||||
ParBilinearForm a(&h1fes);
|
||||
GridFunctionCoefficient f1gc(&f1_g);
|
||||
TransformedCoefficient tf_c(&f1gc, [](double f) { return f * f; });
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator(tf_c));
|
||||
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
a.Assemble();
|
||||
a.Finalize();
|
||||
|
||||
Vector y2(h1fes.TrueVSize()), diff(h1fes.TrueVSize());
|
||||
a.Mult(x, y2);
|
||||
y2.HostRead();
|
||||
diff = y2;
|
||||
diff -= y;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
out << "||F(u) - ex||_l2 = " << diff.Norml2() << "\n";
|
||||
print_vector(diff);
|
||||
print_vector(y);
|
||||
print_vector(y2);
|
||||
return 1;
|
||||
}
|
||||
|
||||
// Test linearization here as well
|
||||
auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {mesh_nodes});
|
||||
dFdu->Mult(x, y);
|
||||
|
||||
// fd jacobian test
|
||||
{
|
||||
double eps = 1.0e-6;
|
||||
Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
|
||||
v *= eps;
|
||||
xpv += v;
|
||||
xmv -= v;
|
||||
dop.Mult(xpv, fxpv);
|
||||
dop.Mult(xmv, fxmv);
|
||||
fxpv -= fxmv;
|
||||
fxpv /= (2.0*eps);
|
||||
|
||||
fxpv -= y;
|
||||
if (fxpv.Norml2() > eps)
|
||||
{
|
||||
out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
// ParBilinearForm da(&h1fes);
|
||||
// TransformedCoefficient dtf_c(&f1gc, [](double f) { return 2.0 * f; });
|
||||
// da.AddDomainIntegrator(new DiffusionIntegrator(dtf_c));
|
||||
// da.SetAssemblyLevel(AssemblyLevel::PARTIAL);
|
||||
// da.Assemble();
|
||||
// da.Finalize();
|
||||
|
||||
// if (dFdu->Height() != h1fes.GetTrueVSize())
|
||||
// {
|
||||
// out << "dFdu unexpected height of " << dFdu->Height() << "\n";
|
||||
// return 1;
|
||||
// }
|
||||
|
||||
// dFdu->Mult(x, y);
|
||||
// print_vector(y);
|
||||
// da.Mult(x, y2);
|
||||
// print_vector(y2);
|
||||
// y2 -= y;
|
||||
// out << "||dFdu x - A x||_l2 = " << y2.Norml2() << "\n";
|
||||
// if (y2.Norml2() > 1e-10)
|
||||
// {
|
||||
// out << "||dFdu u^* - ex||_l2 = " << y2.Norml2() << "\n";
|
||||
// }
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_nonlinear_diffusion);
|
||||
@@ -0,0 +1,268 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
#include "fem/pfespace.hpp"
|
||||
#include "linalg/hypre.hpp"
|
||||
#include "linalg/operator.hpp"
|
||||
#include "linalg/solvers.hpp"
|
||||
#include <fstream>
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
class FDJacobian : public Operator
|
||||
{
|
||||
public:
|
||||
FDJacobian(const Operator &op, const Vector &x) :
|
||||
Operator(op.Height()),
|
||||
op(op),
|
||||
x(x)
|
||||
{
|
||||
f.SetSize(Height());
|
||||
xpev.SetSize(Height());
|
||||
op.Mult(x, f);
|
||||
xnorm = x.Norml2();
|
||||
}
|
||||
|
||||
void Mult(const Vector &v, Vector &y) const override
|
||||
{
|
||||
x.HostRead();
|
||||
|
||||
// See [1] for choice of eps.
|
||||
//
|
||||
// [1] Woodward, C.S., Gardner, D.J. and Evans, K.J., 2015. On the use of
|
||||
// finite difference matrix-vector products in Newton-Krylov solvers for
|
||||
// implicit climate dynamics with spectral elements. Procedia Computer
|
||||
// Science, 51, pp.2036-2045.
|
||||
real_t eps = lambda * (lambda + xnorm / v.Norml2());
|
||||
|
||||
for (int i = 0; i < x.Size(); i++)
|
||||
{
|
||||
xpev(i) = x(i) + eps * v(i);
|
||||
}
|
||||
|
||||
// y = f(x + eps * v)
|
||||
op.Mult(xpev, y);
|
||||
|
||||
// y = (f(x + eps * v) - f(x)) / eps
|
||||
for (int i = 0; i < x.Size(); i++)
|
||||
{
|
||||
y(i) = (y(i) - f(i)) / eps;
|
||||
}
|
||||
}
|
||||
|
||||
virtual MemoryClass GetMemoryClass() const override
|
||||
{
|
||||
return Device::GetDeviceMemoryClass();
|
||||
}
|
||||
|
||||
private:
|
||||
const Operator &op;
|
||||
Vector x, f;
|
||||
mutable Vector xpev;
|
||||
real_t lambda = 1.0e-6;
|
||||
real_t xnorm;
|
||||
};
|
||||
|
||||
template <typename elasticity_t>
|
||||
class ElasticityOperator : public Operator
|
||||
{
|
||||
template <typename elasticity_du_t>
|
||||
class ElasticityJacobianOperator : public Operator
|
||||
{
|
||||
public:
|
||||
ElasticityJacobianOperator(const ElasticityOperator *elasticity,
|
||||
std::shared_ptr<elasticity_du_t> dRdu) :
|
||||
Operator(elasticity->Height()),
|
||||
elasticity(elasticity),
|
||||
dRdu(dRdu),
|
||||
x_ess(dRdu->Height())
|
||||
{
|
||||
}
|
||||
|
||||
void Mult(const Vector &x, Vector &y) const override
|
||||
{
|
||||
x_ess = x;
|
||||
x_ess.SetSubVector(elasticity->ess_tdofs, 0.0);
|
||||
|
||||
dRdu->Mult(x_ess, y);
|
||||
|
||||
for (int i = 0; i < elasticity->ess_tdofs.Size(); i++)
|
||||
{
|
||||
y[elasticity->ess_tdofs[i]] = x[elasticity->ess_tdofs[i]];
|
||||
}
|
||||
}
|
||||
|
||||
const ElasticityOperator *elasticity = nullptr;
|
||||
std::shared_ptr<elasticity_du_t> dRdu;
|
||||
mutable Vector x_ess;
|
||||
};
|
||||
|
||||
public:
|
||||
ElasticityOperator(ParFiniteElementSpace &fes, elasticity_t &elasticity,
|
||||
Array<int> &ess_tdofs) :
|
||||
Operator(fes.GetTrueVSize()),
|
||||
fes(fes),
|
||||
elasticity(elasticity),
|
||||
ess_tdofs(ess_tdofs) {}
|
||||
|
||||
void Mult(const Vector &x, Vector &r) const override
|
||||
{
|
||||
elasticity.Mult(x, r);
|
||||
r.SetSubVector(ess_tdofs, 0.0);
|
||||
}
|
||||
|
||||
Operator &GetGradient(const Vector &x) const override
|
||||
{
|
||||
ParGridFunction u(const_cast<ParFiniteElementSpace *>
|
||||
(*std::get_if<const ParFiniteElementSpace *>
|
||||
(&elasticity.solutions[0].data)));
|
||||
|
||||
u.SetFromTrueDofs(x);
|
||||
auto dRdu = elasticity.template GetDerivativeWrt<0>({&u}, {mesh_nodes});
|
||||
|
||||
jacobian.reset(
|
||||
new ElasticityJacobianOperator<
|
||||
typename std::remove_pointer<decltype(dRdu.get())>::type> (this, dRdu));
|
||||
|
||||
// jacobian.reset(new FDJacobian(*this, x));
|
||||
|
||||
return *jacobian;
|
||||
}
|
||||
|
||||
void SetParameters(ParGridFunction &mesh_nodes)
|
||||
{
|
||||
elasticity.SetParameters({&mesh_nodes});
|
||||
this->mesh_nodes = &mesh_nodes;
|
||||
}
|
||||
|
||||
ParFiniteElementSpace &fes;
|
||||
elasticity_t &elasticity;
|
||||
Array<int> ess_tdofs;
|
||||
mutable ParGridFunction *mesh_nodes = nullptr;
|
||||
mutable std::shared_ptr<Operator> jacobian;
|
||||
};
|
||||
|
||||
int test_nonlinear_elasticity_3d(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
constexpr int dim = 3;
|
||||
constexpr int vdim = dim;
|
||||
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(polynomial_order);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
Array<int> ess_tdof_list, ess_bdr(mesh.bdr_attributes.Max());
|
||||
ess_bdr = 0;
|
||||
ess_bdr[0] = 1;
|
||||
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof_list);
|
||||
|
||||
const IntegrationRule& ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(),
|
||||
h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(0)->GetOrder() + h1fes.GetFE(
|
||||
0)->GetDim() - 1);
|
||||
|
||||
out << "#qp: " << ir.GetNPoints() << "\n";
|
||||
out << "#dof: " << h1fes.GetNDofs() << "\n";
|
||||
|
||||
ParGridFunction u(&h1fes);
|
||||
|
||||
auto elasticity_kernel = [] MFEM_HOST_DEVICE
|
||||
(const tensor<real_t, dim, dim> &dudxi,
|
||||
const tensor<real_t, dim, dim> &J,
|
||||
const double &w)
|
||||
{
|
||||
// shear modulus
|
||||
mfem::real_t D1 = 0.1e6;
|
||||
// bulk modulus
|
||||
mfem::real_t C1 = 1.0e6;
|
||||
constexpr auto I = mfem::internal::IsotropicIdentity<dim>();
|
||||
auto invJ = inv(J);
|
||||
auto dudx = dudxi * invJ;
|
||||
real_t F = det(I + dudx);
|
||||
real_t p = -2.0 * D1 * F * (F - 1);
|
||||
auto devB = dev(dudx + transpose(dudx) + dot(dudx, transpose(dudx)));
|
||||
auto sigma = -(p / F) * I + 2.0 * (C1 / pow(F, 5.0 / 3.0)) * devB;
|
||||
|
||||
return mfem::tuple{sigma * det(J) * w * transpose(invJ)};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators{Gradient{"displacement"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator{Gradient{"displacement"}};
|
||||
|
||||
ElementOperator op{elasticity_kernel, argument_operators, output_operator};
|
||||
|
||||
std::array solutions{FieldDescriptor{&h1fes, "displacement"}};
|
||||
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop{solutions, parameters, mfem::tuple{op}, mesh, ir};
|
||||
|
||||
ElasticityOperator elasticity(h1fes, dop, ess_tdof_list);
|
||||
|
||||
VectorArrayCoefficient f(dim);
|
||||
for (int i = 0; i < dim-1; i++)
|
||||
{
|
||||
f.Set(i, new ConstantCoefficient(0.0));
|
||||
}
|
||||
{
|
||||
Vector pull_force(mesh.bdr_attributes.Max());
|
||||
pull_force = 0.0;
|
||||
pull_force(1) = -1.0e-2;
|
||||
f.Set(dim-1, new PWConstCoefficient(pull_force));
|
||||
}
|
||||
|
||||
ParLinearForm b(&h1fes);
|
||||
b.AddBoundaryIntegrator(new VectorBoundaryLFIntegrator(f));
|
||||
b.UseFastAssembly(true);
|
||||
b.Assemble();
|
||||
auto B = b.ParallelAssemble();
|
||||
|
||||
Vector X = u.GetTrueVector();
|
||||
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetRelTol(1e-8);
|
||||
cg.SetMaxIter(1000);
|
||||
cg.SetPrintLevel(IterativeSolver::PrintLevel().Summary());
|
||||
|
||||
NewtonSolver newton(MPI_COMM_WORLD);
|
||||
newton.SetSolver(cg);
|
||||
newton.SetOperator(elasticity);
|
||||
newton.SetRelTol(1e-6);
|
||||
newton.SetMaxIter(100);
|
||||
newton.SetAdaptiveLinRtol();
|
||||
newton.SetPrintLevel(IterativeSolver::PrintLevel().Iterations());
|
||||
|
||||
elasticity.SetParameters(*mesh_nodes);
|
||||
|
||||
// Vector zero;
|
||||
newton.Mult(*B, X);
|
||||
|
||||
u.SetFromTrueDofs(X);
|
||||
|
||||
ParaViewDataCollection paraview_dc("dfem", &mesh);
|
||||
paraview_dc.SetPrefixPath("ParaView");
|
||||
paraview_dc.SetLevelsOfDetail(polynomial_order);
|
||||
paraview_dc.SetDataFormat(VTKFormat::BINARY);
|
||||
paraview_dc.SetHighOrderOutput(true);
|
||||
paraview_dc.SetCycle(0);
|
||||
paraview_dc.SetTime(0.0);
|
||||
paraview_dc.RegisterField("displacement", &u);
|
||||
paraview_dc.Save();
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_nonlinear_elasticity_3d);
|
||||
@@ -0,0 +1,82 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
#include "fem/coefficient.hpp"
|
||||
#include "fem/pgridfunc.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_ordering(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
constexpr int dim = 2;
|
||||
constexpr int vdim = dim;
|
||||
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(polynomial_order);
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(mesh_fes.GetFE(0)->GetGeomType(),
|
||||
2 * mesh_fes.FEColl()->GetOrder() - 1);
|
||||
|
||||
for (int q = 0; q < ir.GetNPoints(); q++)
|
||||
{
|
||||
out << "(" << ir.IntPoint(q).x << ", " << ir.IntPoint(q).y << ")\n";
|
||||
}
|
||||
|
||||
ParGridFunction u(&mesh_fes);
|
||||
auto f = [](const Vector &coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
u(0) = x*x*y + 1.0;
|
||||
u(1) = y*y*x*x + 2.0;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient uc(dim, f);
|
||||
u.ProjectCoefficient(uc);
|
||||
|
||||
auto kernel = [](const tensor<double, dim> &xi,
|
||||
const tensor<double, vdim, dim> &J,
|
||||
const tensor<double, dim> &u,
|
||||
const tensor<double, vdim, dim> &dudxi)
|
||||
{
|
||||
out << "xi: " << xi << "\n";
|
||||
out << "J: " << J << "\n";
|
||||
out << "u: " << u << "\n";
|
||||
out << "dudxi: " << dudxi << "\n\n";
|
||||
return mfem::tuple{J};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators{Value{"coordinates"}, Gradient{"coordinates"}, Value{"potential"}, Gradient{"potential"}};
|
||||
mfem::tuple output_operator{Gradient{"potential"}};
|
||||
|
||||
ElementOperator op{kernel, argument_operators, output_operator};
|
||||
|
||||
std::array solutions{FieldDescriptor{&mesh_fes, "potential"}};
|
||||
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop{solutions, parameters, mfem::tuple{op}, mesh, ir};
|
||||
|
||||
Vector y(u);
|
||||
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(u, y);
|
||||
|
||||
print_vector(y);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_ordering);
|
||||
@@ -0,0 +1,102 @@
|
||||
#include "dfem/dfem.hpp"
|
||||
#include "dfem/dfem_test_macro.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
|
||||
int test_vector_diffusion(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
const int vdim = dim;
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, vdim);
|
||||
|
||||
Array<int> ess_bdr(mesh.bdr_attributes.Max());
|
||||
Array<int> ess_tdof;
|
||||
ess_bdr = 1;
|
||||
h1fes.GetEssentialTrueDofs(ess_bdr, ess_tdof);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder() - 1);
|
||||
|
||||
ParGridFunction u(&h1fes);
|
||||
|
||||
auto f1 = [](const Vector& coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
u(0) = 2.345 + 0.25 * x * x * y + y * y * x;
|
||||
u(1) = 2.345 - 0.25 * x * y * y + y * x * x;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient u_c(dim, f1);
|
||||
u.ProjectCoefficient(u_c);
|
||||
|
||||
auto vector_diffusion_kernel = [](const tensor<double, 2> &xi,
|
||||
const tensor<double, 2, 2> &dudxi,
|
||||
const tensor<double, 2, 2> &J,
|
||||
const double &w)
|
||||
{
|
||||
out << "xi: " << xi << "\n";
|
||||
out << "dudxi: " << dudxi << "\n";
|
||||
return mfem::tuple{dudxi * inv(J) * det(J) * w * transpose(inv(J))};
|
||||
// return mfem::tuple{dudxi};
|
||||
};
|
||||
|
||||
mfem::tuple argument_operators{Value{"coordinates"}, Gradient{"potential"}, Gradient{"coordinates"}, Weight{}};
|
||||
mfem::tuple output_operator{Gradient{"potential"}};
|
||||
|
||||
ElementOperator op{vector_diffusion_kernel, argument_operators, output_operator};
|
||||
|
||||
std::array solutions{FieldDescriptor{&h1fes, "potential"}};
|
||||
std::array parameters{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop{solutions, parameters, mfem::tuple{op}, mesh, ir};
|
||||
|
||||
Vector x(u), y1(h1fes.GetTrueVSize()),
|
||||
y2(h1fes.GetTrueVSize());
|
||||
|
||||
ParBilinearForm A_form(&h1fes);
|
||||
auto A_integ = new VectorDiffusionIntegrator(vdim);
|
||||
A_integ->SetIntegrationRule(ir);
|
||||
A_form.AddDomainIntegrator(A_integ);
|
||||
A_form.Assemble();
|
||||
A_form.Finalize();
|
||||
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y1);
|
||||
y1.HostRead();
|
||||
|
||||
A_form.Mult(x, y2);
|
||||
y2.HostRead();
|
||||
|
||||
Vector diff(y2);
|
||||
diff -= y1;
|
||||
if (diff.Norml2() > 1e-10)
|
||||
{
|
||||
out << "||F(u) - ex||_l2 = " << diff.Norml2() << "\n";
|
||||
print_vector(diff);
|
||||
print_vector(y1);
|
||||
print_vector(y2);
|
||||
return 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
DFEM_TEST_MAIN(test_vector_diffusion);
|
||||
@@ -0,0 +1,122 @@
|
||||
#include <tuple>
|
||||
#include <type_traits>
|
||||
#include <iostream>
|
||||
#include <enzyme/enzyme>
|
||||
|
||||
template <typename T>
|
||||
constexpr auto get_type_name() -> std::string_view
|
||||
{
|
||||
#if defined(__clang__)
|
||||
constexpr auto prefix = std::string_view {"[T = "};
|
||||
constexpr auto suffix = "]";
|
||||
constexpr auto function = std::string_view{__PRETTY_FUNCTION__};
|
||||
#elif defined(__GNUC__)
|
||||
constexpr auto prefix = std::string_view {"with T = "};
|
||||
constexpr auto suffix = "; ";
|
||||
constexpr auto function = std::string_view{__PRETTY_FUNCTION__};
|
||||
#elif defined(_MSC_VER)
|
||||
constexpr auto prefix = std::string_view {"get_type_name<"};
|
||||
constexpr auto suffix = ">(void)";
|
||||
constexpr auto function = std::string_view{__FUNCSIG__};
|
||||
#else
|
||||
#error Unsupported compiler
|
||||
#endif
|
||||
|
||||
const auto start = function.find(prefix) + prefix.size();
|
||||
const auto end = function.find(suffix);
|
||||
const auto size = end - start;
|
||||
|
||||
return function.substr(start, size);
|
||||
}
|
||||
|
||||
template <typename ... Ts>
|
||||
constexpr auto decay_types(std::tuple<Ts...> const &)
|
||||
-> std::tuple<std::remove_cv_t<std::remove_reference_t<Ts>>...>;
|
||||
|
||||
template <typename T>
|
||||
using decay_tuple = decltype(decay_types(std::declval<T>()));
|
||||
|
||||
template <class F> struct FunctionSignature;
|
||||
|
||||
template <typename output_t, typename... input_ts>
|
||||
struct FunctionSignature<output_t(input_ts...)>
|
||||
{
|
||||
using return_t = output_t;
|
||||
using parameter_ts = std::tuple<input_ts...>;
|
||||
};
|
||||
|
||||
template <class T> struct create_function_signature;
|
||||
|
||||
template <typename output_t, typename T, typename... input_ts>
|
||||
struct create_function_signature<output_t (T::*)(input_ts...) const>
|
||||
{
|
||||
using type = FunctionSignature<output_t(input_ts...)>;
|
||||
};
|
||||
|
||||
template <typename arg_ts, std::size_t... Is>
|
||||
auto create_enzyme_args(arg_ts &args,
|
||||
arg_ts &shadow_args,
|
||||
std::index_sequence<Is...>)
|
||||
{
|
||||
// (std::cout << ... << std::get<Is>(shadow_args));
|
||||
return std::tuple<enzyme::Duplicated<decltype(std::get<Is>(args))>...>
|
||||
{
|
||||
{ std::get<Is>(args), std::get<Is>(shadow_args) }...
|
||||
};
|
||||
}
|
||||
|
||||
template <typename kernel_t, typename arg_ts>
|
||||
auto fwddiff_apply_enzyme(kernel_t kernel, arg_ts &&args, arg_ts &&shadow_args)
|
||||
{
|
||||
auto arg_indices =
|
||||
std::make_index_sequence<std::tuple_size_v<std::remove_reference_t<arg_ts>>> {};
|
||||
|
||||
auto enzyme_args = create_enzyme_args(args, shadow_args, arg_indices);
|
||||
|
||||
// using kf_return_t = typename create_function_signature<
|
||||
// decltype(&kernel_t::operator())>::type::return_t;
|
||||
|
||||
std::cout << "\n";
|
||||
std::cout << "args is " << get_type_name<decltype(args)>() << "\n\n";
|
||||
std::cout << "enzyme_args type is " << get_type_name<decltype(enzyme_args)>() <<
|
||||
"\n\n";
|
||||
// std::cout << "return type is " << get_type_name<decltype(kf_return_t{})>() <<
|
||||
// "\n\n";
|
||||
|
||||
std::cout << "args " << std::get<0>(args) << "\n";
|
||||
std::cout << "shadow args " << std::get<0>(shadow_args) << "\n";
|
||||
|
||||
return std::apply([&](auto &&...args)
|
||||
{
|
||||
// std::cout << enzyme::autodiff<enzyme::Forward>(+kernel, args...) << "\n";
|
||||
return enzyme::get<0>
|
||||
(enzyme::autodiff<enzyme::Forward>(+kernel, args...));
|
||||
},
|
||||
enzyme_args);
|
||||
}
|
||||
|
||||
int main()
|
||||
{
|
||||
|
||||
auto func = [](const double &x, double &y)
|
||||
{
|
||||
std::cout << "func( x = " << x << " )\n";
|
||||
return x*x;
|
||||
};
|
||||
|
||||
using kf_param_ts = typename create_function_signature<
|
||||
decltype(&decltype(func)::operator())>::type::parameter_ts;
|
||||
using kf_output_t = typename create_function_signature<
|
||||
decltype(&decltype(func)::operator())>::type::return_t;
|
||||
auto kernel_args = decay_tuple<kf_param_ts> {};
|
||||
auto kernel_shadow_args = decay_tuple<kf_param_ts> {};
|
||||
|
||||
std::get<0>(kernel_args) = 3;
|
||||
std::get<0>(kernel_shadow_args) = 1;
|
||||
|
||||
auto dx = fwddiff_apply_enzyme(func, kernel_args, kernel_shadow_args);
|
||||
|
||||
std::cout << "dfdx = " << dx << "\n";
|
||||
|
||||
return 0;
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,305 @@
|
||||
#include "dfem/dfem_refactor.hpp"
|
||||
#include "linalg/hypre.hpp"
|
||||
|
||||
using namespace mfem;
|
||||
using mfem::internal::tensor;
|
||||
using mfem::internal::dual;
|
||||
|
||||
int test_diffusion_integrator(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder());
|
||||
|
||||
ParGridFunction f1_g(&h1fes);
|
||||
ParGridFunction rho_g(&h1fes);
|
||||
|
||||
auto rho_f = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
return x + y;
|
||||
};
|
||||
|
||||
FunctionCoefficient rho_c(rho_f);
|
||||
rho_g.ProjectCoefficient(rho_c);
|
||||
|
||||
auto kernel = [](const tensor<dual<double, double>, 2> &grad_u,
|
||||
const dual<double, double> &rho,
|
||||
const tensor<double, 2, 2> &J,
|
||||
const double &w)
|
||||
{
|
||||
auto invJ = inv(J);
|
||||
return std::tuple{rho*rho * grad_u * invJ * transpose(invJ) * det(J) * w};
|
||||
};
|
||||
|
||||
std::tuple argument_operators = {Gradient{"potential"}, Value{"density"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
|
||||
std::tuple output_operator = {Gradient{"potential"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = std::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "potential"}};
|
||||
auto parameters = std::array
|
||||
{
|
||||
FieldDescriptor{&h1fes, "density"},
|
||||
FieldDescriptor{&mesh_fes, "coordinates"}
|
||||
};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
auto f1 = [](const Vector &coords)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
return 2.345 + 0.25 * x*x*y + y*y*x;
|
||||
};
|
||||
|
||||
FunctionCoefficient f1_c(f1);
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
|
||||
Vector x(f1_g), y(h1fes.TrueVSize());
|
||||
dop.SetParameters({&rho_g, mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
ParBilinearForm a(&h1fes);
|
||||
TransformedCoefficient rho_c2(&rho_c, [](double c) {return c*c;});
|
||||
a.AddDomainIntegrator(new DiffusionIntegrator(rho_c2));
|
||||
a.Assemble();
|
||||
a.Finalize();
|
||||
|
||||
Vector y2(h1fes.TrueVSize());
|
||||
a.Mult(x, y2);
|
||||
y2 -= y;
|
||||
if (y2.Norml2() > 1e-10)
|
||||
{
|
||||
out << "||F(u) - ex||_l2 = " << y2.Norml2() << "\n";
|
||||
return 1;
|
||||
}
|
||||
|
||||
// Test linearization here as well
|
||||
auto dFdu = dop.GetDerivativeWrt<0>({&f1_g}, {&rho_g, mesh_nodes});
|
||||
|
||||
// HypreParMatrix A;
|
||||
// dFdu->Assemble(A);
|
||||
|
||||
if (dFdu->Height() != h1fes.GetTrueVSize())
|
||||
{
|
||||
out << "dFdu unexpected height of " << dFdu->Height() << "\n";
|
||||
return 1;
|
||||
}
|
||||
|
||||
dFdu->Mult(x, y);
|
||||
a.Mult(x, y2);
|
||||
y2 -= y;
|
||||
if (y2.Norml2() > 1e-10)
|
||||
{
|
||||
out << "||dFdu u^* - ex||_l2 = " << y2.Norml2() << "\n";
|
||||
return 1;
|
||||
}
|
||||
|
||||
// fd jacobian test
|
||||
{
|
||||
double eps = 1.0e-6;
|
||||
Vector v(x), xpv(x), xmv(x), fxpv(x.Size()), fxmv(x.Size());
|
||||
v *= eps;
|
||||
xpv += v;
|
||||
xmv -= v;
|
||||
dop.Mult(xpv, fxpv);
|
||||
dop.Mult(xmv, fxmv);
|
||||
fxpv -= fxmv;
|
||||
fxpv /= (2.0*eps);
|
||||
|
||||
fxpv -= y;
|
||||
if (fxpv.Norml2() > eps)
|
||||
{
|
||||
out << "||dFdu_FD u^* - ex||_l2 = " << fxpv.Norml2() << "\n";
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
f1_g.ProjectCoefficient(f1_c);
|
||||
rho_g.ProjectCoefficient(rho_c);
|
||||
auto dFdrho = dop.GetDerivativeWrt<1>({&f1_g}, {&rho_g, mesh_nodes});
|
||||
if (dFdrho->Height() != h1fes.GetTrueVSize())
|
||||
{
|
||||
out << "dFdrho unexpected height of " << dFdrho->Height() << "\n";
|
||||
return 1;
|
||||
}
|
||||
|
||||
dFdrho->Mult(rho_g, y);
|
||||
|
||||
// fd test
|
||||
{
|
||||
double eps = 1.0e-6;
|
||||
Vector v(rho_g), rhopv(rho_g), rhomv(rho_g), frhopv(x.Size()), frhomv(x.Size());
|
||||
v *= eps;
|
||||
rhopv += v;
|
||||
rhomv -= v;
|
||||
dop.SetParameters({&rhopv, mesh_nodes});
|
||||
dop.Mult(x, frhopv);
|
||||
dop.SetParameters({&rhomv, mesh_nodes});
|
||||
dop.Mult(x, frhomv);
|
||||
frhopv -= frhomv;
|
||||
frhopv /= (2.0*eps);
|
||||
|
||||
frhopv -= y;
|
||||
if (frhopv.Norml2() > eps)
|
||||
{
|
||||
out << "||dFdu_FD u^* - ex||_l2 = " << frhopv.Norml2() << "\n";
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int test_qoi(std::string mesh_file,
|
||||
int refinements,
|
||||
int polynomial_order)
|
||||
{
|
||||
Mesh mesh_serial = Mesh(mesh_file);
|
||||
for (int i = 0; i < refinements; i++)
|
||||
{
|
||||
mesh_serial.UniformRefinement();
|
||||
}
|
||||
ParMesh mesh(MPI_COMM_WORLD, mesh_serial);
|
||||
|
||||
mesh.SetCurvature(1);
|
||||
const int dim = mesh.Dimension();
|
||||
mesh_serial.Clear();
|
||||
|
||||
ParGridFunction* mesh_nodes = static_cast<ParGridFunction *>(mesh.GetNodes());
|
||||
ParFiniteElementSpace &mesh_fes = *mesh_nodes->ParFESpace();
|
||||
|
||||
H1_FECollection h1fec(polynomial_order, dim);
|
||||
ParFiniteElementSpace h1fes(&mesh, &h1fec, dim);
|
||||
|
||||
const IntegrationRule &ir =
|
||||
IntRules.Get(h1fes.GetFE(0)->GetGeomType(), 2 * h1fec.GetOrder());
|
||||
|
||||
ParGridFunction rho_g(&h1fes);
|
||||
|
||||
auto rho_f = [](const Vector &coords, Vector &u)
|
||||
{
|
||||
const double x = coords(0);
|
||||
const double y = coords(1);
|
||||
u(0) = x + y;
|
||||
u(1) = x + y;
|
||||
};
|
||||
|
||||
VectorFunctionCoefficient rho_c(dim, rho_f);
|
||||
rho_g.ProjectCoefficient(rho_c);
|
||||
|
||||
auto kernel = [](const tensor<dual<double, double>, 2> &rho,
|
||||
const tensor<dual<double, double>, 2, 2> &drhodxi,
|
||||
const tensor<double, 2, 2> &J,
|
||||
const double &w)
|
||||
{
|
||||
const double eps = 1.2345;
|
||||
const auto drhodx = drhodxi * inv(J);
|
||||
return std::tuple{(0.5 * eps * dot(rho, rho) + ddot(drhodx, drhodx)) * det(J) * w};
|
||||
};
|
||||
|
||||
std::tuple argument_operators = {Value{"density"}, Gradient{"density"}, Gradient{"coordinates"}, Weight{"integration_weights"}};
|
||||
std::tuple output_operator = {One{"density"}};
|
||||
|
||||
ElementOperator eop = {kernel, argument_operators, output_operator};
|
||||
auto ops = std::tuple{eop};
|
||||
|
||||
auto solutions = std::array{FieldDescriptor{&h1fes, "density"}};
|
||||
auto parameters = std::array{FieldDescriptor{&mesh_fes, "coordinates"}};
|
||||
|
||||
DifferentiableOperator dop(solutions, parameters, ops, mesh, ir);
|
||||
|
||||
Vector x(rho_g), y(1);
|
||||
dop.SetParameters({mesh_nodes});
|
||||
dop.Mult(x, y);
|
||||
|
||||
// print_vector(y);
|
||||
|
||||
auto dFdrho = dop.GetDerivativeWrt<0>({&rho_g}, {mesh_nodes});
|
||||
// Vector dFdrho_vec;
|
||||
// dFdrho->Assemble(dFdrho_vec);
|
||||
|
||||
// print_vector(dFdrho_vec);
|
||||
|
||||
// fd jacobian test
|
||||
{
|
||||
double eps = 1.0e-8;
|
||||
Vector v(x), fxpv(1), fxmv(1), dfdx(x.Size());
|
||||
for (int i = 0; i < x.Size(); i++)
|
||||
{
|
||||
v(i) += eps;
|
||||
dop.Mult(v, fxpv);
|
||||
v(i) -= 2.0 * eps;
|
||||
dop.Mult(v, fxmv);
|
||||
fxpv -= fxmv;
|
||||
fxpv /= (2.0*eps);
|
||||
dfdx(i) = fxpv(0);
|
||||
}
|
||||
|
||||
// print_vector(dfdx);
|
||||
dfdx -= dFdrho_vec;
|
||||
if (dfdx.Norml2() > 1e-6)
|
||||
{
|
||||
out << "||dFdu_FD u^* - ex||_l2 = " << dfdx.Norml2() << "\n";
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
Mpi::Init();
|
||||
|
||||
std::cout << std::setprecision(9);
|
||||
|
||||
const char *mesh_file = "../data/star.mesh";
|
||||
int polynomial_order = 1;
|
||||
int ir_order = 2;
|
||||
int refinements = 0;
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
|
||||
args.AddOption(&polynomial_order, "-o", "--order", "");
|
||||
args.AddOption(&refinements, "-r", "--r", "");
|
||||
args.AddOption(&ir_order, "-iro", "--iro", "");
|
||||
args.ParseCheck();
|
||||
|
||||
out << std::setprecision(12);
|
||||
|
||||
int ret;
|
||||
|
||||
ret = test_diffusion_integrator(mesh_file,
|
||||
refinements,
|
||||
polynomial_order);
|
||||
out << "test_diffusion_integrator";
|
||||
ret ? out << " FAILURE\n" : out << " OK\n";
|
||||
|
||||
ret = test_qoi(mesh_file, refinements, polynomial_order);
|
||||
out << "test_qoi";
|
||||
ret ? out << " FAILURE\n" : out << " OK\n";
|
||||
|
||||
return 0;
|
||||
}
|
||||
+8
-8
@@ -21,16 +21,16 @@
|
||||
* documentation (https://enzyme.mit.edu) for more information.
|
||||
*/
|
||||
|
||||
extern int enzyme_dup;
|
||||
extern int enzyme_dupnoneed;
|
||||
extern int enzyme_out;
|
||||
extern int enzyme_const;
|
||||
// extern int enzyme_dup;
|
||||
// extern int enzyme_dupnoneed;
|
||||
// extern int enzyme_out;
|
||||
// extern int enzyme_const;
|
||||
|
||||
template <typename return_type, typename... Args>
|
||||
return_type __enzyme_autodiff(Args...);
|
||||
// template <typename return_type, typename... Args>
|
||||
// return_type __enzyme_autodiff(Args...);
|
||||
|
||||
template <typename return_type, typename... Args>
|
||||
return_type __enzyme_fwddiff(Args...);
|
||||
// template <typename return_type, typename... Args>
|
||||
// return_type __enzyme_fwddiff(Args...);
|
||||
|
||||
#define MFEM_ENZYME_INACTIVENOFREE __attribute__((enzyme_inactive, enzyme_nofree))
|
||||
#define MFEM_ENZYME_INACTIVE __attribute__((enzyme_inactive))
|
||||
|
||||
@@ -871,6 +871,8 @@ public:
|
||||
inline static void* __enzyme_allocation_like2[4] = {(void*)static_cast<void*(*)(void*, size_t, MemoryType, MemoryType, unsigned, unsigned&)>(MemoryManager::New_),
|
||||
(void*)1, (void*)"-1,2,4", (void*)MemoryManager::Delete_
|
||||
};
|
||||
__attribute__((used))
|
||||
inline static void* __enzyme_function_like[2] = {(void*)MemoryManager::Delete_, (void*)"free"};
|
||||
#endif
|
||||
};
|
||||
|
||||
|
||||
+13
-1
@@ -87,7 +87,9 @@ protected:
|
||||
|
||||
public:
|
||||
/// Default constructor
|
||||
DeviceTensor() = delete;
|
||||
// DeviceTensor() = delete;
|
||||
MFEM_HOST_DEVICE
|
||||
DeviceTensor() {}
|
||||
|
||||
/// Constructor to initialize a tensor from the Scalar array data_
|
||||
template <typename... Args> MFEM_HOST_DEVICE
|
||||
@@ -122,6 +124,16 @@ public:
|
||||
{
|
||||
return data[i];
|
||||
}
|
||||
|
||||
MFEM_HOST_DEVICE inline std::array<int, Dim> GetShape() const
|
||||
{
|
||||
std::array<int, Dim> s;
|
||||
for (int i = 0; i < Dim; i++)
|
||||
{
|
||||
s[i] = sizes[i];
|
||||
}
|
||||
return s;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -19,6 +19,8 @@
|
||||
#define MFEM_INTERNAL_TENSOR_HPP
|
||||
|
||||
#include "dual.hpp"
|
||||
#include "general/backends.hpp"
|
||||
#include <limits>
|
||||
#include <type_traits> // for std::false_type
|
||||
|
||||
namespace mfem
|
||||
@@ -436,6 +438,15 @@ tensor<decltype(f(n1, n2, n3, n4)), n1, n2, n3, n4>
|
||||
return A;
|
||||
}
|
||||
|
||||
template <typename T, int n1, int n2> MFEM_HOST_DEVICE
|
||||
tensor<T, n2> get_col(tensor<T, n1, n2> A, int j)
|
||||
{
|
||||
tensor<T, n2> c{};
|
||||
c(0) = A(0, j);
|
||||
c(1) = A(1, j);
|
||||
return c;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief return the sum of two tensors
|
||||
* @tparam S the underlying type of the lefthand argument
|
||||
@@ -697,6 +708,20 @@ auto outer(S A, T B) -> decltype(A * B)
|
||||
return A * B;
|
||||
}
|
||||
|
||||
template <typename T, int n, int m> MFEM_HOST_DEVICE
|
||||
tensor<T, n + m> flatten(tensor<T, n, m> A)
|
||||
{
|
||||
tensor<T, n + m> B{};
|
||||
for (int i = 0; i < n; i++)
|
||||
{
|
||||
for (int j = 0; j < m; j++)
|
||||
{
|
||||
B(i + j * m) = A(i, j);
|
||||
}
|
||||
}
|
||||
return B;
|
||||
}
|
||||
|
||||
/**
|
||||
* @overload
|
||||
* @note this overload implements the case where the left argument is a scalar, and the right argument is a tensor
|
||||
@@ -1051,6 +1076,18 @@ decltype(S {} * T{})
|
||||
return AB;
|
||||
}
|
||||
|
||||
template <typename T, int m> MFEM_HOST_DEVICE
|
||||
auto dot(const tensor<T, m>& A, const tensor<T, m>& B) ->
|
||||
decltype(T {})
|
||||
{
|
||||
decltype(T{}) AB{};
|
||||
for (int i = 0; i < m; i++)
|
||||
{
|
||||
AB += A[i] * B[i];
|
||||
}
|
||||
return AB;
|
||||
}
|
||||
|
||||
template <typename S, typename T, int m, int... n> MFEM_HOST_DEVICE
|
||||
auto dot(const tensor<S, m>& A, const tensor<T, m, n...>& B) ->
|
||||
tensor<decltype(S {} * T{}), n...>
|
||||
@@ -1335,6 +1372,133 @@ T det(const tensor<T, 3, 3>& A)
|
||||
A[2][0];
|
||||
}
|
||||
|
||||
template <typename T> MFEM_HOST_DEVICE
|
||||
std::tuple<tensor<T, 2>, tensor<T, 2, 2>> eig(tensor<T, 2, 2> &A)
|
||||
{
|
||||
tensor<T, 2> e;
|
||||
tensor<T, 2, 2> v;
|
||||
|
||||
double d0 = A(0, 0);
|
||||
double d2 = A(0, 1);
|
||||
double d3 = A(1, 1);
|
||||
double c, s;
|
||||
|
||||
if (d2 == 0.0)
|
||||
{
|
||||
c = 1.0;
|
||||
s = 0.0;
|
||||
}
|
||||
else
|
||||
{
|
||||
double t;
|
||||
const double zeta = (d3 - d0) / (2.0 * d2);
|
||||
const double azeta = fabs(zeta);
|
||||
if (azeta < std::sqrt(1.0/std::numeric_limits<T>::epsilon()))
|
||||
{
|
||||
t = copysign(1./(azeta + std::sqrt(1. + zeta*zeta)), zeta);
|
||||
}
|
||||
else
|
||||
{
|
||||
t = copysign(0.5/azeta, zeta);
|
||||
}
|
||||
c = std::sqrt(1./(1. + t*t));
|
||||
s = c*t;
|
||||
t *= d2;
|
||||
d0 -= t;
|
||||
d3 += t;
|
||||
}
|
||||
|
||||
if (d0 <= d3)
|
||||
{
|
||||
e(0) = d0;
|
||||
e(1) = d3;
|
||||
v(0, 0) = c;
|
||||
v(1, 0) = -s;
|
||||
v(0, 1) = s;
|
||||
v(1, 1) = c;
|
||||
}
|
||||
else
|
||||
{
|
||||
e(0) = d3;
|
||||
e(1) = d0;
|
||||
v(0, 0) = s;
|
||||
v(1, 0) = c;
|
||||
v(0, 1) = c;
|
||||
v(1, 1) = -s;
|
||||
}
|
||||
|
||||
return {e, v};
|
||||
}
|
||||
|
||||
template <typename T> MFEM_HOST_DEVICE
|
||||
void GetScalingFactor(const T &d_max, T &mult)
|
||||
{
|
||||
int d_exp;
|
||||
if (d_max > 0.)
|
||||
{
|
||||
mult = frexp(d_max, &d_exp);
|
||||
if (d_exp == std::numeric_limits<T>::max_exponent)
|
||||
{
|
||||
mult *= std::numeric_limits<T>::radix;
|
||||
}
|
||||
mult = d_max/mult;
|
||||
}
|
||||
else
|
||||
{
|
||||
mult = 1.;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Compute the i-th singular value of a 2x2 matrix A
|
||||
*/
|
||||
template <typename T> MFEM_HOST_DEVICE
|
||||
T calcsv(const tensor<T, 2, 2> A, const int i)
|
||||
{
|
||||
double mult;
|
||||
double d0, d1, d2, d3;
|
||||
d0 = A(0, 0);
|
||||
d1 = A(1, 0);
|
||||
d2 = A(0, 1);
|
||||
d3 = A(1, 1);
|
||||
|
||||
double d_max = fabs(d0);
|
||||
if (d_max < fabs(d1)) { d_max = fabs(d1); }
|
||||
if (d_max < fabs(d2)) { d_max = fabs(d2); }
|
||||
if (d_max < fabs(d3)) { d_max = fabs(d3); }
|
||||
|
||||
GetScalingFactor(d_max, mult);
|
||||
|
||||
d0 /= mult;
|
||||
d1 /= mult;
|
||||
d2 /= mult;
|
||||
d3 /= mult;
|
||||
|
||||
double t = 0.5*((d0+d2)*(d0-d2)+(d1-d3)*(d1+d3));
|
||||
double s = d0*d2 + d1*d3;
|
||||
s = std::sqrt(0.5*(d0*d0 + d1*d1 + d2*d2 + d3*d3) + std::sqrt(t*t + s*s));
|
||||
|
||||
if (s == 0.0)
|
||||
{
|
||||
return 0.0;
|
||||
}
|
||||
t = fabs(d0*d3 - d1*d2) / s;
|
||||
if (t > s)
|
||||
{
|
||||
if (i == 0)
|
||||
{
|
||||
return t*mult;
|
||||
}
|
||||
return s*mult;
|
||||
}
|
||||
if (i == 0)
|
||||
{
|
||||
return s*mult;
|
||||
}
|
||||
return t*mult;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* @brief Return whether a square rank 2 tensor is symmetric
|
||||
*
|
||||
|
||||
@@ -0,0 +1,190 @@
|
||||
using LinearAlgebra
|
||||
using Tullio
|
||||
|
||||
num_qp = 4
|
||||
num_trial_dof = 4;
|
||||
trial_vdim = 1
|
||||
trial_op_dim = [1, 2]
|
||||
test_op_dim = 2
|
||||
test_vdim = 1
|
||||
num_test_dof = num_trial_dof
|
||||
num_rho_dof = 4;
|
||||
space_dim = 2;
|
||||
|
||||
Bu_mem = [0.622008468 0.166666667 0.166666667 0.0446581987 0.166666667 0.622008468 0.0446581987 0.166666667 0.0446581987 0.166666667 0.166666667 0.622008468 0.166666667 0.0446581987 0.622008468 0.166666667]
|
||||
|
||||
Bdu_mem = [-0.788675135 -0.788675135 -0.211324865 -0.211324865 -0.788675135 -0.211324865 -0.788675135 -0.211324865 0.788675135 0.788675135 0.211324865 0.211324865 -0.211324865 -0.788675135 -0.211324865 -0.788675135 0.211324865 0.211324865 0.788675135 0.788675135 0.211324865 0.788675135 0.211324865 0.788675135 -0.211324865 -0.211324865 -0.788675135 -0.788675135 0.788675135 0.211324865 0.788675135 0.211324865]
|
||||
|
||||
Bu = reshape(Bu_mem, (num_qp, trial_op_dim[1], num_trial_dof))
|
||||
Bdu = reshape(Bdu_mem, (num_qp, trial_op_dim[2], num_trial_dof))
|
||||
Bv = Bdu;
|
||||
|
||||
u_e = reshape([2.345 2.345 3.595 2.345], (num_trial_dof, trial_vdim))
|
||||
rho_e = reshape([0.0 1.0 2.0 1.0], (num_trial_dof, trial_vdim))
|
||||
x_e = reshape([0.0 1.0 1.0 0.0 0.0 0.0 1.0 1.0], (num_trial_dof, space_dim))
|
||||
|
||||
rho_qpref = [0.422650 1.000000 1.000000 1.577350]
|
||||
J_qpref = [1.000000 0.000000 0.000000 1.000000 1.000000 0.000000 0.000000 1.000000 1.000000 0.000000 0.000000 1.000000 1.000000 0.000000 0.000000 1.000000]
|
||||
w_qpref = [0.250000 0.250000 0.250000 0.250000]
|
||||
dudxi_qpref = [0.264156 0.264156 0.264156 0.985844 0.985844 0.264156 0.985844 0.985844]
|
||||
|
||||
rho_qp = zeros(Float64, trial_vdim, num_qp)
|
||||
for v = 1:trial_vdim
|
||||
for q = 1:num_qp
|
||||
acc = 0.0
|
||||
for d = 1:num_trial_dof
|
||||
acc += Bu[q, v, d] * rho_e[d, v]
|
||||
end
|
||||
rho_qp[v, q] = acc
|
||||
end
|
||||
end
|
||||
println(rho_qp)
|
||||
|
||||
J_qp = zeros(Float64, space_dim, space_dim, num_qp)
|
||||
for v = 1:space_dim
|
||||
for s = 1:space_dim
|
||||
for q = 1:num_qp
|
||||
acc = 0.0
|
||||
for d = 1:num_trial_dof
|
||||
acc += Bdu[q, s, d] * x_e[d, v]
|
||||
end
|
||||
J_qp[v, s, q] = acc
|
||||
end
|
||||
end
|
||||
end
|
||||
println(J_qp)
|
||||
|
||||
dudxi_qp = zeros(Float64, trial_vdim, space_dim, num_qp)
|
||||
for v = 1:trial_vdim
|
||||
for s = 1:space_dim
|
||||
for q = 1:num_qp
|
||||
acc = 0.0
|
||||
for d = 1:num_trial_dof
|
||||
acc += Bdu[q, s, d] * u_e[d, v]
|
||||
end
|
||||
dudxi_qp[v, s, q] = acc
|
||||
end
|
||||
end
|
||||
end
|
||||
println(dudxi_qp)
|
||||
|
||||
w_qp = w_qpref
|
||||
|
||||
# sum factorization
|
||||
|
||||
u_e = reshape(Float64[2.345 2.345 2.345 3.595], (num_trial_dof, trial_vdim))
|
||||
rho_e = reshape(Float64[0 1 1 2], (num_trial_dof, trial_vdim))
|
||||
x_e = reshape(Float64[0 1 0 1 0 0 1 1], (num_trial_dof, space_dim))
|
||||
|
||||
nq1d = 2
|
||||
nd1d = 2
|
||||
|
||||
B = reshape([0.788675135 0.211324865 0.211324865 0.788675135], (nq1d, nd1d))
|
||||
G = reshape([-1 -1 1 1], (nq1d, nd1d))
|
||||
|
||||
function interpolate_value(u_e)
|
||||
u_e = reshape(u_e, (nd1d, nd1d))
|
||||
S2 = zeros(Float64, nq1d, nq1d)
|
||||
for v = 1:trial_vdim
|
||||
@tullio S1[qx, dy] := u_e[dx, dy] * B[qx, dx]
|
||||
@tullio S2[qx, qy] = B[qy, dy] * S1[qx, dy]
|
||||
end
|
||||
return S2
|
||||
end
|
||||
|
||||
function interpolate_grad(u_e)
|
||||
vdim = size(u_e, 2)
|
||||
u_e = reshape(u_e, (nd1d, nd1d, vdim))
|
||||
dq0 = zeros(Float64, nd1d, nq1d)
|
||||
dq1 = zeros(Float64, nd1d, nq1d)
|
||||
dudxi_qp = zeros(nq1d, nq1d, vdim, space_dim)
|
||||
|
||||
for vd = 1:vdim
|
||||
for dy = 1:nd1d
|
||||
for qx = 1:nq1d
|
||||
u = 0.0
|
||||
v = 0.0
|
||||
for dx = 1:nd1d
|
||||
u += u_e[dx, dy, vd] * B[qx, dx]
|
||||
v += u_e[dx, dy, vd] * G[qx, dx]
|
||||
end
|
||||
dq0[dy, qx] = u
|
||||
dq1[dy, qx] = v
|
||||
end
|
||||
end
|
||||
|
||||
for qy = 1:nq1d
|
||||
for qx = 1:nq1d
|
||||
du = [0.0, 0.0]
|
||||
for dy = 1:nd1d
|
||||
du[1] += dq1[dy, qx] * B[qy, dy]
|
||||
du[2] += dq0[dy, qx] * G[qy, dy]
|
||||
end
|
||||
|
||||
for s = 1:space_dim
|
||||
dudxi_qp[qx, qy, vd, s] = du[s]
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
return dudxi_qp
|
||||
end
|
||||
|
||||
function integrate_grad(f_qp, r)
|
||||
dq0 = zeros(Float64, nd1d, nq1d)
|
||||
dq1 = zeros(Float64, nd1d, nq1d)
|
||||
for qy = 1:nq1d
|
||||
for dx = 1:nd1d
|
||||
u = 0.0
|
||||
v = 0.0
|
||||
for qx = 1:nq1d
|
||||
u += G[qx, dx] * f_qp[qx, qy, 1, 1]
|
||||
v += B[qx, dx] * f_qp[qx, qy, 1, 2]
|
||||
end
|
||||
dq0[dx, qy] = u
|
||||
dq1[dx, qy] = v
|
||||
end
|
||||
end
|
||||
|
||||
for dy = 1:nd1d
|
||||
for dx = 1:nd1d
|
||||
u = 0.0
|
||||
v = 0.0
|
||||
for qy = 1:nq1d
|
||||
u += dq0[dx, qy] * B[qy, dy]
|
||||
v += dq1[dx, qy] * G[qy, dy]
|
||||
end
|
||||
r[dx, dy] += u + v
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
function integrate_value(f_qp, r)
|
||||
end
|
||||
|
||||
function kernel(rho, J, w, dudxi)
|
||||
return rho * rho * ((inv(J) * transpose(inv(J))) * dudxi) * det(J) * w
|
||||
end
|
||||
|
||||
rho_qp = reshape(interpolate_value(rho_e), (nq1d * nq1d,))
|
||||
dudxi_qp = reshape(interpolate_grad(u_e), (nq1d * nq1d, 2))
|
||||
J_qp = reshape(interpolate_grad(x_e), (nq1d * nq1d), 2, 2)
|
||||
f_qp = zeros(nq1d * nq1d, 1, space_dim)
|
||||
for q = 1:nq1d*nq1d
|
||||
f_qp[q, 1, :] = kernel(rho_qp[q], J_qp[q, :, :], w_qp[q], dudxi_qp[q, :])
|
||||
end
|
||||
|
||||
w_qp = reshape(w_qp, nq1d, nq1d)
|
||||
rho_qp = interpolate_value(rho_e)
|
||||
dudxi_qp = reshape(interpolate_grad(u_e), (nq1d, nq1d, 2))
|
||||
J_qp = interpolate_grad(x_e)
|
||||
f_qp = zeros(nq1d, nq1d, 1, space_dim)
|
||||
for qx = 1:nq1d
|
||||
for qy = 1:nq1d
|
||||
f_qp[qx, qy, 1, :] = kernel(rho_qp[qx, qy], J_qp[qx, qy, :, :], w_qp[qx, qy], dudxi_qp[qx, qy, :])
|
||||
end
|
||||
end
|
||||
|
||||
r = zeros(Float64, nd1d, nd1d)
|
||||
integrate_grad(f_qp, r)
|
||||
println(reshape(r, nd1d * nd1d))
|
||||
@@ -0,0 +1,285 @@
|
||||
using LinearAlgebra
|
||||
using DelimitedFiles
|
||||
|
||||
Q = 4
|
||||
D = 3
|
||||
space_dim = 3;
|
||||
num_trial_dof = D^3;
|
||||
test_vdim = 1
|
||||
output_op_dim = 3;
|
||||
|
||||
R_data = Int.(readdlm("/Users/andrej1/repos/mfem/build-debug/r_mat.mtx", ' ', Float64))
|
||||
R_N = maximum(R_data[:, 1:2])
|
||||
R = zeros(Float64, (R_N, R_N))
|
||||
for e = 1:size(R_data, 1)
|
||||
rij = R_data[e, :, :]
|
||||
R[rij[1], rij[2]] = 1.0
|
||||
end
|
||||
|
||||
u_l = reshape(Float64[2.345 3.345 4.345 2.345 2.345 4.595 5.595 2.345 2.845 3.845 3.345 2.345 3.47 5.095 3.97 2.345 2.345 3.97 4.97 2.345 3.095 3.1575 4.47 3.6575 2.345 3.72 3.4075], (num_trial_dof, test_vdim))
|
||||
u_e = R * u_l
|
||||
x_l = transpose(reshape(Float64[0 0 0 1 0 0 1 1 0 0 1 0 0 0 1 1 0 1 1 1 1 0 1 1 0.5 0 0 1 0.5 0 0.5 1 0 0 0.5 0 0.5 0 1 1 0.5 1 0.5 1 1 0 0.5 1 0 0 0.5 1 0 0.5 1 1 0.5 0 1 0.5 0.5 0.5 0 0.5 0 0.5 1 0.5 0.5 0.5 1 0.5 0 0.5 0.5 0.5 0.5 1 0.5 0.5 0.5], (space_dim, num_trial_dof)))
|
||||
x_e = R * x_l
|
||||
|
||||
w_qp = [0.00526143468632 0.00986393947438 0.00986393947438 0.00526143468632 0.00986393947438 0.0184925420071 0.0184925420071 0.00986393947438 0.00986393947438 0.0184925420071 0.0184925420071 0.00986393947438 0.00526143468632 0.00986393947438 0.00986393947438 0.00526143468632 0.00986393947438 0.0184925420071 0.0184925420071 0.00986393947438 0.0184925420071 0.0346691208692 0.0346691208692 0.0184925420071 0.0184925420071 0.0346691208692 0.0346691208692 0.0184925420071 0.00986393947438 0.0184925420071 0.0184925420071 0.00986393947438 0.00986393947438 0.0184925420071 0.0184925420071 0.00986393947438 0.0184925420071 0.0346691208692 0.0346691208692 0.0184925420071 0.0184925420071 0.0346691208692 0.0346691208692 0.0184925420071 0.00986393947438 0.0184925420071 0.0184925420071 0.00986393947438 0.00526143468632 0.00986393947438 0.00986393947438 0.00526143468632 0.00986393947438 0.0184925420071 0.0184925420071 0.00986393947438 0.00986393947438 0.0184925420071 0.0184925420071 0.00986393947438 0.00526143468632 0.00986393947438 0.00986393947438 0.00526143468632]
|
||||
B = reshape([0.80134602937 0.227784076791 -0.112196966794 -0.0597902822241 0.258444252854 0.884412890003 0.884412890003 0.258444252854 -0.0597902822241 -0.112196966794 0.227784076791 0.80134602937], (Q, D))
|
||||
G = reshape([-2.72227262319 -1.67996208717 -0.32003791283 0.722272623188 3.44454524638 1.35992417434 -1.35992417434 -3.44454524638 -0.722272623188 0.32003791283 1.67996208717 2.72227262319], (Q, D))
|
||||
|
||||
function interpolate_value(f_e)
|
||||
vdim = size(f_e, 2)
|
||||
f_e = reshape(f_e, (D, D, D, vdim))
|
||||
f_qp = zeros(vdim, Q, Q, Q)
|
||||
s1 = zeros(Float64, D, D, Q)
|
||||
s2 = zeros(Float64, D, Q, Q)
|
||||
|
||||
for vd = 1:vdim
|
||||
for dz = 1:D
|
||||
for dy = 1:D
|
||||
for qx = 1:Q
|
||||
acc = 0.0
|
||||
for dx = 1:D
|
||||
acc += f_e[dx, dy, dz, vd] * B[qx, dx]
|
||||
end
|
||||
s1[dz, dy, qx] = acc
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
for dz = 1:D
|
||||
for qx = 1:Q
|
||||
for qy = 1:Q
|
||||
acc = 0.0
|
||||
for dy = 1:D
|
||||
acc += s1[dz, dy, qx] * B[qy, dy]
|
||||
end
|
||||
s2[dz, qy, qx] = acc
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
for qz = 1:Q
|
||||
for qy = 1:Q
|
||||
for qx = 1:Q
|
||||
acc = 0.0
|
||||
for dz = 1:D
|
||||
acc += s2[dz, qy, qx] * B[qz, dz]
|
||||
end
|
||||
f_qp[vd, qx, qy, qz] = acc
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
return f_qp
|
||||
end
|
||||
|
||||
function interpolate_grad(f_e)
|
||||
vdim = size(f_e, 2)
|
||||
f_e = reshape(f_e, (D, D, D, vdim))
|
||||
dudxi_qp = zeros(vdim, space_dim, Q, Q, Q)
|
||||
|
||||
s1 = zeros(Float64, D, D, Q)
|
||||
s2 = zeros(Float64, D, D, Q)
|
||||
s3 = zeros(Float64, D, Q, Q)
|
||||
s4 = zeros(Float64, D, Q, Q)
|
||||
s5 = zeros(Float64, D, Q, Q)
|
||||
uvw = zeros(Float64, 3)
|
||||
|
||||
for vd = 1:vdim
|
||||
for dz = 1:D
|
||||
for dy = 1:D
|
||||
for qx = 1:Q
|
||||
uvw .= 0.0
|
||||
for dx = 1:D
|
||||
f = f_e[dx, dy, dz, vd]
|
||||
uvw[1] += f * B[qx, dx]
|
||||
uvw[2] += f * G[qx, dx]
|
||||
end
|
||||
s1[dz, dy, qx] = uvw[1]
|
||||
s2[dz, dy, qx] = uvw[2]
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
for dz = 1:D
|
||||
for qy = 1:Q
|
||||
for qx = 1:Q
|
||||
uvw .= 0.0
|
||||
for dy = 1:D
|
||||
uvw[1] += s2[dz, dy, qx] * B[qy, dy]
|
||||
uvw[2] += s1[dz, dy, qx] * G[qy, dy]
|
||||
uvw[3] += s1[dz, dy, qx] * B[qy, dy]
|
||||
end
|
||||
s3[dz, qy, qx] = uvw[1]
|
||||
s4[dz, qy, qx] = uvw[2]
|
||||
s5[dz, qy, qx] = uvw[3]
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
for qz = 1:Q
|
||||
for qy = 1:Q
|
||||
for qx = 1:Q
|
||||
uvw .= 0.0
|
||||
for dz = 1:D
|
||||
uvw[1] += s3[dz, qy, qx] * B[qz, dz]
|
||||
uvw[2] += s4[dz, qy, qx] * B[qz, dz]
|
||||
uvw[3] += s5[dz, qy, qx] * G[qz, dz]
|
||||
end
|
||||
dudxi_qp[vd, 1, qx, qy, qz] = uvw[1]
|
||||
dudxi_qp[vd, 2, qx, qy, qz] = uvw[2]
|
||||
dudxi_qp[vd, 3, qx, qy, qz] = uvw[3]
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
return dudxi_qp
|
||||
end
|
||||
|
||||
function integrate_value(f_qp)
|
||||
r = zeros(Float64, D, D, D, test_vdim)
|
||||
f_qp = reshape(f_qp, (test_vdim, output_op_dim, Q, Q, Q))
|
||||
s1 = zeros(Float64, Q, Q, D)
|
||||
s2 = zeros(Float64, Q, D, D)
|
||||
|
||||
for vd = 1:test_vdim
|
||||
for qy = 1:Q
|
||||
for dx = 1:D
|
||||
for qz = 1:Q
|
||||
acc = 0.0
|
||||
for qx = 1:Q
|
||||
acc += f_qp[test_vdim, 1, qx, qy, qz] * B[qx, dx]
|
||||
end
|
||||
s1[qz, qy, dx] = acc
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# for i = 1:Q
|
||||
# for j = 1:Q
|
||||
# for k = 1:D
|
||||
# print(s1[i,j,k], " ")
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
# println()
|
||||
|
||||
for dy = 1:D
|
||||
for dx = 1:D
|
||||
for qz = 1:Q
|
||||
acc = 0.0
|
||||
for qy = 1:Q
|
||||
acc += s1[qz, qy, dx] * B[qy, dy]
|
||||
end
|
||||
s2[qz, dy, dx] = acc
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
for dy = 1:D
|
||||
for dx = 1:D
|
||||
for dz = 1:D
|
||||
acc = 0.0
|
||||
for qz = 1:Q
|
||||
acc += s2[qz, dy, dx] * B[qz, dz]
|
||||
end
|
||||
r[dx, dy, dz, vd] += acc
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
return r
|
||||
end
|
||||
|
||||
function integrate_grad(f_qp)
|
||||
r = zeros(Float64, D, D, D, test_vdim)
|
||||
f_qp = reshape(f_qp, (test_vdim, output_op_dim, Q, Q, Q))
|
||||
s0 = zeros(Float64, Q, Q, D)
|
||||
s1 = zeros(Float64, Q, Q, D)
|
||||
s2 = zeros(Float64, Q, Q, D)
|
||||
s3 = zeros(Float64, Q, D, D)
|
||||
s4 = zeros(Float64, Q, D, D)
|
||||
s5 = zeros(Float64, Q, D, D)
|
||||
|
||||
for vd = 1:test_vdim
|
||||
for qz = 1:Q
|
||||
for qy = 1:Q
|
||||
for dx = 1:D
|
||||
uvw = zeros(Float64, 3)
|
||||
for qx = 1:Q
|
||||
uvw[1] += f_qp[vd, 1, qx, qy, qz] * G[qx, dx]
|
||||
uvw[2] += f_qp[vd, 2, qx, qy, qz] * B[qx, dx]
|
||||
uvw[3] += f_qp[vd, 3, qx, qy, qz] * B[qx, dx]
|
||||
end
|
||||
s0[qz, qy, dx] = uvw[1]
|
||||
s1[qz, qy, dx] = uvw[2]
|
||||
s2[qz, qy, dx] = uvw[3]
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
for qz = 1:Q
|
||||
for dy = 1:D
|
||||
for dx = 1:D
|
||||
uvw = zeros(Float64, 3)
|
||||
for qy = 1:Q
|
||||
uvw[1] += s0[qz, qy, dx] * B[qy, dy]
|
||||
uvw[2] += s1[qz, qy, dx] * G[qy, dy]
|
||||
uvw[3] += s2[qz, qy, dx] * B[qy, dy]
|
||||
end
|
||||
s3[qz, dy, dx] = uvw[1]
|
||||
s4[qz, dy, dx] = uvw[2]
|
||||
s5[qz, dy, dx] = uvw[3]
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
for dz = 1:D
|
||||
for dy = 1:D
|
||||
for dx = 1:D
|
||||
uvw = zeros(Float64, 3)
|
||||
for qz = 1:Q
|
||||
uvw[1] += s3[qz, dy, dx] * B[qz, dz]
|
||||
uvw[2] += s4[qz, dy, dx] * B[qz, dz]
|
||||
uvw[3] += s5[qz, dy, dx] * G[qz, dz]
|
||||
end
|
||||
r[dx, dy, dz, vd] += sum(uvw)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
for i = 1:D
|
||||
for j = 1:D
|
||||
for k = 1:D
|
||||
print(f_qp[1, 1, i,j,k], " ")
|
||||
end
|
||||
end
|
||||
end
|
||||
println()
|
||||
end
|
||||
|
||||
return r
|
||||
end
|
||||
|
||||
function kernel(dudxi, J, w)
|
||||
transpose(inv(J)) * inv(J) * dudxi * det(J) * w
|
||||
end
|
||||
|
||||
dudxi_qp = reshape(interpolate_grad(u_e), (space_dim, Q * Q * Q))
|
||||
J_qp = reshape(interpolate_grad(x_e), (space_dim, space_dim, Q * Q * Q))
|
||||
f_qp = zeros(test_vdim, output_op_dim, Q * Q * Q)
|
||||
|
||||
for q = 1:Q*Q*Q
|
||||
r_qp = kernel(dudxi_qp[:, q], J_qp[:, :, q], w_qp[q])
|
||||
for od = 1:output_op_dim
|
||||
f_qp[1, od, q] += r_qp[od]
|
||||
end
|
||||
end
|
||||
|
||||
r = integrate_grad(f_qp)
|
||||
# println(transpose(R) * reshape(r, (D * D * D)))
|
||||
|
||||
println(reshape(r, D * D * D))
|
||||
+145
@@ -0,0 +1,145 @@
|
||||
using Enzyme, ForwardDiff, LinearAlgebra
|
||||
|
||||
num_qp = 4;
|
||||
num_trial_dof = 4;
|
||||
trial_vdim = 1
|
||||
trial_op_dim = [1, 2]
|
||||
num_test_dof = num_trial_dof
|
||||
test_op_dim = 2
|
||||
test_vdim = 1
|
||||
|
||||
Bu_u_mem = [0.622008468 0.166666667 0.166666667 0.0446581987 0.166666667 0.622008468 0.0446581987 0.166666667 0.0446581987 0.166666667 0.166666667 0.622008468 0.166666667 0.0446581987 0.622008468 0.166666667]
|
||||
Bu_du_mem = [-0.788675135 -0.788675135 -0.211324865 -0.211324865 -0.788675135 -0.211324865 -0.788675135 -0.211324865 0.788675135 0.788675135 0.211324865 0.211324865 -0.211324865 -0.788675135 -0.211324865 -0.788675135 0.211324865 0.211324865 0.788675135 0.788675135 0.211324865 0.788675135 0.211324865 0.788675135 -0.211324865 -0.211324865 -0.788675135 -0.788675135 0.788675135 0.211324865 0.788675135 0.211324865]
|
||||
Bu_u = reshape(Bu_u_mem, (num_qp, trial_op_dim[1], num_trial_dof))
|
||||
Bu_du = reshape(Bu_du_mem, (num_qp, trial_op_dim[2], num_trial_dof))
|
||||
Bu = [Bu_u, Bu_du]
|
||||
Bv = Bu_du;
|
||||
|
||||
det(A) = A[1, 1] * A[2, 2] - A[1, 2] * A[2, 1]
|
||||
|
||||
# kernel(u, J, w) = u * det(J) * w
|
||||
|
||||
# u = [1.0, 1.0];
|
||||
# du = [0.0, 0.0];
|
||||
J = [1 0; 0 1];
|
||||
w = 0.25;
|
||||
|
||||
### ParamtricFunction test
|
||||
|
||||
# ParamtricFunction as quadrature data
|
||||
pf_size_on_qp = 4
|
||||
pf = zeros(num_qp * pf_size_on_qp)
|
||||
|
||||
residual_size_on_qp = zeros(length(pf))
|
||||
|
||||
function kernel(J, w)
|
||||
return J * w
|
||||
end
|
||||
|
||||
function Trho(u, e, q, size_on_qp, num_qp)
|
||||
b = q * size_on_qp
|
||||
c = q * size_on_qp + ((e - 1) * num_qp * size_on_qp)
|
||||
return u[b-size_on_qp+1:c]
|
||||
end
|
||||
|
||||
function TrhoT(u, uq, e, q, size_on_qp, num_qp)
|
||||
b = q * size_on_qp
|
||||
c = q * size_on_qp + ((e - 1) * num_qp * size_on_qp)
|
||||
return u[b-size_on_qp+1:c] = uq
|
||||
end
|
||||
|
||||
for e = 1:1
|
||||
for q = 1:num_qp
|
||||
pfq = Trho(pf, e, q, pf_size_on_qp, num_qp)
|
||||
pfq[:] = kernel(J, w)
|
||||
TrhoT(pf, pfq, e, q, pf_size_on_qp, num_qp)
|
||||
end
|
||||
end
|
||||
|
||||
# println(pf)
|
||||
|
||||
nqp = 4
|
||||
data = reshape([1 1 1 1 0 0 0 0 0 0 0 0 1 1 1 1], (16,))
|
||||
m = 2
|
||||
n = 2
|
||||
arg = zeros(n, m)
|
||||
for q = 1:nqp
|
||||
for i = 1:m
|
||||
for j = 1:n
|
||||
arg[j, i] = data[(i * m) + j]
|
||||
end
|
||||
end
|
||||
println(arg)
|
||||
end
|
||||
|
||||
|
||||
# function kernel(u, dudxi, J, w)
|
||||
# invJ = J
|
||||
# dudx = dudxi * invJ
|
||||
# return u[1, 1] * dudx * det(J) * w * transpose(invJ)
|
||||
# end
|
||||
|
||||
# u = ones((1, 1))
|
||||
# du = [zeros((1, 1)), zeros((2, 2))]
|
||||
# dudxi = zeros((2, 2))
|
||||
# J = [1 0; 0 1];
|
||||
# w = 0.25;
|
||||
|
||||
# wrap(x) = kernel(x, J, w)
|
||||
|
||||
# D = zeros(test_vdim, test_op_dim, trial_vdim, sum(trial_op_dim), num_qp)
|
||||
# for q = 1:num_qp
|
||||
# for j = 1:trial_vdim
|
||||
# m_offset = 0
|
||||
# local trial_op_dim = size(Bu[s])[2]
|
||||
# for s = 1:length(Bu)
|
||||
# for m = 1:trial_op_dim
|
||||
# du[s][j, m] = 1.0
|
||||
# df = autodiff(Forward, kernel, Duplicated(u, du[1]), Duplicated(dudxi, du[2]), Const(J), Const(w))[1]
|
||||
# du[s][j, m] = 0.0
|
||||
# for i = 1:test_vdim
|
||||
# for k = 1:test_op_dim
|
||||
# D[i, k, j, m+m_offset, q] = df[i, k]
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
# m_offset += trial_op_dim
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
|
||||
# Ae = zeros(num_test_dof, test_vdim, num_trial_dof, trial_vdim)
|
||||
|
||||
# for J = 1:num_trial_dof
|
||||
# for j = 1:trial_vdim
|
||||
# fhat = zeros(test_vdim, test_op_dim, num_qp)
|
||||
# m_offset = 0
|
||||
# for s = 1:length(Bu)
|
||||
# trial_op_dim = size(Bu[s])[2]
|
||||
# # precompute fhat for trial dof J column
|
||||
# for qp = 1:num_qp
|
||||
# for i = 1:test_vdim
|
||||
# for k = 1:test_op_dim
|
||||
# for m = 1:trial_op_dim
|
||||
# fhat[i, k, qp] += D[i, k, j, m+m_offset, qp] * Bu[s][qp, m, J]
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
# m_offset += trial_op_dim
|
||||
# end
|
||||
|
||||
# # this imitates what 'map_quadrature_data_to_fields' does
|
||||
# for I = 1:num_test_dof
|
||||
# for i = 1:test_vdim
|
||||
# for qp = 1:num_qp
|
||||
# for k = 1:test_op_dim
|
||||
# Ae[I, i, J, j] += fhat[i, k, qp] * Bv[qp, k, I]
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
# end
|
||||
|
||||
# display(reshape(Ae, (num_test_dof * test_vdim, num_trial_dof * trial_vdim)))
|
||||
Reference in New Issue
Block a user