Compare commits

..
Author SHA1 Message Date
camierjs f2a424720b Merge branch 'main' into hpcftools/flowSolver-gpu 2025-06-30 11:17:56 -07:00
camierjs e5fad8067c Merge branch 'master' into hpcftools/flowSolver-gpu 2025-06-30 11:17:45 -07:00
camierjs 0b5bb52996 miniapps/navier/incompressible_navier_dfem 2025-06-30 11:17:11 -07:00
camierjs 563bfe9514 Merge branch 'master' 2025-06-30 08:43:14 -07:00
camierjs 9afef578d5 Merge branch 'master' into hpcftools/flowSolver-gpu 2025-05-02 09:59:44 -07:00
camierjs 9ecc414c62 dot reduced tests 2024-12-04 18:25:05 -08:00
camierjs 263b9d32c1 Merge branch 'hpcftools/flowSolver' 2024-12-04 11:59:05 -08:00
Mathias Rainer Schmidt 74e4ad3e2c - updated flow solver
- added comments to Blf and Lf contributions
- split setup and step into vel, auxiliary and pressure part
2024-11-25 11:04:30 -08:00
camierjs 79039e0f6f Switched to PA 2024-11-21 10:52:39 -08:00
Mathias Rainer Schmidt cf9fcd8dde Merge remote-tracking branch 'origin/master' into hpcftools/flowSolver 2024-11-14 13:52:27 -08:00
camierjs c857fde13b Merge branch 'hpcftools/flowSolver' 2024-11-01 15:59:50 -07:00
Mathias Rainer Schmidt fa8617ada3 - added partial assembly option 2024-11-01 13:20:29 -07:00
camierjs c06cbb69d5 Setup and cleanup 2024-10-30 11:37:00 -07:00
Mathias Rainer Schmidt f7e5db2cea - added ortho solver to phi field 2024-10-24 16:06:37 -07:00
Mathias Rainer Schmidt 60c11776b6 - added executable 2024-10-21 15:57:23 -07:00
Mathias Rainer Schmidt 6238f8ca76 - update solution 2024-10-16 16:00:08 -07:00
Mathias Rainer Schmidt 6a256db9aa - added linear solvers to step 2024-10-16 15:56:46 -07:00
Mathias Rainer Schmidt e0fb9658ca - added linear form integrators 2024-10-15 17:30:09 -07:00
Mathias Rainer Schmidt a1089efac3 - added BilinearForms 2024-10-15 13:46:36 -07:00
Mathias Rainer Schmidt 83f7f769dc - inital flow solver commit 2024-10-15 12:44:25 -07:00
139 changed files with 3535 additions and 4815 deletions
-154
View File
@@ -1,154 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: Sanitizer Config
description: Sets up environment variables for MFEM sanitizer workflow
inputs:
DEBUG:
description: If true, use intermediate caches to speed up the workflow
by reusing previous builds.
default: false
REPOSITORY:
description: Repository to checkout
default: mfem/mfem
BRANCH:
description: Branch to checkout
default: ubsan
CLANG_VER:
description: CLANG version to use
default: 18
# https://github.com/llvm/llvm-project/releases
LLVM_VER:
description: LLVM version to use
default: 19.1.7
# https://github.com/hypre-space/hypre/releases
HYPRE_VER:
description: HYPRE version to use
default: 2.19.0
METIS_VER:
description: METIS version to use
default: 4.0.3
CTEST:
description: CTest command to use
default: ctest -j --test-load $(nproc)
--schedule-random
--stop-on-failure --output-on-failure
--test-dir
# https://clang.llvm.org/docs/AddressSanitizer.html
ASAN_OPTIONS:
default: detect_leaks=1,
strict_init_order=1,
strict_string_checks=1,
check_initialization_order=1,
detect_stack_use_after_return=1
ASAN_CXXFLAGS:
default: -fsanitize=address
-fsanitize-address-use-after-scope
ASAN_LDFLAGS:
default: -fsanitize=address
# https://clang.llvm.org/docs/UndefinedBehaviorSanitizer.html
UBSAN_OPTIONS:
default: halt_on_error=1, print_stacktrace=1
UBSAN_CXXFLAGS:
default: -fsanitize=undefined
UBSAN_LDFLAGS:
default: -fsanitize=undefined
# https://clang.llvm.org/docs/MemorySanitizer.html
MSAN_OPTIONS:
default: "poison_in_dtor=1"
MSAN_CXXFLAGS:
default: -fsanitize=memory
-fsanitize-memory-track-origins
-fsanitize-memory-use-after-dtor
MSAN_LDFLAGS:
default: -fsanitize=memory
LSAN_DIR:
description: LSAN suppression directory
default: lsan
LSAN_FILE:
description: LSAN suppression file
default: lsan.supp
NO_FLAGS:
description: If true, do not set any CXXFLAGS or LDFLAGS.
default: false
runs:
using: 'composite'
steps:
- name: Env (Inputs)
run: |
echo DEBUG=${{inputs.DEBUG}} >> $GITHUB_ENV
echo REPOSITORY=${{inputs.REPOSITORY}} >> $GITHUB_ENV
echo BRANCH=${{inputs.BRANCH}} >> $GITHUB_ENV
echo CLANG_VER=${{inputs.CLANG_VER}} >> $GITHUB_ENV
echo LLVM_VER=${{inputs.LLVM_VER}} >> $GITHUB_ENV
echo HYPRE_VER=${{inputs.HYPRE_VER}} >> $GITHUB_ENV
echo METIS_VER=${{inputs.METIS_VER}} >> $GITHUB_ENV
echo CTEST=${{inputs.CTEST}} >> $GITHUB_ENV
echo ASAN_OPTIONS=${{inputs.ASAN_OPTIONS}} >> $GITHUB_ENV
echo UBSAN_OPTIONS=${{inputs.UBSAN_OPTIONS}} >> $GITHUB_ENV
echo MSAN_OPTIONS=${{inputs.MSAN_OPTIONS}} >> $GITHUB_ENV
echo LSAN_DIR=${{inputs.LSAN_DIR}} >> $GITHUB_ENV
echo LSAN_FILE=${{inputs.LSAN_FILE}} >> $GITHUB_ENV
echo ASAN_CXXFLAGS=${{inputs.ASAN_CXXFLAGS}} >> $GITHUB_ENV
echo ASAN_LDFLAGS=${{inputs.ASAN_LDFLAGS}} >> $GITHUB_ENV
echo UBSAN_CXXFLAGS=${{inputs.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
echo UBSAN_LDFLAGS=${{inputs.UBSAN_LDFLAGS}} >> $GITHUB_ENV
echo MSAN_CXXFLAGS=${{inputs.MSAN_CXXFLAGS}} >> $GITHUB_ENV
echo MSAN_LDFLAGS=${{inputs.MSAN_LDFLAGS}} >> $GITHUB_ENV
shell: bash
- name: Env (dir)
run: |
echo LLVM_DIR=${{github.workspace}}/llvm >> $GITHUB_ENV
echo HYPRE_DIR=hypre-${{inputs.HYPRE_VER}} >> $GITHUB_ENV
echo METIS_DIR=metis-${{inputs.METIS_VER}} >> $GITHUB_ENV
shell: bash
- name: Env (bis)
run: |
echo CC=clang-${{inputs.CLANG_VER}} >> $GITHUB_ENV
echo CXX=clang++-${{inputs.CLANG_VER}} >> $GITHUB_ENV
echo LLVM_INC=${{env.LLVM_DIR}}/include/c++/v1 >> $GITHUB_ENV
echo LLVM_LIB=${{env.LLVM_DIR}}/lib >> $GITHUB_ENV
echo HYPRE_TGZ=v${{inputs.HYPRE_VER}}.tar.gz >> $GITHUB_ENV
echo METIS_TGZ=metis-${{inputs.METIS_VER}}.tar.gz >> $GITHUB_ENV
LSAN_SUPPRESSIONS="${{github.workspace}}/${{inputs.LSAN_DIR}}/${{inputs.LSAN_FILE}}"
echo "LSAN_OPTIONS=suppressions=$LSAN_SUPPRESSIONS" >> $GITHUB_ENV
shell: bash
- name: Env (ter)
if: ${{ inputs.NO_FLAGS != 'true' }}
run: |
echo LLVM_CXXFLAGS=-stdlib=libc++ -I${{env.LLVM_INC}} -Isystem${{env.LLVM_INC}} >> $GITHUB_ENV
echo LLVM_LDFLAGS=-L${{env.LLVM_LIB}} -lc++abi -Wl,-rpath,${{env.LLVM_LIB}} >> $GITHUB_ENV
shell: bash
- name: Env (quater)
if: ${{ inputs.NO_FLAGS != 'true' }}
run: |
echo CXXFLAGS=${{env.LLVM_CXXFLAGS}} >> $GITHUB_ENV
echo LDFLAGS=${{env.LLVM_LDFLAGS}} >> $GITHUB_ENV
shell: bash
-91
View File
@@ -1,91 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: 'MFEM Compilation'
description: 'MFEM Compilation'
inputs:
par:
description: 'Whether to build for parallel (true/false)'
default: false
sanitizer:
description: 'Sanitizer to use (asan, msan, ubsan)'
default: asan
runs:
using: 'composite'
steps:
- uses: ./.github/actions/sanitize/config
- uses: actions/cache@v4
if: ${{env.DEBUG == 'true'}}
id: debug
with:
path: mfem/build
key: build-${{inputs.par}}-${{inputs.sanitizer}}
- uses: ./.github/actions/sanitize/setup
if: ${{steps.debug.outputs.cache-hit != 'true'}}
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
- name: Build with ASAN
if: inputs.sanitizer == 'asan'
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.ASAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- name: Build with MSAN
if: inputs.sanitizer == 'msan'
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MSAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- name: Build with UBSAN
if: inputs.sanitizer == 'ubsan'
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
shell: bash
- uses: mfem/github-actions/build-mfem@v2.5
if: ${{steps.debug.outputs.cache-hit != 'true'}}
env:
CXXFLAGS: ${{env.CXXFLAGS}}
LDFLAGS: ${{env.LDFLAGS}}
with:
mpi: ${{inputs.par == 'false' && 'seq' || 'par'}}
mfem-dir: mfem
os: ${{runner.os}}
library-only: true
build-system: cmake
hypre-dir: ${{env.HYPRE_DIR}}
metis-dir: ${{env.METIS_DIR}}
config-options: >-
-GNinja
-DMPICXX=${{env.CXX}}
-DCMAKE_CXX_STANDARD=17
-DMFEM_USE_MEMALLOC=OFF
-DCMAKE_BUILD_TYPE=Release
-DCMAKE_VERBOSE_MAKEFILE=ON
-DCMAKE_CXX_COMPILER=${{env.CXX}}
-DCMAKE_CXX_FLAGS_RELEASE='-g -O1 -fno-omit-frame-pointer'
- name: Delete object files
if: ${{steps.debug.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: find . -type f -name '*.o' -delete
shell: bash
- uses: actions/upload-artifact@v4
with:
name: build-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build
if-no-files-found: error
retention-days: 1
overwrite: false
-33
View File
@@ -1,33 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: 'Install MPI'
description: 'Installs MPI and set up its environment variables'
runs:
using: 'composite'
steps:
- name: Install
run: sudo apt-get install openmpi-bin libopenmpi-dev
shell: bash
- name: Env
run: |
echo PRTE_MCA_rmaps_default_mapping_policy=:oversubscribe >> $GITHUB_ENV
echo MPI_INC=$(mpicxx --showme:compile) >> $GITHUB_ENV
echo MPI_LIB=$(mpicxx --showme:link) >> $GITHUB_ENV
shell: bash
- name: Env (bis)
run: |
echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MPI_INC}} >> $GITHUB_ENV
echo LDFLAGS=${{env.LDFLAGS}} ${{env.MPI_LIB}} >> $GITHUB_ENV
shell: bash
@@ -1,71 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: 'Restore state'
description: 'Restore state to be able to run checks, tests'
inputs:
par:
description: 'Whether to build for parallel (true/false)'
default: false
sanitizer:
description: 'Sanitizer to use (asan, msan, ubsan)'
default: asan
cache-path:
description: 'path to what needs to be restored'
default: none
cache-skip:
description: 'Skip cache restoration'
default: false
outputs:
cache-hit:
description: 'Output from a specific step'
value: ${{steps.debug.outputs.cache-hit}}
runs:
using: 'composite'
steps:
- uses: ./.github/actions/sanitize/config
- uses: actions/cache@v4
if: ${{env.DEBUG == 'true' && inputs.cache-skip != 'true'}}
id: debug
with:
path: ${{inputs.cache-path}}
key: ${{github.job}}-${{inputs.par}}-${{inputs.sanitizer}}
- uses: ./.github/actions/sanitize/setup
if: ${{steps.debug.outputs.cache-hit != 'true'}}
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
- uses: actions/download-artifact@v4
with:
name: build-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build
- name: Ninja Patch
working-directory: mfem/build
run: |
sed -i -e 's/CXX_STATIC_LIBRARY_LINKER__mfem_Release.*/CUSTOM_COMMAND/' build.ninja
sed -i -e '/build tests\/unit\/all:/ s/tests\/unit\/[^ ]*unit_tests[^ ]*//g' build.ninja
sed -i -e '/^add_test(\[=\[\(unit_tests\|punit_tests\)\]=\]/ s/)/ "--input-file .\/list-test-names-${{matrix.tag}}" "--min-duration 1")/' tests/unit/CTestTestfile.cmake
shell: bash
- name: Copy Data
if: ${{steps.debug.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
ninja cmake_object_order_depends_target_unit_tests
cp -pR ../tests/unit/data tests/unit
shell: bash
-64
View File
@@ -1,64 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: 'Setup state'
description: 'Sets up the state to be able to run build & run'
inputs:
par:
description: 'Whether to build for parallel (true/false)'
default: false
sanitizer:
description: 'Sanitizer to use (asan, msan, ubsan)'
default: asan
runs:
using: 'composite'
steps:
- uses: actions/cache/restore@v4 # Cache for LLVM libcxx
with:
path: ${{env.LLVM_DIR}}
fail-on-cache-miss: true
key: build-libcxx-${{env.LLVM_VER}}-${{inputs.sanitizer}}
- uses: ./.github/actions/sanitize/mpi
if: ${{inputs.par == 'true'}}
- uses: actions/cache/restore@v4 # Cache for Hypre
if: ${{inputs.par == 'true'}}
with:
path: ${{env.HYPRE_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
- uses: actions/cache/restore@v4 # Cache for Metis
if: ${{inputs.par == 'true'}}
with:
path: ${{env.METIS_DIR}}
fail-on-cache-miss: true
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
- name: Hypre/Metis links
if: ${{inputs.par == 'true'}}
run: ln -s -f ${{env.HYPRE_DIR}} hypre && ln -s -f ${{env.METIS_DIR}} metis-4.0
shell: bash
- uses: actions/cache/restore@v4 # Cache for LSAN suppression file
with:
path: ${{env.LSAN_DIR}}
fail-on-cache-miss: true
key: build-lsan-suppression-file
- uses: actions/checkout@v4 # Checkout the repository
with:
path: mfem
# ref: ${{env.BRANCH}}
# repository: ${{env.REPOSITORY}}
+7 -26
View File
@@ -7,17 +7,18 @@
https://mfem.org
This directory contains the GitHub CI scripts for MFEM.
Note that some of these scripts use the shared MFEM GitHub Actions from the external mfem/github-actions repository:
<https://github.com/mfem/github-actions>
https://github.com/mfem/github-actions
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch in the above from which the action is taken.
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.1`, the `v2.1` suffix denotes the branch in the above from which the action is taken.
The current CI workflows are:
## `repo-check.yml`
### `repo-check.yml`
Runs a number of static repository-level sanity checks.
@@ -29,39 +30,19 @@ Runs a number of static repository-level sanity checks.
- `branch-history` guards against accidental commits of large files using the `--history` option of the `config/githooks/pre-push` script.
## `mfem-analysis.yml` (`build-analysis`)
### `mfem-analysis.yml` (`build-analysis`)
Checks if the code builds and satisfies minimal requirements.
- `gitignore` builds hypre, METIS, and MFEM using `mfem/github-actions/build-hypre`, `mfem/github-actions/build-metis`, and `mfem/github-actions/build-mfem` and checks for correct `.gitignore` settings by running the `tests/scripts/gitignore` script.
## `builds-and-tests.yml`
### `builds-and-tests.yml`
Runs a matrix of builds and tests runs with different compilers, OS, mfem/hypre settings, etc. Also processes and upload Codecov reports.
Uses the following GitHub Actions from <https://github.com/mfem/github-actions>:
Uses the following GitHub Actions from https://github.com/mfem/github-actions:
- `mfem/github-actions/build-hypre`
- `mfem/github-actions/build-metis`
- `mfem/github-actions/build-mfem`
- `mfem/github-actions/upload-coverage`
## Sanitizer Workflow for MFEM Verification
This workflow validates MFEM unit tests, examples, and miniapps using sanitizer tools.
- `sanitizers.yml` orchestrates:
- Building and caching dependencies: HYPRE, METIS, LSAN suppression file, and LLVM libcxx.
- Launching fine-grained jobs for serial (ASAN, MSAN, UBSAN) and parallel (ASAN, UBSAN) sanitizers.
- `sanitize-tests.yml` is a reusable workflow accepting `par` mode (`true` for parallel) and `sanitizer` (ASAN, MSAN, or UBSAN) as inputs. It executes the following jobs:
- **Build**: Compiles the MFEM library with specified parallel and sanitizer settings.
- **Check**: Runs verification checks.
- Parallel jobs to test the following: **Examples**, **Miniapps** and **Unit tests**
The workflow leverages composite actions in `.github/actions/sanitize/`:
- `config`: Centralizes settings for the sanitizer workflow.
- `mfem`: Manages the MFEM library build process.
- `mpi`: Installs MPI and applies additional compilation flags.
- `restore`: Restores the testing environment state.
- `setup`: Builds or restores cached dependencies.
+4 -4
View File
@@ -289,10 +289,10 @@ jobs:
run: |
export HOMEBREW_NO_INSTALL_CLEANUP=1
brew update
brew install llvm@20 enzyme
echo "LLVM_PREFIX=$(brew --prefix llvm@20)" >> $GITHUB_ENV
echo "OMPI_CC=$(brew --prefix llvm@20)/bin/clang" >> $GITHUB_ENV
echo "OMPI_CXX=$(brew --prefix llvm@20)/bin/clang++" >> $GITHUB_ENV
brew install llvm@19 enzyme
echo "LLVM_PREFIX=$(brew --prefix llvm@19)" >> $GITHUB_ENV
echo "OMPI_CC=$(brew --prefix llvm@19)/bin/clang" >> $GITHUB_ENV
echo "OMPI_CXX=$(brew --prefix llvm@19)/bin/clang++" >> $GITHUB_ENV
# MFEM build and test
- name: build
+69
View File
@@ -0,0 +1,69 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
name: "Sanitizer"
permissions:
actions: write
on:
push:
branches:
- master
- next
pull_request:
workflow_dispatch:
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
jobs:
Serial:
runs-on: ubuntu-24.04
steps:
- name: MFEM Checkout
uses: actions/checkout@v4
with:
path: mfem
- name: MFEM Build
uses: mfem/github-actions/build-mfem@v2.5
with:
os: ${{ runner.os }}
target: opt
mpi: seq
hypre-dir: unused-hypre-dir
metis-dir: unused-metis-dir
mfem-dir: mfem
build-system: make
library-only: false
config-options:
CXX="clang++-18"
CXXFLAGS="-g -O1 -std=c++17
-fsanitize=address
-fno-omit-frame-pointer
-fsanitize-address-use-after-scope"
- name: MFEM Info
working-directory: mfem
run: make info
- name: MFEM Sanitize
working-directory: mfem
run:
ASAN_OPTIONS="detect_leaks=1,
strict_init_order=1,
strict_string_checks=1,
check_initialization_order=1,
detect_stack_use_after_return=1"
make test
@@ -1,39 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: build-hypre
on:
workflow_call:
jobs:
build-hypre:
runs-on: ubuntu-latest
name: 2.19.0
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
with:
path: ${{env.HYPRE_DIR}}
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-hypre@v2.5
with:
archive: ${{env.HYPRE_TGZ}}
dir: ${{env.HYPRE_DIR}}
target: int32
precision: fp64
build-system: make
@@ -1,76 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: build-libcxx
on:
workflow_call:
jobs:
build-llvm-libcxx:
runs-on: ubuntu-latest
strategy:
matrix:
sanitizer: [asan, msan, ubsan]
include:
- sanitizer: asan
llvm_use_sanitizer: "Address"
- sanitizer: msan
llvm_use_sanitizer: "MemoryWithOrigins"
- sanitizer: ubsan
llvm_use_sanitizer: "Undefined"
name: ${{matrix.sanitizer}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/config
with:
NO_FLAGS: true
- name: Cache
id: cache
uses: actions/cache@v4
with:
path: ${{env.LLVM_DIR}}
key: build-libcxx-${{env.LLVM_VER}}-${{matrix.sanitizer}}
- name: Clone
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
run: >
git clone --filter=blob:none --depth=1
--branch llvmorg-${{env.LLVM_VER}}
--no-checkout https://github.com/llvm/llvm-project.git llvm-project
- name: Checkout
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
working-directory: llvm-project
run: |
git sparse-checkout set --cone
git checkout llvmorg-${{env.LLVM_VER}}
git sparse-checkout set cmake llvm/cmake runtimes libcxx libcxxabi
- name: Mkdir
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
run: mkdir ${{env.LLVM_DIR}}
- name: CMake
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
working-directory: ${{env.LLVM_DIR}}
run: >
VERBOSE=1
cmake -GNinja ../llvm-project/runtimes/
-DCMAKE_C_COMPILER=${{env.CC}}
-DCMAKE_CXX_COMPILER=${{env.CXX}}
-DCMAKE_BUILD_TYPE=RelWithDebInfo
-DCMAKE_INSTALL_PREFIX=/usr
-DLLVM_USE_SANITIZER=${{matrix.llvm_use_sanitizer}}
-DLLVM_BUILD_32_BITS=OFF
-DLIBCXXABI_USE_LLVM_UNWINDER=OFF
-DLLVM_INCLUDE_TESTS=OFF
-DLIBCXX_INCLUDE_TESTS=OFF
-DLIBCXX_INCLUDE_BENCHMARKS=OFF
-DLLVM_ENABLE_RUNTIMES='libcxx;libcxxabi'
- name: Build
if: ${{ steps.cache.outputs.cache-hit != 'true' }}
working-directory: ${{env.LLVM_DIR}}
run: cmake --build . -- cxx cxxabi
-38
View File
@@ -1,38 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: build-file-lsan
on:
workflow_call:
jobs:
build-file-lsan:
runs-on: ubuntu-latest
name: lsan.supp
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
with:
path: ${{env.LSAN_DIR}}
key: build-lsan-suppression-file
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
run: |
mkdir -p ${{env.LSAN_DIR}}
cat << EOF > ${{env.LSAN_DIR}}/${{env.LSAN_FILE}}
leak:libevent_core-2.1.so
leak:ompi_mpi_finalize
leak:ompi_mpi_init
leak:PMPI_Init
leak:strdup
EOF
@@ -1,36 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: build-metis
on:
workflow_call:
jobs:
build-metis:
runs-on: ubuntu-latest
name: 4.0.3
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/config
- name: Cache
id: cache
uses: actions/cache@v4
with:
path: ${{env.METIS_DIR}}
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
- name: Setup
if: steps.cache.outputs.cache-hit != 'true'
uses: ./.github/actions/sanitize/mpi
- name: Build
if: steps.cache.outputs.cache-hit != 'true'
uses: mfem/github-actions/build-metis@v2.5
with:
archive: ${{env.METIS_TGZ}}
dir: ${{env.METIS_DIR}}
-197
View File
@@ -1,197 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: Sanitize
on:
workflow_call:
inputs:
par:
description: 'Whether to build for parallel (true/false)'
required: false
default: false
type: boolean
sanitizer:
description: 'Sanitizer to use (asan, msan, ubsan)'
required: true
default: asan
type: string
jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/mfem
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
check:
needs: [build]
runs-on: ubuntu-latest
env:
ex: ${{inputs.par && 'ex1p' || 'ex1'}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/examples/${{env.ex}}
- name: MFEM Check
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v check
examples:
needs: [check]
runs-on: ubuntu-latest
env:
exclude: ${{inputs.par && '-E "_ser"' || ''}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/examples/ex1
- name: Build Examples
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v examples
- name: Test Examples
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
${{env.CTEST}} examples ${{env.exclude}} --show-only
${{env.CTEST}} examples ${{env.exclude}}
miniapps:
needs: [check]
runs-on: ubuntu-latest
env:
exclude: ${{inputs.par && '-E "_ser"' || ''}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/miniapps/meshing/minimal-surface
- name: Build Miniapps
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v miniapps
- name: Test Miniapps
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
${{env.CTEST}} miniapps ${{env.exclude}} --show-only
${{env.CTEST}} miniapps ${{env.exclude}}
tests-miniapps:
needs: [check]
runs-on: ubuntu-latest
env:
run: ${{inputs.par && '-R "_cpu_np"' || ''}}
exclude: ${{inputs.par && '"unit_tests|debug"' || '"^unit_tests$|debug"'}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/tests/unit/sedov_tests_cpu
- name: Build Tests Unit Miniapps
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v tests/unit/all
- name: Run Tests Unit Miniapps
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
${{env.CTEST}} tests/unit -E ${{env.exclude}} ${{env.run}} --show-only
${{env.CTEST}} tests/unit -E ${{env.exclude}} ${{env.run}}
tests-unit-build:
needs: [check]
runs-on: ubuntu-latest
env:
unit_tests: ${{inputs.par && 'punit_tests' || 'unit_tests'}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/tests/unit/${{env.unit_tests}}
- name: Build Unit Tests
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: ninja -v ${{env.unit_tests}}
- name: Delete object files
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build/tests/unit
run: find . -type f -name '*.o' -delete
- uses: actions/upload-artifact@v4
with:
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build/tests/unit/${{env.unit_tests}}
if-no-files-found: error
retention-days: 1
overwrite: false
tests-unit-run:
needs: [tests-unit-build]
runs-on: ubuntu-latest
strategy:
matrix:
tag: [0, 1, 2, 3]
name: tests-unit-run-${{matrix.tag}}
env:
unit_tests: ${{inputs.par && 'punit_tests' || 'unit_tests'}}
np: ${{inputs.par && '_np=2' || ''}}
steps:
- uses: actions/checkout@v4
- uses: ./.github/actions/sanitize/restore
id: restore
with:
par: ${{inputs.par}}
sanitizer: ${{inputs.sanitizer}}
cache-path: mfem/build/tests/unit/${{env.unit_tests}}
- uses: actions/download-artifact@v4
if: ${{steps.restore.outputs.cache-hit != 'true'}}
with:
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
path: mfem/build/tests/unit
- name: Split Unit Tests
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build/tests/unit
run: |
chmod 755 ${{env.unit_tests}}
./${{env.unit_tests}} --list-test-names-only | tail -n +2 > list-test-names
shuf list-test-names -o list-test-names
split --verbose -n l/4 -d -a 1 list-test-names list-test-names-
- name: Cat Unit Tests ${{matrix.tag}}
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build/tests/unit
run: cat list-test-names-${{matrix.tag}}
- name: Run Unit Tests ${{matrix.tag}}
if: ${{steps.restore.outputs.cache-hit != 'true'}}
working-directory: mfem/build
run: |
${{env.CTEST}} tests/unit -R "${{env.unit_tests}}${{env.np}}" --show-only
${{env.CTEST}} tests/unit -R "${{env.unit_tests}}${{env.np}}"
-73
View File
@@ -1,73 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
---
name: Sanitizers
permissions:
actions: write
on:
push:
branches: ["master", "next"]
pull_request:
workflow_dispatch:
concurrency:
group: ${{github.workflow}}-${{github.ref}}
cancel-in-progress: true
jobs:
# Build steps for dependencies
build-hypre:
uses: ./.github/workflows/sanitize-build-hypre.yml
build-metis:
uses: ./.github/workflows/sanitize-build-metis.yml
build-lsan:
uses: ./.github/workflows/sanitize-build-lsan.yml
build-libcxx:
uses: ./.github/workflows/sanitize-build-libcxx.yml
# Serial sanitizers: asan, msan, ubsan
seq-asan:
needs: [build-libcxx]
uses: ./.github/workflows/sanitize-tests.yml
with:
sanitizer: asan
seq-msan:
needs: [build-libcxx]
uses: ./.github/workflows/sanitize-tests.yml
with:
sanitizer: msan
seq-ubsan:
needs: [build-libcxx]
uses: ./.github/workflows/sanitize-tests.yml
with:
sanitizer: ubsan
# Parallel sanitizers: asan, ubsan
par-asan:
needs: [build-libcxx, build-hypre, build-metis]
uses: ./.github/workflows/sanitize-tests.yml
with:
par: true
sanitizer: asan
par-ubsan:
needs: [build-libcxx, build-hypre, build-metis]
uses: ./.github/workflows/sanitize-tests.yml
with:
par: true
sanitizer: ubsan
+9 -8
View File
@@ -19,6 +19,9 @@ CMakeFiles/
# Clangd server cache
*.cache*
# VSCode configuration
/.vscode/
# Backup files
*~
@@ -211,7 +214,7 @@ miniapps/electromagnetics/joule
miniapps/electromagnetics/Volta-AMR*
miniapps/electromagnetics/Tesla-AMR*
miniapps/electromagnetics/Maxwell-Parallel*
miniapps/electromagnetics/Joule_[0-9]*
miniapps/electromagnetics/Joule_*
miniapps/gslib/field-diff
miniapps/gslib/field-interp
@@ -267,9 +270,9 @@ miniapps/meshing/bounding-box*
miniapps/meshing/jacobian-determinant*
miniapps/mtop/parheat
miniapps/mtop/ParHeat/*
miniapps/mtop/ParHeat*
miniapps/mtop/seqheat
miniapps/mtop/SeqHeat/*
miniapps/mtop/SeqHeat*
miniapps/autodiff/paradiff
miniapps/autodiff/seqadiff
@@ -277,7 +280,7 @@ miniapps/autodiff/seqtest
miniapps/autodiff/par_example
miniapps/autodiff/seq_example
miniapps/autodiff/seq_test
miniapps/autodiff/Example/*
miniapps/autodiff/Exampl*
miniapps/navier/navier_mms
miniapps/navier/navier_kovasznay
@@ -300,7 +303,6 @@ miniapps/nurbs/nurbs_solenoidal
miniapps/nurbs/nurbs_printfunc
miniapps/nurbs/nurbs_patch_ex1
miniapps/nurbs/nurbs_curveint
miniapps/nurbs/nurbs_surface
miniapps/nurbs/refined.mesh
miniapps/nurbs/mesh.*
miniapps/nurbs/sol_?.gf
@@ -319,7 +321,6 @@ miniapps/nurbs/nurbs_naca_cmesh
miniapps/nurbs/naca-cmesh.mesh
miniapps/nurbs/glvis_naca-cmesh.mesh
miniapps/nurbs/Naca_cmesh
miniapps/nurbs/*-Surface.mesh
miniapps/performance/ex1
miniapps/performance/ex1p
@@ -415,8 +416,8 @@ miniapps/diag-smoothers/mg-abs-l1-jacobi
tests/unit/output_meshes
tests/unit/unit_tests
tests/unit/punit_tests
tests/unit/gpu_unit_tests
tests/unit/pgpu_unit_tests
tests/unit/cunit_tests
tests/unit/pcunit_tests
tests/unit/sedov_tests_*
tests/unit/psedov_tests_*
tests/unit/tmop_pa_tests_*
+1 -1
View File
@@ -52,4 +52,4 @@ variables:
- echo ${JOBID}
- echo ${MFEM_DATA_DIR}
- echo ${SPEC}
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) --reservation=ci -t 60 -N 1 tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) --reservation=ci -t 45 -N 1 tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
+1 -26
View File
@@ -29,14 +29,9 @@ Discretization improvements
Meshing improvements
--------------------
- Added support for higher order meshes in Mesh::MakeSimplicial and
ParMesh::MakeSimplicial.
- Added a new miniapp for interpolating a surface grid of points in 3D using a
smooth NURBS surface, that can then be sampled at arbitrary resolution while
staying close to the original geometry. See miniapps/nurbs/nurbs_surface.
GPU computing
-------------
- The function Vector::SetSubVector(const Array<int> &, const real_t) now
@@ -44,13 +39,6 @@ GPU computing
set. This is most often used for setting constant essential boundary
conditions. A new function Vector::SetSubVectorHost has been added in cases
where host execution is always needed (e.g. when the DOFs array is small).
- Introduced MFEM_FOREACH_THREAD_DIRECT, which directly maps loop tasks to GPU
threads, assigning one task per thread.
- Implemented a GPU-accelerated matrix-free AMR derefinement `GridFunction`
update operator. This supports mixed geometry meshes and variable order
spaces, and is the default derefinement operator constructed by
`FiniteElementSpace::Update` and `ParFiniteElementSpace::Update`.
The operator requires `FiniteElementSpace::Nonconforming() == true`.
New and updated examples and miniapps
-------------------------------------
@@ -60,26 +48,13 @@ New and updated examples and miniapps
operators as smoothers.
These miniapps can be found in `miniapps/diag-smoothers`.
API changes
API changes:
-----------
- mfem::internal::tensor and mfem::internal::dual have been moved to
mfem::future::tensor and mfem::future::dual.
- API addition: in class `Operator`, added virtual functions: `AbsMult`, and
`AbsMultTranspose`; in class `Vector`, added `Abs` and `Pow`.
Miscellaneous
-------------
- Added the "gpu", "raja-gpu", and "ceed-gpu" backend aliases/shortcuts which
automatically select between CUDA or HIP.
- The CUDA-specific names used by some of the unit tests like 'cunit_tests' and
'pcunit_tests' were replaced by names using 'gpu' instead of 'c' (short for
CUDA) or 'cuda'. These tests automatically run the CUDA/HIP tests based on the
MFEM build configuration.
- Added the option to enable GPU-aware MPI in MFEM using the environment
variable 'MFEM_GPU_AWARE_MPI' set to any value. Setting this environment
variable is an alternative to calling 'Device::SetGPUAwareMPI(true)'.
- Added parallel Address Sanitizer, serial and parallel Undefined Behavior
Sanitizer and serial Memory Sanitizer GitHub actions tests on Ubuntu.
Version 4.8, released on Apr 9, 2025
====================================
+5 -14
View File
@@ -598,20 +598,14 @@ set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
ALGOIM ENZYME)
# Add all created targets and *_FOUND libraries in the variables TPL_TARGETS and
# TPL_LIBRARIES, respectively.
set(TPL_TARGETS)
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
set(TPL_LIBRARIES "")
set(TPL_INCLUDE_DIRS "")
foreach(TPL IN LISTS MFEM_TPLS)
if (${TPL}_FOUND OR TARGET ${TPL})
if (${TPL}_FOUND)
message(STATUS "MFEM: using package ${TPL}")
if (TARGET ${TPL})
list(APPEND TPL_TARGETS ${TPL})
else()
list(APPEND TPL_LIBRARIES ${${TPL}_LIBRARIES})
list(APPEND TPL_INCLUDE_DIRS ${${TPL}_INCLUDE_DIRS})
endif()
list(APPEND TPL_LIBRARIES ${${TPL}_LIBRARIES})
list(APPEND TPL_INCLUDE_DIRS ${${TPL}_INCLUDE_DIRS})
endif()
endforeach(TPL)
list(REVERSE TPL_LIBRARIES)
@@ -686,10 +680,7 @@ set(MFEM_INSTALL_DIR ${CMAKE_INSTALL_PREFIX})
# Declaring the library
mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES} ${TPL_TARGETS})
if (TPL_TARGETS)
add_dependencies(mfem ${TPL_TARGETS})
endif()
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES})
if (MINGW)
target_link_libraries(mfem PRIVATE ws2_32)
endif()
-15
View File
@@ -121,11 +121,6 @@ Parallel build:
make -j 4
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
Parallel build with fetching of hypre and METIS:
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DFETCH_TPLS=YES
make -j 4
CUDA build:
(this build requires CMake 3.17 or newer)
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
@@ -847,7 +842,6 @@ The specific libraries and their options are:
- HIP (optional), used when MFEM_USE_HIP = YES.
URL: https://rocmdocs.amd.com
Options: HIP_CXX, HIP_ARCH, HIP_OPT, HIP_LIB.
Versions: ROCm >= 5.6.1.
- OCCA (optional), used when MFEM_USE_OCCA = YES.
URL: https://libocca.org
@@ -1080,9 +1074,6 @@ The following options are CMake specific:
MFEM_ENABLE_TESTING - Enable the ctest framework for testing.
MFEM_ENABLE_EXAMPLES - Build all of the examples by default.
MFEM_ENABLE_MINIAPPS - Build all of the miniapps by default.
FETCH_TPLS - Enable fetching of all supported third-party libraries.
HYPRE_FETCH - Enable fetching of hypre.
METIS_FETCH - Enable fetching of metis.
External libraries (CMake):
---------------------------
@@ -1144,12 +1135,6 @@ The following built-in CMake packages are also used:
set the <LIBNAME>_LIBRARIES option directly; the configuration option
<LIBNAME>_DIR is not supported.
The MFEM CMake build system also provides fetching (automated building) for the
packages/libraries listed below. Note that when fetching is enabled, any related
auto-detection functionality is disabled.
- HYPRE
- METIS
Building without GNU make or CMake
==================================
+2 -54
View File
@@ -9,18 +9,15 @@
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Defines the following variables if fetching of TPLs is disabled (default):
# Defines the following variables:
# - HYPRE_FOUND
# - HYPRE_LIBRARIES
# - HYPRE_INCLUDE_DIRS
# - HYPRE_VERSION
# - HYPRE_USING_CUDA (internal)
# - HYPRE_USING_HIP (internal)
# otherwise, the following are defined:
# - HYPRE (imported library target)
# - HYPRE_VERSION (cache variable)
if (HYPRE_FOUND OR TARGET HYPRE)
if (HYPRE_FOUND)
if (HYPRE_USING_CUDA)
find_package(CUDAToolkit REQUIRED)
endif()
@@ -36,55 +33,6 @@ if (HYPRE_FOUND OR TARGET HYPRE)
endif()
endif()
if (HYPRE_FETCH OR FETCH_TPLS)
set(HYPRE_FETCH_VERSION 2.33.0)
add_library(HYPRE STATIC IMPORTED)
# set options and associated dependencies
set(CMAKE_OPTIONS)
list(APPEND CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
if (MFEM_USE_CUDA)
list(APPEND CMAKE_OPTIONS -DHYPRE_WITH_CUDA:BOOL=ON)
find_package(CUDAToolkit REQUIRED)
target_link_libraries(HYPRE INTERFACE CUDA::cusparse CUDA::curand CUDA::cublas)
elseif (MFEM_USE_HIP)
list(APPEND CMAKE_OPTIONS -DHYPRE_WITH_HIP:BOOL=ON)
find_package(rocsparse REQUIRED)
find_package(rocrand REQUIRED)
target_link_libraries(HYPRE INTERFACE rocsparse rocrand)
endif()
if (MFEM_USE_SINGLE)
list(APPEND CMAKE_OPTIONS -DHYPRE_ENABLE_SINGLE:BOOL=ON)
endif()
# define external project and create future include directory so it is present
# to pass CMake checks at end of MFEM configuration step
message(STATUS "Will fetch HYPRE ${HYPRE_FETCH_VERSION} to be built with ${CMAKE_OPTIONS}")
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/hypre)
include(ExternalProject)
ExternalProject_Add(hypre
GIT_REPOSITORY https://github.com/hypre-space/hypre.git
GIT_TAG v${HYPRE_FETCH_VERSION}
GIT_SHALLOW TRUE
UPDATE_DISCONNECTED TRUE
SOURCE_SUBDIR src
PREFIX ${PREFIX}
CMAKE_CACHE_ARGS -DCMAKE_INSTALL_PREFIX:PATH=${PREFIX} -DCMAKE_INSTALL_LIBDIR:PATH=lib ${CMAKE_OPTIONS})
file(MAKE_DIRECTORY ${PREFIX}/include)
# set imported library target properties
add_dependencies(HYPRE hypre)
set_target_properties(HYPRE PROPERTIES
IMPORTED_LOCATION ${PREFIX}/lib/libHYPRE.a
INTERFACE_INCLUDE_DIRECTORIES ${PREFIX}/include)
# convert HYPRE version to integer
string(REGEX MATCHALL "[0-9]+" HYPRE_SPLIT_VERSION ${HYPRE_FETCH_VERSION})
list(GET HYPRE_SPLIT_VERSION 0 HYPRE_MAJOR_VERSION)
list(GET HYPRE_SPLIT_VERSION 1 HYPRE_MINOR_VERSION)
list(GET HYPRE_SPLIT_VERSION 2 HYPRE_PATCH_VERSION)
math(EXPR HYPRE_VERSION "10000*${HYPRE_MAJOR_VERSION} + 100*${HYPRE_MINOR_VERSION} + ${HYPRE_PATCH_VERSION}")
# set cache variables that would otherwise be set after mfem_find_package call
set(HYPRE_VERSION ${HYPRE_VERSION} CACHE STRING "HYPRE version." FORCE)
return()
endif()
include(MfemCmakeUtilities)
mfem_find_package(HYPRE HYPRE HYPRE_DIR "include" "HYPRE.h" "lib" "HYPRE"
"Paths to headers required by HYPRE." "Libraries required by HYPRE."
+1 -29
View File
@@ -9,38 +9,10 @@
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Defines the following variables if fetching of TPLs is disabled (default):
# Defines the following variables:
# - METIS_FOUND
# - METIS_LIBRARIES
# - METIS_INCLUDE_DIRS
# - METIS_VERSION_5
# otherwise, the following are defined:
# - METIS (imported library target)
# - METIS_VERSION_5 (cache variable)
if (METIS_FETCH OR FETCH_TPLS)
set(METIS_FETCH_VERSION 4.0.3)
add_library(METIS STATIC IMPORTED)
# define external project
message(STATUS "Will fetch METIS ${METIS_FETCH_VERSION} to be built with default options")
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/metis)
include(ExternalProject)
ExternalProject_Add(metis
GIT_REPOSITORY https://github.com/mfem/tpls
GIT_TAG b60352fbe9675d374b00828055e55be4584c7995 # tag from 1/16/25
GIT_SHALLOW TRUE
UPDATE_DISCONNECTED TRUE
PREFIX ${PREFIX}
CONFIGURE_COMMAND tar -xzf ../metis/metis-${METIS_FETCH_VERSION}-mac.tgz --strip=1
INSTALL_COMMAND mkdir -p ${PREFIX}/lib && cp libmetis.a ${PREFIX}/lib/)
# set imported library target properties
add_dependencies(METIS metis)
set_target_properties(METIS PROPERTIES
IMPORTED_LOCATION ${PREFIX}/lib/libmetis.a)
# set cache variables that would otherwise be set after mfem_find_package call
set(METIS_VERSION_5 FALSE CACHE BOOL "Is METIS version 5?")
return()
endif()
include(MfemCmakeUtilities)
mfem_find_package(METIS METIS METIS_DIR "include;Lib" "metis.h"
+1 -1
View File
@@ -27,7 +27,7 @@ namespace mfem
{
#if (defined(MFEM_USE_CUDA) && defined(__CUDACC__)) || \
(defined(MFEM_USE_HIP) && defined(__HIP__))
(defined(MFEM_USE_HIP) && defined(__HIPCC__))
#define MFEM_HOST_DEVICE __host__ __device__
#else
#define MFEM_HOST_DEVICE
-6
View File
@@ -89,12 +89,6 @@ option(MFEM_ENABLE_EXAMPLES "Build all of the examples" OFF)
option(MFEM_ENABLE_MINIAPPS "Build all of the miniapps" OFF)
option(MFEM_ENABLE_BENCHMARKS "Build all of the benchmarks" OFF)
# Allow a user to specify fetching of certain third-party libraries instead of
# searching for existing installations.
option(FETCH_TPLS "Enable fetching of all supported third-party libraries" OFF)
option(HYPRE_FETCH "Enable fetching of hypre" OFF)
option(METIS_FETCH "Enable fetching of METIS" OFF)
# Setting CXX/MPICXX on the command line or in user.cmake will overwrite the
# autodetected C++ compiler.
# set(CXX g++)
-3
View File
@@ -82,8 +82,6 @@ set(SRCS
fe/fe_ser.cpp
fe_coll.cpp
fespace.cpp
derefmat_op.cpp
pderefmat_op.cpp
geom.cpp
gridfunc.cpp
hybridization.cpp
@@ -249,7 +247,6 @@ set(HDRS
nonlinearform_ext.hpp
nonlininteg.hpp
qfunction.hpp
qinterp/det.hpp
qinterp/eval.hpp
qinterp/eval_hdiv.hpp
qinterp/grad.hpp
-1
View File
@@ -515,7 +515,6 @@ struct InvTNewtonSolver<Geometry::SEGMENT, SDim, SType, max_team_x>
phys_tol += pptr[idx + d * npts] * pptr[idx + d * npts];
}
phys_tol = fmax(phys_rtol * phys_rtol, phys_tol * phys_rtol * phys_rtol);
hit_bdr[0] = prev_hit_bdr[0] = false;
}
// for each iteration
while (true)
+10 -10
View File
@@ -812,7 +812,7 @@ protected:
const FiniteElement & test_fe) const
{
return (trial_fe.GetDim() == 1 && test_fe.GetDim() == 1 &&
trial_fe.GetDerivType() == mfem::FiniteElement::GRAD &&
trial_fe.GetDerivType() == mfem::FiniteElement::GRAD &&
test_fe.GetRangeType() == mfem::FiniteElement::SCALAR );
}
@@ -884,7 +884,7 @@ protected:
const FiniteElement & trial_fe,
const FiniteElement & test_fe) const
{
return (trial_fe.GetDerivType() == mfem::FiniteElement::DIV &&
return (trial_fe.GetDerivType() == mfem::FiniteElement::DIV &&
test_fe.GetRangeType() == mfem::FiniteElement::SCALAR );
}
@@ -919,7 +919,7 @@ protected:
const FiniteElement & trial_fe,
const FiniteElement & test_fe) const
{
return (trial_fe.GetDerivType() == mfem::FiniteElement::DIV &&
return (trial_fe.GetDerivType() == mfem::FiniteElement::DIV &&
test_fe.GetRangeType() == mfem::FiniteElement::VECTOR );
}
@@ -1600,7 +1600,7 @@ public:
{
return (trial_fe.GetCurlDim() == 3 && test_fe.GetRangeDim() == 3 &&
trial_fe.GetRangeType() == mfem::FiniteElement::VECTOR &&
trial_fe.GetDerivType() == mfem::FiniteElement::CURL &&
trial_fe.GetDerivType() == mfem::FiniteElement::CURL &&
test_fe.GetRangeType() == mfem::FiniteElement::VECTOR );
}
@@ -1635,7 +1635,7 @@ public:
{
return (trial_fe.GetDim() == 2 && test_fe.GetDim() == 2 &&
trial_fe.GetRangeType() == mfem::FiniteElement::VECTOR &&
trial_fe.GetDerivType() == mfem::FiniteElement::CURL &&
trial_fe.GetDerivType() == mfem::FiniteElement::CURL &&
test_fe.GetRangeType() == mfem::FiniteElement::VECTOR );
}
@@ -1669,7 +1669,7 @@ public:
{
return (trial_fe.GetDim() == 2 && test_fe.GetDim() == 2 &&
trial_fe.GetRangeType() == mfem::FiniteElement::SCALAR &&
trial_fe.GetDerivType() == mfem::FiniteElement::GRAD &&
trial_fe.GetDerivType() == mfem::FiniteElement::GRAD &&
test_fe.GetRangeType() == mfem::FiniteElement::SCALAR );
}
@@ -1760,7 +1760,7 @@ public:
const FiniteElement & test_fe) const
{
return (trial_fe.GetRangeType() == mfem::FiniteElement::SCALAR &&
trial_fe.GetDerivType() == mfem::FiniteElement::GRAD &&
trial_fe.GetDerivType() == mfem::FiniteElement::GRAD &&
test_fe.GetRangeType() == mfem::FiniteElement::SCALAR );
}
@@ -1793,7 +1793,7 @@ public:
const FiniteElement & test_fe) const
{
return (trial_fe.GetRangeType() == mfem::FiniteElement::SCALAR &&
trial_fe.GetDerivType() == mfem::FiniteElement::GRAD &&
trial_fe.GetDerivType() == mfem::FiniteElement::GRAD &&
test_fe.GetRangeType() == mfem::FiniteElement::VECTOR &&
test_fe.GetDerivType() == mfem::FiniteElement::DIV );
}
@@ -1832,7 +1832,7 @@ public:
const FiniteElement & test_fe) const
{
return (trial_fe.GetRangeType() == mfem::FiniteElement::VECTOR &&
trial_fe.GetDerivType() == mfem::FiniteElement::DIV &&
trial_fe.GetDerivType() == mfem::FiniteElement::DIV &&
test_fe.GetRangeType() == mfem::FiniteElement::SCALAR &&
test_fe.GetDerivType() == mfem::FiniteElement::GRAD
);
@@ -1973,7 +1973,7 @@ protected:
const FiniteElement & test_fe) const override
{
return (trial_fe.GetCurlDim() == 3 && test_fe.GetRangeDim() == 3 &&
trial_fe.GetDerivType() == mfem::FiniteElement::CURL &&
trial_fe.GetDerivType() == mfem::FiniteElement::CURL &&
test_fe.GetRangeType() == mfem::FiniteElement::VECTOR );
}
+4 -4
View File
@@ -912,7 +912,7 @@ ConduitDataCollection::GridFunctionToBlueprintField(mfem::GridFunction *gf,
if (vdim == 1) // scalar case
{
n_field["values"].set_external(const_cast<real_t *>(gf->HostRead()),
n_field["values"].set_external(gf->GetData(),
ndofs);
}
else // vector case
@@ -925,18 +925,18 @@ ConduitDataCollection::GridFunctionToBlueprintField(mfem::GridFunction *gf,
int vdim_stride = (ordering == Ordering::byNODES ? ndofs : 1);
index_t offset = 0;
index_t stride = sizeof(real_t) * entry_stride;
index_t stride = sizeof(double) * entry_stride;
for (int d = 0; d < vdim; d++)
{
std::ostringstream oss;
oss << "v" << d;
std::string comp_name = oss.str();
n_field["values"][comp_name].set_external(const_cast<real_t *>(gf->HostRead()),
n_field["values"][comp_name].set_external(gf->GetData(),
ndofs,
offset,
stride);
offset += sizeof(real_t) * vdim_stride;
offset += sizeof(double) * vdim_stride;
}
}
+10 -12
View File
@@ -764,9 +764,9 @@ ParaViewDataCollectionBase::ParaViewDataCollectionBase(
{
cycle = 0;
#ifdef MFEM_USE_ZLIB
// If we have zlib, enable compression. Otherwise, compression is disabled in
// the DataCollection base class constructor.
compression = true;
compression = true; // if we have zlib, enable compression
#else
compression = false; // otherwise, disable compression
#endif
}
@@ -784,8 +784,13 @@ void ParaViewDataCollectionBase::SetCompressionLevel(int compression_level_)
{
MFEM_ASSERT(compression_level_ >= -1 && compression_level_ <= 9,
"Compression level must be between -1 and 9 (inclusive).");
if (compression_level_ != 0) { SetCompression(true);}
compression_level = compression_level_;
compression = compression_level_ != 0;
}
void ParaViewDataCollectionBase::SetCompression(bool compression_)
{
compression = compression_;
}
int ParaViewDataCollectionBase::GetCompressionLevel() const
@@ -1169,14 +1174,7 @@ const char *ParaViewDataCollection::GetDataTypeString() const
ParaViewHDFDataCollection::ParaViewHDFDataCollection(
const std::string &collection_name, Mesh *mesh)
: ParaViewDataCollectionBase(collection_name, mesh)
{
compression = true;
}
void ParaViewHDFDataCollection::SetCompression(bool compression_)
{
compression = compression_;
}
{ }
void ParaViewHDFDataCollection::EnsureVTKHDF()
{
+7 -6
View File
@@ -537,6 +537,13 @@ public:
/// Any nonzero compression level will enable compression.
void SetCompressionLevel(int compression_level_);
/// @brief Enable or disable zlib compression.
///
/// If the input is true, use the default zlib compression level (unless the
/// compression level has previously been set by calling
/// SetCompressionLevel()).
void SetCompression(bool compression_) override;
/// @brief Sets whether or not to output the data as high-order elements
/// (false by default).
///
@@ -626,12 +633,6 @@ public:
ParaViewHDFDataCollection(const std::string& collection_name,
Mesh *mesh_ = nullptr);
/// @brief Enable or disable compression.
///
/// The compression level can be set with SetCompressionLevel()). VTKHDF
/// compression does not require MFEM to be compiled with zlib support.
void SetCompression(bool compression_) override;
/// Save the collection.
void Save() override;
-266
View File
@@ -1,266 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "derefmat_op.hpp"
#include "fes_kernels.hpp"
/// \cond DO_NOT_DOCUMENT
namespace mfem
{
namespace internal
{
template <Ordering::Type Order, bool Atomic>
static void DerefMultKernelImpl(const DerefineMatrixOp &op, const Vector &x,
Vector &y)
{
DerefineMatrixOpMultFunctor<Order, Atomic> func;
func.xptr = x.Read();
y.UseDevice();
y = 0.;
func.yptr = y.ReadWrite();
func.bsptr = op.block_storage.Read();
func.boptr = op.block_offsets.Read();
func.brptr = op.block_row_idcs_offsets.Read();
func.bcptr = op.block_col_idcs_offsets.Read();
func.rptr = op.row_idcs.Read();
func.cptr = op.col_idcs.Read();
func.vdims = op.fespace->GetVDim();
func.nblocks = op.block_offsets.Size();
func.width = op.Width() / func.vdims;
func.height = op.Height() / func.vdims;
func.Run(op.max_rows);
}
} // namespace internal
DerefineMatrixOp::DerefineMatrixOp(FiniteElementSpace &fespace_, int old_ndofs,
const Table *old_elem_dof,
const Table *old_elem_fos)
: Operator(fespace_.GetVSize(), old_ndofs * fespace_.GetVDim()),
fespace(&fespace_)
{
static Kernels kernels;
constexpr int max_team_size = 256;
/// TODO: Implement DofTransformation support
MFEM_VERIFY(fespace->Nonconforming(),
"Not implemented for conforming meshes.");
MFEM_VERIFY(old_ndofs, "Missing previous (finer) space.");
MFEM_VERIFY(fespace->GetNDofs() <= old_ndofs,
"Previous space is not finer.");
const CoarseFineTransformations &dtrans =
fespace->GetMesh()->ncmesh->GetDerefinementTransforms();
MFEM_ASSERT(dtrans.embeddings.Size() == old_elem_dof->Size(), "");
const bool is_dg = fespace->FEColl()->GetContType()
== FiniteElementCollection::DISCONTINUOUS;
DenseMatrix localRVO; // for variable-order only
DenseTensor localR[Geometry::NumGeom];
int total_rows = 0;
int total_cols = 0;
block_offsets.SetSize(dtrans.embeddings.Size());
block_offsets.HostWrite();
if (fespace->IsVariableOrder())
{
// TODO: any potential for some compression here?
// determine storage size and offsets
block_offsets[0] = 0;
int total_size = 0;
for (int k = 0; k < dtrans.embeddings.Size(); ++k)
{
const Embedding &emb = dtrans.embeddings[k];
const FiniteElement *fe = fespace->GetFE(emb.parent);
const int ldof = fe->GetDof();
if (k + 1 < dtrans.embeddings.Size())
{
block_offsets[k + 1] = block_offsets[k] + ldof * ldof;
}
total_rows += ldof;
total_cols += ldof;
total_size += ldof * ldof;
}
block_storage.SetSize(total_size);
}
else
{
// compression scheme:
// block_offsets is the start of each block, potentially repeated
// only need to store localR for used shapes
Mesh::GeometryList elem_geoms(*fespace->GetMesh());
int geom_offsets[Geometry::NumGeom];
{
int size = 0;
for (int i = 0; i < elem_geoms.Size(); ++i)
{
fespace->GetLocalDerefinementMatrices(elem_geoms[i],
localR[elem_geoms[i]]);
geom_offsets[elem_geoms[i]] = size;
size += localR[elem_geoms[i]].TotalSize();
}
block_storage.SetSize(size);
// copy blocks into block_storage
auto bs_ptr = block_storage.HostWrite();
for (int i = 0; i < elem_geoms.Size(); ++i)
{
std::copy(localR[elem_geoms[i]].Data(),
localR[elem_geoms[i]].Data()
+ localR[elem_geoms[i]].TotalSize(),
bs_ptr);
bs_ptr += localR[elem_geoms[i]].TotalSize();
}
}
for (int k = 0; k < dtrans.embeddings.Size(); ++k)
{
const Embedding &emb = dtrans.embeddings[k];
Geometry::Type geom =
fespace->GetMesh()->GetElementBaseGeometry(emb.parent);
auto size = localR[geom].SizeI() * localR[geom].SizeJ();
total_rows += localR[geom].SizeI();
total_cols += localR[geom].SizeJ();
// set block offsets and sizes
block_offsets[k] = geom_offsets[geom] + size * emb.matrix;
}
}
row_idcs.SetSize(total_rows);
row_idcs.HostWrite();
col_idcs.SetSize(total_cols);
col_idcs.HostWrite();
block_row_idcs_offsets.SetSize(dtrans.embeddings.Size() + 1);
block_row_idcs_offsets.HostWrite();
block_col_idcs_offsets.SetSize(dtrans.embeddings.Size() + 1);
block_col_idcs_offsets.HostWrite();
block_row_idcs_offsets[0] = 0;
block_col_idcs_offsets[0] = 0;
// compute index information
Array<int> dofs, old_dofs;
max_rows = 1;
{
Array<int> mark(fespace->GetNDofs());
mark = 0;
auto bs_ptr = block_storage.HostWrite();
int ridx = 0;
int cidx = 0;
int num_marked = 0;
for (int k = 0; k < dtrans.embeddings.Size(); k++)
{
const Embedding &emb = dtrans.embeddings[k];
Geometry::Type geom =
fespace->GetMesh()->GetElementBaseGeometry(emb.parent);
if (fespace->IsVariableOrder())
{
const FiniteElement *fe = fespace->GetFE(emb.parent);
const DenseTensor &pmats = dtrans.point_matrices[geom];
const int ldof = fe->GetDof();
IsoparametricTransformation isotr;
isotr.SetIdentityTransformation(geom);
localRVO.SetSize(ldof, ldof);
isotr.SetPointMat(pmats(emb.matrix));
// Local restriction is size ldofxldof assuming that the parent
// and child are of same polynomial order.
fe->GetLocalRestriction(isotr, localRVO);
// copy block
auto size = localRVO.Height() * localRVO.Width();
std::copy(localRVO.Data(), localRVO.Data() + size, bs_ptr);
bs_ptr += size;
}
DenseMatrix &lR =
fespace->IsVariableOrder() ? localRVO : localR[geom](emb.matrix);
block_row_idcs_offsets[k + 1] =
block_row_idcs_offsets[k] + lR.Height();
block_col_idcs_offsets[k + 1] = block_col_idcs_offsets[k] + lR.Width();
max_rows = std::max(lR.Height(), max_rows);
// index information
fespace->elem_dof->GetRow(emb.parent, dofs);
old_elem_dof->GetRow(k, old_dofs);
MFEM_VERIFY(old_dofs.Size() == dofs.Size(),
"Parent and child must have same #dofs.");
for (int i = 0; i < lR.Height(); ++i, ++ridx)
{
if (!std::isfinite(lR(i, 0)))
{
row_idcs[ridx] = INT_MAX;
continue;
}
int r = dofs[i];
int m = (r >= 0) ? r : (-1 - r);
if (is_dg || !mark[m])
{
row_idcs[ridx] = r;
mark[m] = 1;
++num_marked;
}
else
{
row_idcs[ridx] = INT_MAX;
}
}
for (int i = 0; i < lR.Width(); ++i, ++cidx)
{
col_idcs[cidx] = old_dofs[i];
}
}
if (!is_dg && !fespace->IsVariableOrder())
{
MFEM_VERIFY(num_marked * fespace->GetVDim() == Height(),
"internal error: not all rows were set.");
}
}
// if not using GPU, set max_rows/max_cols to zero
if (Device::Allows(Backend::DEVICE_MASK))
{
max_rows = std::min(max_rows, max_team_size);
}
else
{
max_rows = 1;
}
}
void DerefineMatrixOp::Mult(const Vector &x, Vector &y) const
{
const bool is_dg = fespace->FEColl()->GetContType()
== FiniteElementCollection::DISCONTINUOUS;
// DG needs atomic summation
MultKernel::Run(fespace->GetOrdering(), is_dg, *this, x, y);
}
DerefineMatrixOp::Kernels::Kernels()
{
MultKernel::Specialization<Ordering::byNODES, false>::Add();
MultKernel::Specialization<Ordering::byVDIM, false>::Add();
MultKernel::Specialization<Ordering::byNODES, true>::Add();
MultKernel::Specialization<Ordering::byVDIM, true>::Add();
}
template <Ordering::Type Order, bool Atomic>
DerefineMatrixOp::MultKernelType DerefineMatrixOp::MultKernel::Kernel()
{
return internal::DerefMultKernelImpl<Order, Atomic>;
}
DerefineMatrixOp::MultKernelType
DerefineMatrixOp::MultKernel::Fallback(Ordering::Type, bool)
{
MFEM_ABORT("invalid MultKernel parameters");
}
} // namespace mfem
/// \endcond DO_NOT_DOCUMENT
-65
View File
@@ -1,65 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_DEREFMAT_OP
#define MFEM_DEREFMAT_OP
#include "fespace.hpp"
#include "kernel_dispatch.hpp"
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
struct DerefineMatrixOp : public Operator
{
FiniteElementSpace *fespace;
/// offsets into block_storage
Array<int> block_offsets;
/// offsets into row_idcs
Array<int> block_row_idcs_offsets;
/// offsets into col_idcs
Array<int> block_col_idcs_offsets;
/// mapping for row dofs, INT_MAX indicates the block row should be ignored.
/// negative means the row data should be negated.
Array<int> row_idcs;
/// mapping for col dofs, negative means the col data should be negated.
Array<int> col_idcs;
/// dense block matrices which can be reused to construct the full matrix
/// operation. These are stored contiguously and blocks have no restrictions
/// on shape (can be rectangle and differ from block to block).
Vector block_storage;
/// maximum height of any block in block_storage for GPU
/// parallelization, or 1 for CPU runs.
int max_rows;
using MultKernelType = void (*)(const DerefineMatrixOp &, const Vector &,
Vector &);
/// template args: ordering, atomic
MFEM_REGISTER_KERNELS(MultKernel, MultKernelType, (Ordering::Type, bool));
struct Kernels
{
Kernels();
};
void Mult(const Vector &x, Vector &y) const;
DerefineMatrixOp(FiniteElementSpace &fespace_, int old_ndofs,
const Table *old_elem_dof, const Table *old_elem_fos);
};
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
-1
View File
@@ -241,7 +241,6 @@ public:
{
MFEM_ASSERT(!action_callbacks.empty(), "no integrators have been set");
prolongation(solutions, solutions_t, solutions_l);
residual_l = 0.0;
for (auto &action : action_callbacks)
{
action(solutions_l, parameters_l, residual_l);
+6 -7
View File
@@ -568,7 +568,7 @@ struct ThreadBlocks
int z = 1;
};
#if defined(MFEM_USE_CUDA_OR_HIP)
#if (defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
template <typename func_t>
__global__ void forall_kernel_shmem(func_t f, int n)
{
@@ -591,7 +591,7 @@ void forall(func_t f,
if (Device::Allows(Backend::CUDA_MASK) ||
Device::Allows(Backend::HIP_MASK))
{
#if defined(MFEM_USE_CUDA_OR_HIP)
#if (defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
// int gridsize = (N + Z - 1) / Z;
int num_bytes = num_shmem * sizeof(decltype(shmem));
dim3 block_size(blocks.x, blocks.y, blocks.z);
@@ -987,7 +987,7 @@ get_restriction_transpose(
{
auto RT = [=](const Vector &v_e, Vector &v_l)
{
v_l += v_e;
v_l = v_e;
};
return std::make_tuple(RT, 1);
}
@@ -996,7 +996,7 @@ get_restriction_transpose(
const Operator *R = get_restriction<entity_t>(f, o);
std::function<void(const Vector&, Vector&)> RT = [=](const Vector &x, Vector &y)
{
R->AddMultTranspose(x, y);
R->MultTranspose(x, y);
};
return std::make_tuple(RT, R->Height());
}
@@ -1708,7 +1708,6 @@ std::array<DofToQuadMap, N> load_dtq_mem(
const auto B = Reshape(&dtq[i].B[0], nqp_b, dim_b, ndof_b);
auto mem_Bi = Reshape(reinterpret_cast<real_t *>(mem) + offset, nqp_b, dim_b,
ndof_b);
MFEM_FOREACH_THREAD(q, x, nqp_b)
{
MFEM_FOREACH_THREAD(d, y, ndof_b)
@@ -2159,7 +2158,7 @@ template <
std::size_t... Is>
std::array<DofToQuadMap, N> create_dtq_maps_impl(
field_operator_ts &fops,
std::vector<const DofToQuad*> &dtqs,
std::vector<const DofToQuad*> dtqs,
const std::array<int, N> &field_map,
std::index_sequence<Is...>)
{
@@ -2244,7 +2243,7 @@ template <
std::size_t num_fields>
std::array<DofToQuadMap, num_fields> create_dtq_maps(
field_operator_ts &fops,
std::vector<const DofToQuad*> &dtqmaps,
std::vector<const DofToQuad*> dtqmaps,
const std::array<int, num_fields> &to_field_map)
{
return create_dtq_maps_impl<entity_t>(
-249
View File
@@ -1,249 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_FES_KERNELS_HPP
#define MFEM_FES_KERNELS_HPP
#include "../general/forall.hpp"
#include <climits>
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
///
/// Implements matrix-vector multiply $y = A x$ for a sparse matrix composed of
/// a sum of smaller dense blocks. There is additional permutation/sign
/// information associated with each block. The base class only implements
/// helper routines such as computing block widths, index into x, index into y,
/// and column in A given sub-block information.
/// @sa DerefineMatrixOpMultFunctor
///
/// @tparam Order vdim ordering for x and y. Note that for Diag = false this is
/// ignored for x as x has a special interleaved order.
/// @tparam Base used for the curious recurring template pattern (CRTP) so the
/// base class can access child class fields without virtual functions
/// @tparam Diag true if this corresponds to the diagonal block (coarse element
/// and fine element are on our rank), false otherwise (coarse element is on our
/// rank, fine element is on a different rank).
///
template <Ordering::Type Order, class Base, bool Diag = true>
struct DerefineMatrixOpFunctorBase;
template <class Base>
struct DerefineMatrixOpFunctorBase<Ordering::byNODES, Base, true>
{
/// block column indices offsets
const int *bcptr;
/// column indices
const int *cptr;
int MFEM_HOST_DEVICE BlockWidth(int k) const
{
return bcptr[k + 1] - bcptr[k];
}
void MFEM_HOST_DEVICE Col(int j, int k, int &col, int &sign) const
{
col = cptr[bcptr[k] + j];
if (col < 0)
{
col = -1 - col;
sign = -sign;
}
}
int MFEM_HOST_DEVICE IndexX(int col, int vdim, int) const
{
return col + vdim * static_cast<const Base *>(this)->width;
}
int MFEM_HOST_DEVICE IndexY(int row, int vdim) const
{
return row + vdim * static_cast<const Base *>(this)->height;
}
};
template <class Base>
struct DerefineMatrixOpFunctorBase<Ordering::byVDIM, Base, true>
{
/// block column indices offsets
const int *bcptr;
/// column indices
const int *cptr;
int MFEM_HOST_DEVICE BlockWidth(int k) const
{
return bcptr[k + 1] - bcptr[k];
}
void MFEM_HOST_DEVICE Col(int j, int k, int &col, int &sign) const
{
col = cptr[bcptr[k] + j];
if (col < 0)
{
col = -1 - col;
sign = -sign;
}
}
int MFEM_HOST_DEVICE IndexX(int col, int vdim, int) const
{
return vdim + col * static_cast<const Base *>(this)->vdims;
}
int MFEM_HOST_DEVICE IndexY(int row, int vdim) const
{
return vdim + row * static_cast<const Base *>(this)->vdims;
}
};
template <class Base>
struct DerefineMatrixOpFunctorBase<Ordering::byNODES, Base, false>
{
/// receive segment offsets
const int *segptr;
/// receive segment index
const int *rsptr;
/// off-diagonal block column offsets
const int *coptr;
/// off-diagonal block widths
const int *bwptr;
int MFEM_HOST_DEVICE BlockWidth(int k) const { return bwptr[k]; }
void MFEM_HOST_DEVICE Col(int j, int k, int &col, int &sign) const
{
col = coptr[k] + j;
}
int MFEM_HOST_DEVICE IndexX(int col, int vdim, int k) const
{
int tmp = rsptr[k];
int segwidth = segptr[tmp + 1] - segptr[tmp];
return segptr[tmp] * static_cast<const Base *>(this)->vdims + col +
vdim * segwidth;
}
int MFEM_HOST_DEVICE IndexY(int row, int vdim) const
{
return row + vdim * static_cast<const Base *>(this)->height;
}
};
template <class Base>
struct DerefineMatrixOpFunctorBase<Ordering::byVDIM, Base, false>
{
/// receive segment offsets
const int *segptr;
/// receive segment index
const int *rsptr;
/// off-diagonal block column offsets
const int *coptr;
/// off-diagonal block widths
const int *bwptr;
int MFEM_HOST_DEVICE BlockWidth(int k) const { return bwptr[k]; }
void MFEM_HOST_DEVICE Col(int j, int k, int &col, int &sign) const
{
col = coptr[k] + j;
}
int MFEM_HOST_DEVICE IndexX(int col, int vdim, int k) const
{
int tmp = rsptr[k];
int segwidth = segptr[tmp + 1] - segptr[tmp];
return segptr[tmp] * static_cast<const Base *>(this)->vdims + col +
vdim * segwidth;
}
int MFEM_HOST_DEVICE IndexY(int row, int vdim) const
{
return vdim + row * static_cast<const Base *>(this)->vdims;
}
};
/// internally used to implement the derefinement operator Mult diagonal
/// block
template <Ordering::Type Order, bool Atomic, bool Diag = true>
struct DerefineMatrixOpMultFunctor
: public DerefineMatrixOpFunctorBase<
Order, DerefineMatrixOpMultFunctor<Order, Atomic, Diag>, Diag>
{
const real_t *xptr;
real_t *yptr;
/// block storage
const real_t *bsptr;
/// block offsets
const int *boptr;
/// block row index offsets
const int *brptr;
/// row indices
const int *rptr;
// number of blocks
int nblocks;
// number of components
int vdims;
/// overall operator height (for vdim = 1)
int height;
/// overall operator width (for vdim = 1)
int width;
void MFEM_HOST_DEVICE operator()(int kidx) const
{
int k = kidx % nblocks;
int vdim = kidx / nblocks;
int block_height = brptr[k + 1] - brptr[k];
int block_width = this->BlockWidth(k);
MFEM_FOREACH_THREAD(i, x, block_height)
{
int row = rptr[brptr[k] + i];
int rsign = 1;
if (row < 0)
{
row = -1 - row;
rsign = -1;
}
if (row < INT_MAX)
{
// row not marked as unused
real_t sum = 0;
for (int j = 0; j < block_width; ++j)
{
int col, sign = rsign;
this->Col(j, k, col, sign);
sum += sign * bsptr[boptr[k] + i + j * block_height] *
xptr[this->IndexX(col, vdim, k)];
}
#if defined(__CUDA_ARCH__) or defined(__HIP_DEVICE_COMPILE__)
if (Atomic)
{
atomicAdd(yptr + this->IndexY(row, vdim), sum);
}
else
#endif
{
yptr[this->IndexY(row, vdim)] += sum;
}
}
}
}
/// N is the max block row size (doesn't have to be a power of 2)
void Run(int N) const { forall_2D(nblocks * vdims, N, 1, *this); }
};
} // namespace internal
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
+6 -13
View File
@@ -17,9 +17,6 @@
#include "fem.hpp"
#include "ceed/interface/util.hpp"
#include "derefmat_op.hpp"
#include <algorithm>
#include <cmath>
#include <cstdarg>
@@ -27,9 +24,9 @@ using namespace std;
namespace mfem
{
template <>
void Ordering::DofsToVDofs<Ordering::byNODES>(int ndofs, int vdim,
Array<int> &dofs)
template <> void Ordering::
DofsToVDofs<Ordering::byNODES>(int ndofs, int vdim, Array<int> &dofs)
{
// static method
int size = dofs.Size();
@@ -43,9 +40,8 @@ void Ordering::DofsToVDofs<Ordering::byNODES>(int ndofs, int vdim,
}
}
template <>
void Ordering::DofsToVDofs<Ordering::byVDIM>(int ndofs, int vdim,
Array<int> &dofs)
template <> void Ordering::
DofsToVDofs<Ordering::byVDIM>(int ndofs, int vdim, Array<int> &dofs)
{
// static method
int size = dofs.Size();
@@ -59,6 +55,7 @@ void Ordering::DofsToVDofs<Ordering::byVDIM>(int ndofs, int vdim,
}
}
FiniteElementSpace::FiniteElementSpace()
: mesh(NULL), fec(NULL), vdim(0), ordering(Ordering::byNODES),
ndofs(0), nvdofs(0), nedofs(0), nfdofs(0), nbdofs(0),
@@ -4247,11 +4244,7 @@ void FiniteElementSpace::Update(bool want_transform)
case Mesh::DEREFINE:
{
BuildConformingInterpolation();
#if 0
Th.Reset(DerefinementMatrix(old_ndofs, old_elem_dof, old_elem_fos));
#else
Th.Reset(new DerefineMatrixOp(*this, old_ndofs, old_elem_dof, old_elem_fos));
#endif
if (IsVariableOrder())
{
if (cP && cR_hp)
+1 -2
View File
@@ -113,7 +113,7 @@ class QuadratureSpace;
class QuadratureInterpolator;
class FaceQuadratureInterpolator;
class PRefinementTransferOperator;
struct DerefineMatrixOp;
/** @brief Class FiniteElementSpace - responsible for providing FEM view of the
mesh, mainly managing the set of degrees of freedom.
@@ -246,7 +246,6 @@ class FiniteElementSpace
friend class PRefinementTransferOperator;
friend void Mesh::Swap(Mesh &, bool);
friend class LORBase;
friend struct DerefineMatrixOp;
protected:
/// The mesh that FE space lives on (not owned).
+1 -1
View File
@@ -4334,7 +4334,7 @@ real_t LSZZErrorEstimator(BilinearFormIntegrator &blfi, // input
u.GetSubVector(udofs, ul);
utrans.InvTransformPrimal(ul);
Transf = ufes->GetElementTransformation(ielem);
const auto *dummy = ufes->GetFE(ielem);
FiniteElement *dummy = nullptr;
blfi.ComputeElementFlux(*ufes->GetFE(ielem), *Transf, ul,
*dummy, fl, with_coeff, ir);
+25 -26
View File
@@ -1009,7 +1009,6 @@ inline void SmemPADiffusionApply3D(const int NE,
auto d = Reshape(d_.Read(), Q1D, Q1D, Q1D, symmetric ? 6 : 9, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, NE);
MFEM_VERIFY(D1D <= Q1D, "THREAD_DIRECT requires D1D <= Q1D");
mfem::forall_3D(NE, Q1D, Q1D, Q1D, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -1039,11 +1038,11 @@ inline void SmemPADiffusionApply3D(const int NE,
real_t (*QDD0)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+0);
real_t (*QDD1)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+1);
real_t (*QDD2)[MD1][MD1] = (real_t (*)[MD1][MD1]) (sm0+2);
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx,x,D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
X[dz][dy][dx] = x(dx,dy,dz,e);
}
@@ -1051,9 +1050,9 @@ inline void SmemPADiffusionApply3D(const int NE,
}
if (MFEM_THREAD_ID(z) == 0)
{
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
B[qx][dy] = b(qx,dy);
G[qx][dy] = g(qx,dy);
@@ -1061,11 +1060,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t u = 0.0, v = 0.0;
MFEM_UNROLL(MD1)
@@ -1081,11 +1080,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(MD1)
@@ -1102,11 +1101,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(MD1)
@@ -1137,9 +1136,9 @@ inline void SmemPADiffusionApply3D(const int NE,
MFEM_SYNC_THREAD;
if (MFEM_THREAD_ID(z) == 0)
{
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx,x,Q1D)
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
Bt[dy][qx] = b(qx,dy);
Gt[dy][qx] = g(qx,dy);
@@ -1147,11 +1146,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy,y,Q1D)
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx,x,D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(MQ1)
@@ -1168,11 +1167,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(qz,z,Q1D)
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx,x,D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(Q1D)
@@ -1189,11 +1188,11 @@ inline void SmemPADiffusionApply3D(const int NE,
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dz,z,D1D)
MFEM_FOREACH_THREAD(dz,z,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy,y,D1D)
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx,x,D1D)
MFEM_FOREACH_THREAD(dx,x,D1D)
{
real_t u = 0.0, v = 0.0, w = 0.0;
MFEM_UNROLL(MQ1)
+1 -1
View File
@@ -62,7 +62,7 @@ void MassIntegrator::AssemblePA(const FiniteElementSpace &fes)
const int NE = ne;
const int Q1D = quad1D;
const int NQ = static_cast<int>(std::pow(Q1D, dim));
const int NQ = pow(Q1D, dim);
const bool const_c = coeff.Size() == 1;
const bool by_val = map_type == FiniteElement::VALUE;
const auto W = Reshape(ir->GetWeights().Read(), NQ);
+29 -29
View File
@@ -346,13 +346,13 @@ private:
template<typename T>
T operator() (const blitz::TinyVector<T,3>& x) const
{
const int el_order = el->GetOrder();
std::vector<T> u1(el_order+1);
std::vector<T> u2(el_order+1);
std::vector<T> u3(el_order+1);
TmplPoly_1D::CalcBernstein(el_order, x[0], u1.data());
TmplPoly_1D::CalcBernstein(el_order, x[1], u2.data());
TmplPoly_1D::CalcBernstein(el_order, x[2], u3.data());
int el_order=el->GetOrder();
T u1[el_order+1];
T u2[el_order+1];
T u3[el_order+1];
TmplPoly_1D::CalcBernstein(el_order, x[0], u1);
TmplPoly_1D::CalcBernstein(el_order, x[1], u2);
TmplPoly_1D::CalcBernstein(el_order, x[2], u3);
const Array<int>& dof_map=el->GetDofMap();
@@ -370,17 +370,17 @@ private:
template<typename T>
blitz::TinyVector<T,3> grad(const blitz::TinyVector<T,3>& x) const
{
const int el_order = el->GetOrder();
std::vector<T> u1(el_order+1);
std::vector<T> u2(el_order+1);
std::vector<T> u3(el_order+1);
std::vector<T> d1(el_order+1);
std::vector<T> d2(el_order+1);
std::vector<T> d3(el_order+1);
int el_order=el->GetOrder();
T u1[el_order+1];
T u2[el_order+1];
T u3[el_order+1];
T d1[el_order+1];
T d2[el_order+1];
T d3[el_order+1];
TmplPoly_1D::CalcBernstein(el_order,x[0], u1.data(), d1.data());
TmplPoly_1D::CalcBernstein(el_order,x[1], u2.data(), d2.data());
TmplPoly_1D::CalcBernstein(el_order,x[2], u3.data(), d3.data());
TmplPoly_1D::CalcBernstein(el_order,x[0], u1, d1);
TmplPoly_1D::CalcBernstein(el_order,x[1], u2, d2);
TmplPoly_1D::CalcBernstein(el_order,x[2], u3, d3);
blitz::TinyVector<T,3> res(T(0.0),T(0.0),T(0.0));
@@ -415,11 +415,11 @@ private:
template<typename T>
T operator() (const blitz::TinyVector<T,2>& x) const
{
const int el_order = el->GetOrder();
std::vector<T> u1(el_order+1);
std::vector<T> u2(el_order+1);
TmplPoly_1D::CalcBernstein(el_order, x[0], u1.data());
TmplPoly_1D::CalcBernstein(el_order, x[1], u2.data());
int el_order=el->GetOrder();
T u1[el_order+1];
T u2[el_order+1];
TmplPoly_1D::CalcBernstein(el_order, x[0], u1);
TmplPoly_1D::CalcBernstein(el_order, x[1], u2);
const Array<int>& dof_map=el->GetDofMap();
@@ -437,14 +437,14 @@ private:
template<typename T>
blitz::TinyVector<T,2> grad(const blitz::TinyVector<T,2>& x) const
{
const int el_order = el->GetOrder();
std::vector<T> u1(el_order+1);
std::vector<T> u2(el_order+1);
std::vector<T> d1(el_order+1);
std::vector<T> d2(el_order+1);
int el_order=el->GetOrder();
T u1[el_order+1];
T u2[el_order+1];
T d1[el_order+1];
T d2[el_order+1];
TmplPoly_1D::CalcBernstein(el_order,x[0], u1.data(), d1.data());
TmplPoly_1D::CalcBernstein(el_order,x[1], u2.data(), d2.data());
TmplPoly_1D::CalcBernstein(el_order,x[0], u1, d1);
TmplPoly_1D::CalcBernstein(el_order,x[1], u2, d2);
blitz::TinyVector<T,2> res(T(0.0),T(0.0));
+1 -1
View File
@@ -673,7 +673,7 @@ public:
int myid;
MPI_Comm_rank(comm, &myid);
int seed = (seed_ > 0) ? seed_ + myid : time(nullptr) + myid;
int seed = (seed_ > 0) ? seed_ + myid : (int)time(0) + myid;
SetSeed(seed);
}
#else
-591
View File
@@ -1,591 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "pderefmat_op.hpp"
#ifdef MFEM_USE_MPI
#include "fes_kernels.hpp"
/// \cond DO_NOT_DOCUMENT
namespace mfem
{
namespace internal
{
template <Ordering::Type Order, bool Atomic>
static void ParDerefMultKernelImpl(const ParDerefineMatrixOp &op,
const Vector &x, Vector &y)
{
// pack sends
if (op.xghost_send.Size())
{
auto src = x.Read();
auto idcs = op.send_permutations.Read();
auto dst = Device::GetGPUAwareMPI() ? op.xghost_send.Write()
: op.xghost_send.HostWrite();
auto vdims = op.fespace->GetVDim();
auto sptr = op.send_segment_idcs.Read();
auto lptr = op.send_segments.Read();
auto old_ndofs = x.Size() / vdims;
forall(op.send_permutations.Size(), [=] MFEM_HOST_DEVICE(int i)
{
int seg = sptr[i];
int width = lptr[seg + 1] - lptr[seg];
auto tdst = dst + i + lptr[seg] * vdims;
int sign = 1;
int col = idcs[i];
if (col < 0)
{
sign = -1;
col = -1 - col;
}
for (int vdim = 0; vdim < vdims; ++vdim)
{
tdst[vdim * width] =
sign
* src[Order == Ordering::byNODES ? (col + vdim * old_ndofs)
: (col * vdims + vdim)];
}
});
// TODO: is this needed so we can send the packed data correctly?
// unclear for GPU-aware MPI, definitely required otherwise
MFEM_DEVICE_SYNC;
}
// initialize off-diagonal receive and send
op.requests.clear();
if (op.xghost_recv.Size())
{
auto vdims = op.fespace->GetVDim();
auto rcv = Device::GetGPUAwareMPI() ? op.xghost_recv.Write()
: op.xghost_recv.HostWrite();
for (int i = 0; i < op.recv_ranks.Size(); ++i)
{
op.requests.emplace_back();
MPI_Irecv(rcv + op.recv_segments[i] * vdims,
(op.recv_segments[i + 1] - op.recv_segments[i]) * vdims,
MPITypeMap<real_t>::mpi_type, op.recv_ranks[i],
MessageTag::DEREFINEMENT_MATRIX_CONSTRUCTION_DATA,
op.fespace->GetComm(), &op.requests.back());
}
}
if (op.xghost_send.Size())
{
auto vdims = op.fespace->GetVDim();
// only is a GPU mem ptr if GPU-aware MPI is enabled
auto dst = Device::GetGPUAwareMPI() ? op.xghost_send.Write()
: op.xghost_send.HostWrite();
for (int i = 0; i < op.send_ranks.Size(); ++i)
{
op.requests.emplace_back();
MPI_Isend(dst + op.send_segments[i] * vdims,
(op.send_segments[i + 1] - op.send_segments[i]) * vdims,
MPITypeMap<real_t>::mpi_type, op.send_ranks[i],
MessageTag::DEREFINEMENT_MATRIX_CONSTRUCTION_DATA,
op.fespace->GetComm(), &op.requests.back());
}
}
{
// diagonal
DerefineMatrixOpMultFunctor<Order, Atomic, true> func;
func.xptr = x.Read();
y.UseDevice();
y = 0.;
func.yptr = y.ReadWrite();
func.bsptr = op.block_storage.Read();
func.boptr = op.block_offsets.Read();
func.brptr = op.block_row_idcs_offsets.Read();
func.bcptr = op.block_col_idcs_offsets.Read();
func.rptr = op.row_idcs.Read();
func.cptr = op.col_idcs.Read();
func.vdims = op.fespace->GetVDim();
func.nblocks = op.block_offsets.Size();
func.width = op.Width() / func.vdims;
func.height = op.Height() / func.vdims;
func.Run(op.max_rows);
}
// wait for comm to finish, if any
if (op.requests.size())
{
MPI_Waitall(op.requests.size(), op.requests.data(), MPI_STATUSES_IGNORE);
if (op.xghost_recv.Size())
{
// off-diagonal kernel
DerefineMatrixOpMultFunctor<Order, Atomic, false> func;
// directly read from host-pinned memory if not using GPU-aware MPI
func.xptr = Device::GetGPUAwareMPI() ? op.xghost_recv.Read()
: op.xghost_recv.HostRead();
func.yptr = y.ReadWrite();
func.bsptr = op.block_storage.Read();
func.boptr = op.off_diag_block_offsets.Read();
func.brptr = op.block_off_diag_row_idcs_offsets.Read();
func.rsptr = op.recv_segment_idcs.Read();
func.segptr = op.recv_segments.Read();
func.coptr = op.block_off_diag_col_offsets.Read();
func.bwptr = op.block_off_diag_widths.Read();
func.rptr = op.row_off_diag_idcs.Read();
func.vdims = op.fespace->GetVDim();
func.nblocks = op.off_diag_block_offsets.Size();
func.width = op.xghost_recv.Size() / func.vdims;
func.height = op.Height() / func.vdims;
func.Run(op.max_rows);
}
}
}
} // namespace internal
template <Ordering::Type Order, bool Atomic>
ParDerefineMatrixOp::MultKernelType ParDerefineMatrixOp::MultKernel::Kernel()
{
return internal::ParDerefMultKernelImpl<Order, Atomic>;
}
ParDerefineMatrixOp::MultKernelType
ParDerefineMatrixOp::MultKernel::Fallback(Ordering::Type, bool)
{
MFEM_ABORT("invalid MultKernel parameters");
}
ParDerefineMatrixOp::Kernels::Kernels()
{
MultKernel::Specialization<Ordering::byNODES, false>::Add();
MultKernel::Specialization<Ordering::byVDIM, false>::Add();
MultKernel::Specialization<Ordering::byNODES, true>::Add();
MultKernel::Specialization<Ordering::byVDIM, true>::Add();
}
void ParDerefineMatrixOp::Mult(const Vector &x, Vector &y) const
{
const bool is_dg = fespace->FEColl()->GetContType()
== FiniteElementCollection::DISCONTINUOUS;
// DG needs atomic summation
MultKernel::Run(fespace->GetOrdering(), is_dg, *this, x, y);
// use this to prevent xghost* from being re-purposed for subsequent Mult
// calls
MFEM_DEVICE_SYNC;
}
ParDerefineMatrixOp::ParDerefineMatrixOp(ParFiniteElementSpace &fespace_,
int old_ndofs,
const Table *old_elem_dof,
const Table *old_elem_fos)
: Operator(fespace_.GetVSize(), old_ndofs * fespace_.GetVDim()),
fespace(&fespace_)
{
static Kernels kernels;
constexpr int max_team_size = 256;
const int NRanks = fespace->GetNRanks();
const int nrk = HYPRE_AssumedPartitionCheck() ? 2 : NRanks;
MFEM_VERIFY(fespace->Nonconforming(),
"Not implemented for conforming meshes.");
MFEM_VERIFY(fespace->old_dof_offsets[nrk],
"Missing previous (finer) space.");
const int MyRank = fespace->GetMyRank();
ParNCMesh *old_pncmesh = fespace->GetParMesh()->pncmesh;
const CoarseFineTransformations &dtrans =
old_pncmesh->GetDerefinementTransforms();
const Array<int> &old_ranks = old_pncmesh->GetDerefineOldRanks();
const bool is_dg = fespace->FEColl()->GetContType()
== FiniteElementCollection::DISCONTINUOUS;
DenseMatrix localRVO; // for variable-order only
DenseTensor localR[Geometry::NumGeom];
int diag_rows = 0;
int off_diag_rows = 0;
int diag_cols = 0;
auto get_ldofs = [&](int k) -> int
{
const Embedding &emb = dtrans.embeddings[k];
if (fespace->IsVariableOrder())
{
const FiniteElement *fe = fespace->GetFE(emb.parent);
return fe->GetDof();
}
else
{
Geometry::Type geom =
fespace->GetParMesh()->GetElementBaseGeometry(emb.parent);
return fespace->FEColl()->FiniteElementForGeometry(geom)->GetDof();
}
};
Array<int> dofs, old_dofs;
max_rows = 1;
// first pass:
// - determine memory block lengths
// - identify dofs in x we need to send/receive
// don't need to send the indices, fine rank will re-arrange and sign
// change x before transmitting the ghost data
// key: coarse rank to send to
// value: old dofs to send (with sign)
std::map<int, std::vector<int>> to_send;
// key: fine rank
// value: indices into dtrans.embeddings
std::map<int, std::vector<int>> od_ks;
// key: fine rank
// value: recv segment length
std::map<int, int> od_seg_lens;
int send_len = 0;
int recv_len = 0;
// size of block_storage, if fespace->IsVariableOrder()
// otherwise unused
int total_size = 0;
int num_diagonal_blocks = 0;
int num_offdiagonal_blocks = 0;
for (int k = 0; k < dtrans.embeddings.Size(); ++k)
{
const Embedding &emb = dtrans.embeddings[k];
int fine_rank = old_ranks[k];
int coarse_rank = (emb.parent < 0) ? (-1 - emb.parent)
: old_pncmesh->ElementRank(emb.parent);
if (coarse_rank != MyRank && fine_rank == MyRank)
{
// this rank needs to send data in x to course_rank
old_elem_dof->GetRow(k, old_dofs);
auto &tmp = to_send[coarse_rank];
send_len += old_dofs.Size();
for (int i = 0; i < old_dofs.Size(); ++i)
{
tmp.emplace_back(old_dofs[i]);
}
}
else if (coarse_rank == MyRank && fine_rank != MyRank)
{
// this rank needs to receive data in x from fine_rank
MFEM_ASSERT(emb.parent >= 0, "");
auto ldofs = get_ldofs(k);
off_diag_rows += ldofs;
recv_len += ldofs;
od_ks[fine_rank].emplace_back(k);
od_seg_lens[fine_rank] += ldofs;
++num_offdiagonal_blocks;
if (fespace->IsVariableOrder())
{
total_size += ldofs * ldofs;
}
}
else if (coarse_rank == MyRank && fine_rank == MyRank)
{
MFEM_ASSERT(emb.parent >= 0, "");
// diagonal
++num_diagonal_blocks;
auto ldofs = get_ldofs(k);
diag_rows += ldofs;
diag_cols += ldofs;
if (fespace->IsVariableOrder())
{
total_size += ldofs * ldofs;
}
}
}
send_segments.SetSize(to_send.size() + 1);
send_segments.HostWrite();
send_ranks.SetSize(to_send.size());
send_ranks.HostWrite();
{
int idx = 0;
send_segments[0] = 0;
for (auto &tmp : to_send)
{
send_ranks[idx] = tmp.first;
send_segments[idx + 1] = send_segments[idx] + tmp.second.size();
++idx;
}
}
recv_segment_idcs.SetSize(off_diag_rows);
recv_segment_idcs.HostWrite();
recv_segments.SetSize(od_ks.size() + 1);
recv_segments.HostWrite();
recv_ranks.SetSize(od_ks.size());
recv_ranks.HostWrite();
// set sizes
row_idcs.SetSize(diag_rows);
row_idcs.HostWrite();
row_off_diag_idcs.SetSize(off_diag_rows);
row_off_diag_idcs.HostWrite();
col_idcs.SetSize(diag_cols);
col_idcs.HostWrite();
block_row_idcs_offsets.SetSize(num_diagonal_blocks + 1);
block_row_idcs_offsets.HostWrite();
block_col_idcs_offsets.SetSize(num_diagonal_blocks + 1);
block_col_idcs_offsets.HostWrite();
block_off_diag_row_idcs_offsets.SetSize(num_offdiagonal_blocks + 1);
block_off_diag_row_idcs_offsets.HostWrite();
block_off_diag_col_offsets.SetSize(num_offdiagonal_blocks);
block_off_diag_col_offsets.HostWrite();
block_off_diag_widths.SetSize(num_offdiagonal_blocks);
block_off_diag_widths.HostWrite();
pack_col_idcs.SetSize(send_len);
// memory manager doesn't appear to have a graceful fallback for
// HOST_PINNED if not built with CUDA or HIP
#if defined(MFEM_USE_CUDA) or defined(MFEM_USE_HIP)
xghost_send.SetSize(send_len * fespace->GetVDim(),
Device::GetGPUAwareMPI() ? MemoryType::DEFAULT
: MemoryType::HOST_PINNED);
xghost_recv.SetSize(recv_len * fespace->GetVDim(),
Device::GetGPUAwareMPI() ? MemoryType::DEFAULT
: MemoryType::HOST_PINNED);
#else
xghost_send.SetSize(send_len * fespace->GetVDim());
xghost_recv.SetSize(recv_len * fespace->GetVDim());
#endif
send_permutations.SetSize(send_len);
send_segment_idcs.SetSize(send_len);
block_offsets.SetSize(num_diagonal_blocks);
block_offsets.HostWrite();
off_diag_block_offsets.SetSize(num_offdiagonal_blocks);
off_diag_block_offsets.HostWrite();
int geom_offsets[Geometry::NumGeom];
real_t *bs_ptr;
if (fespace->IsVariableOrder())
{
block_storage.SetSize(total_size);
bs_ptr = block_storage.HostWrite();
// compute block data later
}
else
{
// compression scheme:
// block_offsets is the start of each block, potentially repeated
// only need to store localR for used shapes
Mesh::GeometryList elem_geoms(*fespace->GetMesh());
int size = 0;
for (int i = 0; i < elem_geoms.Size(); ++i)
{
fespace->GetLocalDerefinementMatrices(elem_geoms[i],
localR[elem_geoms[i]]);
geom_offsets[elem_geoms[i]] = size;
size += localR[elem_geoms[i]].TotalSize();
}
block_storage.SetSize(size);
bs_ptr = block_storage.HostWrite();
// copy blocks into block_storage
for (int i = 0; i < elem_geoms.Size(); ++i)
{
std::copy(localR[elem_geoms[i]].Data(),
localR[elem_geoms[i]].Data()
+ localR[elem_geoms[i]].TotalSize(),
bs_ptr);
bs_ptr += localR[elem_geoms[i]].TotalSize();
}
}
// second pass:
// - initialize buffers
{
auto ptr = send_permutations.HostWrite();
auto ptr2 = send_segment_idcs.HostWrite();
int i = 0;
for (auto &v : to_send)
{
ptr = std::copy(v.second.begin(), v.second.end(), ptr);
for (size_t idx = 0; idx < v.second.size(); ++idx)
{
*ptr2 = i;
++ptr2;
}
++i;
}
}
block_row_idcs_offsets[0] = 0;
block_col_idcs_offsets[0] = 0;
block_off_diag_row_idcs_offsets[0] = 0;
Array<int> mark(fespace->GetNDofs());
mark = 0;
{
int idx = 0;
recv_segments[0] = 0;
for (auto &v : od_seg_lens)
{
recv_ranks[idx] = v.first;
recv_segments[idx + 1] = recv_segments[idx] + v.second;
++idx;
}
}
// key: index into dtrans.embeddings
// value: off-diagonal block offset, od_ridx, seg id
std::unordered_map<int, std::array<int, 3>> ks_map;
{
int od_ridx = 0;
int seg_id = 0;
for (auto &v1 : od_ks)
{
for (auto k : v1.second)
{
auto &tmp = ks_map[k];
tmp[0] = ks_map.size() - 1;
tmp[1] = od_ridx;
tmp[2] = seg_id;
od_ridx += get_ldofs(k);
}
++seg_id;
}
}
int diag_idx = 0;
int var_offset = 0;
int ridx = 0;
int cidx = 0;
// can't break this up into separate diagonals/off-diagonals loops because
// of mark
for (int k = 0; k < dtrans.embeddings.Size(); ++k)
{
const Embedding &emb = dtrans.embeddings[k];
if (emb.parent < 0)
{
continue;
}
int fine_rank = old_ranks[k];
int coarse_rank = (emb.parent < 0) ? (-1 - emb.parent)
: old_pncmesh->ElementRank(emb.parent);
if (coarse_rank == MyRank)
{
// either diagonal or off-diagonal
Geometry::Type geom =
fespace->GetMesh()->GetElementBaseGeometry(emb.parent);
if (fespace->IsVariableOrder())
{
const FiniteElement *fe = fespace->GetFE(emb.parent);
const DenseTensor &pmats = dtrans.point_matrices[geom];
const int ldof = fe->GetDof();
IsoparametricTransformation isotr;
isotr.SetIdentityTransformation(geom);
localRVO.SetSize(ldof, ldof);
isotr.SetPointMat(pmats(emb.matrix));
// Local restriction is size ldofxldof assuming that the parent
// and child are of same polynomial order.
fe->GetLocalRestriction(isotr, localRVO);
// copy block
auto s = localRVO.Height() * localRVO.Width();
std::copy(localRVO.Data(), localRVO.Data() + s, bs_ptr);
bs_ptr += s;
}
DenseMatrix &lR =
fespace->IsVariableOrder() ? localRVO : localR[geom](emb.matrix);
max_rows = std::max(lR.Height(), max_rows);
auto size = lR.Height() * lR.Width();
fespace->elem_dof->GetRow(emb.parent, dofs);
if (fine_rank == MyRank)
{
// diagonal
old_elem_dof->GetRow(k, old_dofs);
MFEM_VERIFY(old_dofs.Size() == dofs.Size(),
"Parent and child must have same #dofs.");
block_row_idcs_offsets[diag_idx + 1] =
block_row_idcs_offsets[diag_idx] + lR.Height();
block_col_idcs_offsets[diag_idx + 1] =
block_col_idcs_offsets[diag_idx] + lR.Width();
if (fespace->IsVariableOrder())
{
block_offsets[diag_idx] = var_offset;
var_offset += size;
}
else
{
block_offsets[diag_idx] = geom_offsets[geom] + size * emb.matrix;
}
for (int i = 0; i < lR.Height(); ++i, ++ridx)
{
if (!std::isfinite(lR(i, 0)))
{
row_idcs[ridx] = INT_MAX;
continue;
}
int r = dofs[i];
int m = (r >= 0) ? r : (-1 - r);
if (is_dg || !mark[m])
{
row_idcs[ridx] = r;
mark[m] = 1;
}
else
{
row_idcs[ridx] = INT_MAX;
}
}
for (int i = 0; i < lR.Width(); ++i, ++cidx)
{
col_idcs[cidx] = old_dofs[i];
}
++diag_idx;
}
else
{
// off-diagonal
auto &tmp = ks_map.at(k);
auto od_idx = tmp[0];
auto od_ridx = tmp[1];
block_off_diag_row_idcs_offsets[od_idx + 1] =
block_off_diag_row_idcs_offsets[od_idx] + lR.Height();
block_off_diag_col_offsets[od_idx] = od_ridx;
block_off_diag_widths[od_idx] = lR.Width();
recv_segment_idcs[od_idx] = tmp[2];
if (fespace->IsVariableOrder())
{
off_diag_block_offsets[od_idx] = var_offset;
var_offset += size;
}
else
{
off_diag_block_offsets[od_idx] =
geom_offsets[geom] + size * emb.matrix;
}
for (int i = 0; i < lR.Height(); ++i, ++od_ridx)
{
if (!std::isfinite(lR(i, 0)))
{
row_off_diag_idcs[od_ridx] = INT_MAX;
continue;
}
int r = dofs[i];
int m = (r >= 0) ? r : (-1 - r);
if (is_dg || !mark[m])
{
row_off_diag_idcs[od_ridx] = r;
mark[m] = 1;
}
else
{
row_off_diag_idcs[od_ridx] = INT_MAX;
}
}
++od_idx;
}
}
}
// if not using GPU, set max_rows/max_cols to zero
if (Device::Allows(Backend::DEVICE_MASK))
{
max_rows = std::min(max_rows, max_team_size);
}
else
{
max_rows = 1;
}
requests.reserve(recv_ranks.Size() + send_ranks.Size());
}
} // namespace mfem
/// \endcond DO_NOT_DOCUMENT
#endif
-111
View File
@@ -1,111 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_PDEREFMAT_OP
#define MFEM_PDEREFMAT_OP
#include "../config/config.hpp"
#ifdef MFEM_USE_MPI
#include "pfespace.hpp"
#include "kernel_dispatch.hpp"
#include <vector>
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
struct ParDerefineMatrixOp : public Operator
{
ParFiniteElementSpace *fespace;
/// offsets into block_storage for diagonal
Array<int> block_offsets;
/// offsets into row_idcs for diagonal
Array<int> block_row_idcs_offsets;
/// offsets into col_idcs for diagonal
Array<int> block_col_idcs_offsets;
/// offsets into block_storage for off-diagonal
Array<int> off_diag_block_offsets;
/// offsets into row_idcs for off-diagonal
Array<int> block_off_diag_row_idcs_offsets;
Array<int> block_off_diag_col_offsets;
Array<int> block_off_diag_widths;
/// mapping for row dofs, INT_MAX indicates the block row should be ignored.
/// negative means the row data should be negated.
/// only for diagonal blocks
Array<int> row_idcs;
/// mapping for col dofs, negative means the col data should be negated.
/// only for diagonal blocks
Array<int> col_idcs;
Array<int> pack_col_idcs;
/// mapping for row dofs, INT_MAX indicates the block row should be ignored.
/// negative means the row data should be negated.
/// only for off-diagonal blocks
Array<int> row_off_diag_idcs;
/// dense block matrices which can be reused to construct the full matrix
/// operation. These are stored contiguously and blocks have no restrictions
/// on shape (can be rectangle and differ from block to block).
/// This is only for the diagonal block.
Vector block_storage;
/// maximum height of any block in block_storage for GPU
/// parallelization, or 1 for CPU runs.
int max_rows;
/// quasi Ordering::byNODES, broken into sections by ranks we need to send
/// the data to
mutable Vector xghost_send;
/// quasi Ordering::byNODES, broken into sections by ranks we received
/// the data from
mutable Vector xghost_recv;
/// maps off-diagonal k to segment
Array<int> recv_segment_idcs;
/// cumulative count of dofs which will be received from other ranks
Array<int> recv_segments;
/// Source rank of each recv segment
Array<int> recv_ranks;
/// What send segment each entry in send_permutations corresponds to
Array<int> send_segment_idcs;
/// cumulative count of dofs which will be sent to other ranks
Array<int> send_segments;
/// Destination rank of each send segment
Array<int> send_ranks;
/// how to permute/sign change values from our local x to send to other ranks
Array<int> send_permutations;
/// internal buffer for MPI requests
mutable std::vector<MPI_Request> requests;
using MultKernelType = void (*)(const ParDerefineMatrixOp &, const Vector &,
Vector &);
/// template args: ordering, atomic
MFEM_REGISTER_KERNELS(MultKernel, MultKernelType, (Ordering::Type, bool));
struct Kernels
{
Kernels();
};
void Mult(const Vector &x, Vector &y) const;
ParDerefineMatrixOp(ParFiniteElementSpace &fespace_, int old_ndofs,
const Table *old_elem_dof, const Table *old_elem_fos);
};
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
#endif
#endif
+33 -51
View File
@@ -22,13 +22,12 @@
#include "../mesh/mesh_headers.hpp"
#include "../general/binaryio.hpp"
#include "pderefmat_op.hpp"
#include <limits>
#include <list>
namespace mfem
{
ParFiniteElementSpace::ParFiniteElementSpace(
const ParFiniteElementSpace &orig, ParMesh *pmesh,
const FiniteElementCollection *fec)
@@ -4488,6 +4487,13 @@ ParFiniteElementSpace::RebalanceMatrix(int old_ndofs,
return M;
}
struct DerefDofMessage
{
std::vector<HYPRE_BigInt> dofs;
MPI_Request request;
};
HypreParMatrix*
ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
const Table* old_elem_dof,
@@ -4530,13 +4536,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
old_pncmesh->GetDerefinementTransforms();
const Array<int> &old_ranks = old_pncmesh->GetDerefineOldRanks();
// key: other rank
// value: send or recieve buffer
std::map<int, std::vector<HYPRE_BigInt>> to_send;
std::map<int, std::vector<HYPRE_BigInt>> to_recv;
// key: index into dtrans.embeddings
// value: [start, stop]
std::unordered_map<int, std::array<size_t, 2>> recv_messages;
std::map<int, DerefDofMessage> messages;
HYPRE_BigInt old_offset = HYPRE_AssumedPartitionCheck()
? old_dof_offsets[0] : old_dof_offsets[MyRank];
@@ -4556,46 +4556,30 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
old_elem_dof->GetRow(k, dofs);
DofsToVDofs(dofs, old_ndofs);
std::vector<HYPRE_BigInt>& send_buf = to_send[coarse_rank];
auto pos = send_buf.size();
send_buf.resize(pos + dofs.Size());
DerefDofMessage &msg = messages[k];
msg.dofs.resize(dofs.Size());
for (int i = 0; i < dofs.Size(); i++)
{
send_buf[pos + i] = old_offset + dofs[i];
msg.dofs[i] = old_offset + dofs[i];
}
MPI_Isend(&msg.dofs[0], static_cast<int>(msg.dofs.size()), HYPRE_MPI_BIG_INT,
coarse_rank, 291, MyComm, &msg.request);
}
else if (coarse_rank == MyRank && fine_rank != MyRank)
{
MFEM_ASSERT(emb.parent >= 0, "");
Geometry::Type geom = mesh->GetElementBaseGeometry(emb.parent);
std::vector<HYPRE_BigInt>& recv_buf = to_recv[fine_rank];
auto& msg = recv_messages[k];
msg[0] = recv_buf.size();
recv_buf.resize(recv_buf.size() + ldof[geom] * vdim);
msg[1] = recv_buf.size();
}
}
DerefDofMessage &msg = messages[k];
msg.dofs.resize(ldof[geom]*vdim);
// assume embedding orders are consistent (i.e. what we expect to receive
// first from a given rank is sent first, etc.)
std::vector<MPI_Request> requests;
requests.reserve(to_send.size() + to_recv.size());
// enqueue recvs
for (auto &v : to_recv)
{
requests.emplace_back();
MPI_Irecv(v.second.data(), v.second.size(), HYPRE_MPI_BIG_INT, v.first,
MessageTag::DEREFINEMENT_MATRIX_CONSTRUCTION_DATA, MyComm,
&requests.back());
}
// enqueue sends
for (auto &v : to_send)
{
requests.emplace_back();
MPI_Isend(v.second.data(), v.second.size(), HYPRE_MPI_BIG_INT, v.first,
MessageTag::DEREFINEMENT_MATRIX_CONSTRUCTION_DATA, MyComm,
&requests.back());
MPI_Irecv(&msg.dofs[0], ldof[geom]*vdim, HYPRE_MPI_BIG_INT,
fine_rank, 291, MyComm, &msg.request);
}
// TODO: coalesce Isends/Irecvs to the same rank. Typically, on uniform
// derefinement, there should be just one send to MyRank-1 and one recv
// from MyRank+1
}
DenseTensor localR[Geometry::NumGeom];
@@ -4653,7 +4637,10 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
diag->Finalize();
// wait for all sends/receives to complete
MPI_Waitall(requests.size(), requests.data(), MPI_STATUSES_IGNORE);
for (auto it = messages.begin(); it != messages.end(); ++it)
{
MPI_Wait(&it->second.request, MPI_STATUS_IGNORE);
}
// create the off-diagonal part of the derefinement matrix
SparseMatrix *offd = new SparseMatrix(ndofs*vdim, 1);
@@ -4674,14 +4661,13 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
elem_dof->GetRow(emb.parent, dofs);
auto& odofs = to_recv.at(fine_rank);
auto &msg = recv_messages[k];
MFEM_ASSERT(msg[1] > msg[0], "");
DerefDofMessage &msg = messages[k];
MFEM_ASSERT(msg.dofs.size(), "");
for (int vd = 0; vd < vdim; vd++)
{
MFEM_ASSERT(ldof[geom], "");
HYPRE_BigInt *remote_dofs = odofs.data() + msg[0] + vd * ldof[geom];
HYPRE_BigInt* remote_dofs = &msg.dofs[vd*ldof[geom]];
for (int i = 0; i < lR.Height(); i++)
{
@@ -4708,6 +4694,7 @@ ParFiniteElementSpace::ParallelDerefinementMatrix(int old_ndofs,
}
}
messages.clear();
offd->Finalize(0);
offd->SetWidth(static_cast<int>(col_map.size()));
@@ -4959,13 +4946,8 @@ void ParFiniteElementSpace::Update(bool want_transform)
case Mesh::DEREFINE:
{
#if 0
Th.Reset(ParallelDerefinementMatrix(old_ndofs, old_elem_dof,
old_elem_fos));
#else
Th.Reset(new ParDerefineMatrixOp(*this, old_ndofs, old_elem_dof,
old_elem_fos));
#endif
if (Nonconforming())
{
Th.SetOperatorOwner(false);
@@ -5277,7 +5259,7 @@ DeviceConformingProlongationOperator::DeviceConformingProlongationOperator(
gc.GetNeighborLTDofTable(nbr_ltdof);
const int nb_connections = nbr_ltdof.Size_of_connections();
shr_ltdof.SetSize(nb_connections);
if (nb_connections > 0) { shr_ltdof.CopyFrom(nbr_ltdof.GetJ()); }
shr_ltdof.CopyFrom(nbr_ltdof.GetJ());
shr_buf.SetSize(nb_connections);
shr_buf.UseDevice(true);
shr_buf_offsets = nbr_ltdof.GetIMemory();
@@ -5306,7 +5288,7 @@ DeviceConformingProlongationOperator::DeviceConformingProlongationOperator(
gc.GetNeighborLDofTable(nbr_ldof);
const int nb_connections = nbr_ldof.Size_of_connections();
ext_ldof.SetSize(nb_connections);
if (nb_connections > 0) { ext_ldof.CopyFrom(nbr_ldof.GetJ()); }
ext_ldof.CopyFrom(nbr_ldof.GetJ());
ext_ldof.GetMemory().UseDevice(true);
ext_buf.SetSize(nb_connections);
ext_buf.UseDevice(true);
-3
View File
@@ -24,12 +24,9 @@
namespace mfem
{
struct ParDerefineMatrixOp;
/// Abstract parallel finite element space.
class ParFiniteElementSpace : public FiniteElementSpace
{
friend struct ParDerefineMatrixOp;
private:
/// MPI data.
MPI_Comm MyComm;
+280 -3
View File
@@ -9,16 +9,278 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "det.hpp"
#include "../quadinterpolator.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../fem/kernels.hpp"
#include "../../linalg/kernels.hpp"
using namespace mfem;
namespace mfem
{
namespace internal
{
namespace quadrature_interpolator
{
static void Det1D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d,
const int q1d,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(b);
MFEM_CONTRACT_VAR(d_buff);
const auto G = Reshape(g, q1d, d1d);
const auto X = Reshape(x, d1d, NE);
auto Y = Reshape(y, q1d, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < q1d; q++)
{
real_t u = 0.0;
for (int d = 0; d < d1d; d++)
{
u += G(q, d) * X(d, e);
}
Y(q, e) = u;
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void Det2D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(d_buff);
static constexpr int SDIM = 2;
static constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, SDIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t XY[SDIM][NBZ][MD1*MD1];
MFEM_SHARED real_t DQ[2*SDIM][NBZ][MD1*MQ1];
MFEM_SHARED real_t QQ[2*SDIM][NBZ][MQ1*MQ1];
kernels::internal::LoadX<MD1,NBZ>(e,D1D,X,XY);
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::internal::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
kernels::internal::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J[4];
kernels::internal::PullGrad<MQ1,NBZ>(Q1D,qx,qy,QQ,J);
Y(qx,qy,e) = kernels::Det<2>(J);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
static void Det2DSurface(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(d_buff);
static constexpr int SDIM = 3;
static constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, SDIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t XYZ[SDIM][NBZ][MD1*MD1];
MFEM_SHARED real_t DQ[2*SDIM][NBZ][MD1*MQ1];
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
// Load XYZ components
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int d = 0; d < SDIM; ++d)
{
XYZ[d][tidz][dx + dy*D1D] = X(dx,dy,d,e);
}
}
}
MFEM_SYNC_THREAD;
ConstDeviceMatrix B_mat(BG[0], D1D, Q1D);
ConstDeviceMatrix G_mat(BG[1], D1D, Q1D);
// x contraction
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
for (int d = 0; d < SDIM; ++d)
{
real_t u = 0.0;
real_t v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const real_t xval = XYZ[d][tidz][dx + dy*D1D];
u += xval * G_mat(dx,qx);
v += xval * B_mat(dx,qx);
}
DQ[d][tidz][dy + qx*D1D] = u;
DQ[3 + d][tidz][dy + qx*D1D] = v;
}
}
}
MFEM_SYNC_THREAD;
// y contraction and determinant computation
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J_[6] = {0.0, 0.0, 0.0, 0.0, 0.0, 0.0};
for (int d = 0; d < SDIM; ++d)
{
for (int dy = 0; dy < D1D; ++dy)
{
J_[d] += DQ[d][tidz][dy + qx*D1D] * B_mat(dy,qy);
J_[3 + d] += DQ[3 + d][tidz][dy + qx*D1D] * G_mat(dy,qy);
}
}
DeviceTensor<2> J(J_, 3, 2);
const real_t E = J(0,0)*J(0,0) + J(1,0)*J(1,0) + J(2,0)*J(2,0);
const real_t F = J(0,0)*J(0,1) + J(1,0)*J(1,1) + J(2,0)*J(2,1);
const real_t G = J(0,1)*J(0,1) + J(1,1)*J(1,1) + J(2,1)*J(2,1);
Y(qx,qy,e) = std::sqrt(E*G - F*F);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, bool SMEM = true>
static void Det3D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr) // used only with SMEM = false
{
constexpr int DIM = 3;
static constexpr int GRID = SMEM ? 0 : 128;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, D1D, DIM, NE);
auto Y = Reshape(y, Q1D, Q1D, Q1D, NE);
real_t *GM = nullptr;
if (!SMEM)
{
const DeviceDofQuadLimits &limits = DeviceDofQuadLimits::Get();
const int max_q1d = T_Q1D ? T_Q1D : limits.MAX_Q1D;
const int max_d1d = T_D1D ? T_D1D : limits.MAX_D1D;
const int max_qd = std::max(max_q1d, max_d1d);
const int mem_size = max_qd * max_qd * max_qd * 9;
d_buff->SetSize(2*mem_size*GRID);
GM = d_buff->Write();
}
mfem::forall_3D_grid(NE, Q1D, Q1D, Q1D, GRID, [=] MFEM_HOST_DEVICE (int e)
{
static constexpr int MQ1 = T_Q1D ? T_Q1D :
(SMEM ? DofQuadLimits::MAX_DET_1D : DofQuadLimits::MAX_Q1D);
static constexpr int MD1 = T_D1D ? T_D1D :
(SMEM ? DofQuadLimits::MAX_DET_1D : DofQuadLimits::MAX_D1D);
static constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
static constexpr int MSZ = MDQ * MDQ * MDQ * 9;
const int bid = MFEM_BLOCK_ID(x);
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t SM0[SMEM?MSZ:1];
MFEM_SHARED real_t SM1[SMEM?MSZ:1];
real_t *lm0 = SMEM ? SM0 : GM + MSZ*bid;
real_t *lm1 = SMEM ? SM1 : GM + MSZ*(GRID+bid);
real_t (*DDD)[MD1*MD1*MD1] = (real_t (*)[MD1*MD1*MD1]) (lm0);
real_t (*DDQ)[MD1*MD1*MQ1] = (real_t (*)[MD1*MD1*MQ1]) (lm1);
real_t (*DQQ)[MD1*MQ1*MQ1] = (real_t (*)[MD1*MQ1*MQ1]) (lm0);
real_t (*QQQ)[MQ1*MQ1*MQ1] = (real_t (*)[MQ1*MQ1*MQ1]) (lm1);
kernels::internal::LoadX<MD1>(e,D1D,X,DDD);
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::internal::GradX<MD1,MQ1>(D1D,Q1D,BG,DDD,DDQ);
kernels::internal::GradY<MD1,MQ1>(D1D,Q1D,BG,DDQ,DQQ);
kernels::internal::GradZ<MD1,MQ1>(D1D,Q1D,BG,DQQ,QQQ);
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J[9];
kernels::internal::PullGrad<MQ1>(Q1D, qx,qy,qz, QQQ, J);
Y(qx,qy,qz,e) = kernels::Det<3>(J);
}
}
}
});
}
void InitDetKernels()
{
using k = QuadratureInterpolator::DetKernels;
@@ -40,12 +302,27 @@ void InitDetKernels()
}
} // namespace quadrature_interpolator
} // namespace internal
/// @cond Suppress_Doxygen_warnings
QuadratureInterpolator::DetKernelType
QuadratureInterpolator::DetKernels::Fallback(
namespace
{
using DetKernel = QuadratureInterpolator::DetKernelType;
}
template<int DIM, int SDIM, int D1D, int Q1D>
DetKernel QuadratureInterpolator::DetKernels::Kernel()
{
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D<D1D, Q1D>; }
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface<D1D, Q1D>; }
else if (DIM == 3) { return internal::quadrature_interpolator::Det3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
}
DetKernel QuadratureInterpolator::DetKernels::Fallback(
int DIM, int SDIM, int D1D, int Q1D)
{
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
-304
View File
@@ -1,304 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#ifndef MFEM_QUADINTERP_DET_HPP
#define MFEM_QUADINTERP_DET_HPP
#include "../quadinterpolator.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../fem/kernels.hpp"
#include "../../linalg/kernels.hpp"
namespace mfem
{
namespace internal
{
namespace quadrature_interpolator
{
inline void Det1D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d,
const int q1d,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(b);
MFEM_CONTRACT_VAR(d_buff);
const auto G = Reshape(g, q1d, d1d);
const auto X = Reshape(x, d1d, NE);
auto Y = Reshape(y, q1d, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < q1d; q++)
{
real_t u = 0.0;
for (int d = 0; d < d1d; d++)
{
u += G(q, d) * X(d, e);
}
Y(q, e) = u;
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
inline void Det2D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(d_buff);
static constexpr int SDIM = 2;
static constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, SDIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t XY[SDIM][NBZ][MD1*MD1];
MFEM_SHARED real_t DQ[2*SDIM][NBZ][MD1*MQ1];
MFEM_SHARED real_t QQ[2*SDIM][NBZ][MQ1*MQ1];
kernels::internal::LoadX<MD1,NBZ>(e,D1D,X,XY);
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::internal::GradX<MD1,MQ1,NBZ>(D1D,Q1D,BG,XY,DQ);
kernels::internal::GradY<MD1,MQ1,NBZ>(D1D,Q1D,BG,DQ,QQ);
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J[4];
kernels::internal::PullGrad<MQ1,NBZ>(Q1D,qx,qy,QQ,J);
Y(qx,qy,e) = kernels::Det<2>(J);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0>
inline void Det2DSurface(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr)
{
MFEM_CONTRACT_VAR(d_buff);
static constexpr int SDIM = 3;
static constexpr int NBZ = 1;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, SDIM, NE);
auto Y = Reshape(y, Q1D, Q1D, NE);
mfem::forall_2D_batch(NE, Q1D, Q1D, NBZ, [=] MFEM_HOST_DEVICE (int e)
{
constexpr int MQ1 = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
constexpr int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int tidz = MFEM_THREAD_ID(z);
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t XYZ[SDIM][NBZ][MD1*MD1];
MFEM_SHARED real_t DQ[2*SDIM][NBZ][MD1*MQ1];
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
// Load XYZ components
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(dx,x,D1D)
{
for (int d = 0; d < SDIM; ++d)
{
XYZ[d][tidz][dx + dy*D1D] = X(dx,dy,d,e);
}
}
}
MFEM_SYNC_THREAD;
ConstDeviceMatrix B_mat(BG[0], D1D, Q1D);
ConstDeviceMatrix G_mat(BG[1], D1D, Q1D);
// x contraction
MFEM_FOREACH_THREAD(dy,y,D1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
for (int d = 0; d < SDIM; ++d)
{
real_t u = 0.0;
real_t v = 0.0;
for (int dx = 0; dx < D1D; ++dx)
{
const real_t xval = XYZ[d][tidz][dx + dy*D1D];
u += xval * G_mat(dx,qx);
v += xval * B_mat(dx,qx);
}
DQ[d][tidz][dy + qx*D1D] = u;
DQ[3 + d][tidz][dy + qx*D1D] = v;
}
}
}
MFEM_SYNC_THREAD;
// y contraction and determinant computation
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J_[6] = {0.0, 0.0, 0.0, 0.0, 0.0, 0.0};
for (int d = 0; d < SDIM; ++d)
{
for (int dy = 0; dy < D1D; ++dy)
{
J_[d] += DQ[d][tidz][dy + qx*D1D] * B_mat(dy,qy);
J_[3 + d] += DQ[3 + d][tidz][dy + qx*D1D] * G_mat(dy,qy);
}
}
DeviceTensor<2> J(J_, 3, 2);
const real_t E = J(0,0)*J(0,0) + J(1,0)*J(1,0) + J(2,0)*J(2,0);
const real_t F = J(0,0)*J(0,1) + J(1,0)*J(1,1) + J(2,0)*J(2,1);
const real_t G = J(0,1)*J(0,1) + J(1,1)*J(1,1) + J(2,1)*J(2,1);
Y(qx,qy,e) = std::sqrt(E*G - F*F);
}
}
});
}
template<int T_D1D = 0, int T_Q1D = 0, bool SMEM = true>
inline void Det3D(const int NE,
const real_t *b,
const real_t *g,
const real_t *x,
real_t *y,
const int d1d = 0,
const int q1d = 0,
Vector *d_buff = nullptr) // used only with SMEM = false
{
constexpr int DIM = 3;
static constexpr int GRID = SMEM ? 0 : 128;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const auto B = Reshape(b, Q1D, D1D);
const auto G = Reshape(g, Q1D, D1D);
const auto X = Reshape(x, D1D, D1D, D1D, DIM, NE);
auto Y = Reshape(y, Q1D, Q1D, Q1D, NE);
real_t *GM = nullptr;
if (!SMEM)
{
const DeviceDofQuadLimits &limits = DeviceDofQuadLimits::Get();
const int max_q1d = T_Q1D ? T_Q1D : limits.MAX_Q1D;
const int max_d1d = T_D1D ? T_D1D : limits.MAX_D1D;
const int max_qd = std::max(max_q1d, max_d1d);
const int mem_size = max_qd * max_qd * max_qd * 9;
d_buff->SetSize(2*mem_size*GRID);
GM = d_buff->Write();
}
mfem::forall_3D_grid(NE, Q1D, Q1D, Q1D, GRID, [=] MFEM_HOST_DEVICE (int e)
{
static constexpr int MQ1 = T_Q1D ? T_Q1D :
(SMEM ? DofQuadLimits::MAX_DET_1D : DofQuadLimits::MAX_Q1D);
static constexpr int MD1 = T_D1D ? T_D1D :
(SMEM ? DofQuadLimits::MAX_DET_1D : DofQuadLimits::MAX_D1D);
static constexpr int MDQ = MQ1 > MD1 ? MQ1 : MD1;
static constexpr int MSZ = MDQ * MDQ * MDQ * 9;
const int bid = MFEM_BLOCK_ID(x);
MFEM_SHARED real_t BG[2][MQ1*MD1];
MFEM_SHARED real_t SM0[SMEM?MSZ:1];
MFEM_SHARED real_t SM1[SMEM?MSZ:1];
real_t *lm0 = SMEM ? SM0 : GM + MSZ*bid;
real_t *lm1 = SMEM ? SM1 : GM + MSZ*(GRID+bid);
real_t (*DDD)[MD1*MD1*MD1] = (real_t (*)[MD1*MD1*MD1]) (lm0);
real_t (*DDQ)[MD1*MD1*MQ1] = (real_t (*)[MD1*MD1*MQ1]) (lm1);
real_t (*DQQ)[MD1*MQ1*MQ1] = (real_t (*)[MD1*MQ1*MQ1]) (lm0);
real_t (*QQQ)[MQ1*MQ1*MQ1] = (real_t (*)[MQ1*MQ1*MQ1]) (lm1);
kernels::internal::LoadX<MD1>(e,D1D,X,DDD);
kernels::internal::LoadBG<MD1,MQ1>(D1D,Q1D,B,G,BG);
kernels::internal::GradX<MD1,MQ1>(D1D,Q1D,BG,DDD,DDQ);
kernels::internal::GradY<MD1,MQ1>(D1D,Q1D,BG,DDQ,DQQ);
kernels::internal::GradZ<MD1,MQ1>(D1D,Q1D,BG,DQQ,QQQ);
MFEM_FOREACH_THREAD(qz,z,Q1D)
{
MFEM_FOREACH_THREAD(qy,y,Q1D)
{
MFEM_FOREACH_THREAD(qx,x,Q1D)
{
real_t J[9];
kernels::internal::PullGrad<MQ1>(Q1D, qx,qy,qz, QQQ, J);
Y(qx,qy,qz,e) = kernels::Det<3>(J);
}
}
}
});
}
} // namespace quadrature_interpolator
} // namespace internal
/// @cond Suppress_Doxygen_warnings
template<int DIM, int SDIM, int D1D, int Q1D>
QuadratureInterpolator::DetKernelType
QuadratureInterpolator::DetKernels::Kernel()
{
if (DIM == 1) { return internal::quadrature_interpolator::Det1D; }
else if (DIM == 2 && SDIM == 2) { return internal::quadrature_interpolator::Det2D<D1D, Q1D>; }
else if (DIM == 2 && SDIM == 3) { return internal::quadrature_interpolator::Det2DSurface<D1D, Q1D>; }
else if (DIM == 3) { return internal::quadrature_interpolator::Det3D<D1D, Q1D>; }
else { MFEM_ABORT(""); }
}
/// @endcond
} // namespace mfem
#endif // MFEM_QUADINTERP_DET_HPP
+27 -9
View File
@@ -5122,32 +5122,33 @@ real_t TMOP_Integrator::GetSurfaceFittingWeight()
void TMOP_Integrator::EnableNormalization(const GridFunction &x)
{
ComputeNormalizationEnergies(x, metric_normal, lim_normal);
ComputeNormalizationEnergies(x, metric_normal, lim_normal, surf_fit_normal);
metric_normal = 1.0 / metric_normal;
lim_normal = 1.0 / lim_normal;
//if (surf_fit_gf) { surf_fit_normal = 1.0 / surf_fit_normal; }
if (surf_fit_gf || surf_fit_pos) { surf_fit_normal = lim_normal; }
}
#ifdef MFEM_USE_MPI
void TMOP_Integrator::ParEnableNormalization(const ParGridFunction &x)
{
real_t loc[2];
ComputeNormalizationEnergies(x, loc[0], loc[1]);
real_t rdc[2];
MPI_Allreduce(loc, rdc, 2, MPITypeMap<real_t>::mpi_type, MPI_SUM,
real_t loc[3];
ComputeNormalizationEnergies(x, loc[0], loc[1], loc[2]);
real_t rdc[3];
MPI_Allreduce(loc, rdc, 3, MPITypeMap<real_t>::mpi_type, MPI_SUM,
x.ParFESpace()->GetComm());
metric_normal = 1.0 / rdc[0];
lim_normal = 1.0 / rdc[1];
// if (surf_fit_gf) { surf_fit_normal = 1.0 / rdc[2]; }
if (surf_fit_gf || surf_fit_pos) { surf_fit_normal = lim_normal; }
}
#endif
void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
real_t &metric_energy,
real_t &lim_energy)
real_t &lim_energy,
real_t &surf_fit_gf_energy)
{
metric_energy = 0.0;
lim_energy = 0.0;
if (PA.enabled)
{
MFEM_VERIFY(PA.E.Size() > 0, "Must be called after AssemblePA!");
@@ -5190,6 +5191,9 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
Jpr.SetSize(dim);
Jpt.SetSize(dim);
metric_energy = 0.0;
lim_energy = 0.0;
surf_fit_gf_energy = 0.0;
for (int i = 0; i < fes->GetNE(); i++)
{
const FiniteElement *fe = fes->GetFE(i);
@@ -5221,7 +5225,21 @@ void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
lim_energy += weight;
}
// TODO: Normalization of the surface fitting term.
// Normalization of the surface fitting term.
if (surf_fit_gf)
{
Array<int> dofs;
Vector sigma_e;
surf_fit_gf->FESpace()->GetElementDofs(i, dofs);
surf_fit_gf->GetSubVector(dofs, sigma_e);
for (int s = 0; s < dofs.Size(); s++)
{
if ((*surf_fit_marker)[dofs[s]] == true)
{
surf_fit_gf_energy += sigma_e(s) * sigma_e(s);
}
}
}
}
// Cases when integration is not over the target element, or when the
+2 -1
View File
@@ -2038,7 +2038,8 @@ protected:
} PA;
void ComputeNormalizationEnergies(const GridFunction &x,
real_t &metric_energy, real_t &lim_energy);
real_t &metric_energy, real_t &lim_energy,
real_t &surf_fit_gf_energy);
void AssembleElementVectorExact(const FiniteElement &el,
ElementTransformation &T,
+1 -69
View File
@@ -190,12 +190,6 @@ public:
/// Prepend an 'el' to the array, resize if necessary.
inline int Prepend(const T &el);
/// Insert @a els into the array at index @a i
inline int Insert(int i, const Array<T> &els);
/// Insert @a el into the array at index @a i
inline int Insert(int i, const T &el) { return Insert(i, Array<T>({el})); }
/// Return the last element in the array.
inline T &Last();
@@ -217,9 +211,6 @@ public:
/// Delete the first entry with value == 'el'.
inline void DeleteFirst(const T &el);
/// Delete entries at @a indices, and resize
inline void DeleteAt(const Array<int> &indices);
/// Delete the whole array.
inline void DeleteAll();
@@ -258,9 +249,6 @@ public:
/// Copy sub array starting from @a offset out to the provided @a sa.
inline void GetSubArray(int offset, int sa_size, Array<T> &sa) const;
/// Set from sub array @sa at @a offset
inline void SetSubArray(int offset, const Array<T> &sa);
/// Prints array to stream with width elements per row.
void Print(std::ostream &out = mfem::out, int width = 4) const;
@@ -338,11 +326,7 @@ public:
the Size to match this Capacity after this.*/
template <typename U>
inline void CopyFrom(const U *src)
{
if (!begin() || size == 0) { return; }
MFEM_ASSERT(begin() && src, "Error in Array::CopyFrom");
std::memcpy(begin(), src, MemoryUsage());
}
{ std::memcpy(begin(), src, MemoryUsage()); }
/// STL-like begin. Returns pointer to the first element of the array.
inline T* begin() { return data; }
@@ -885,22 +869,6 @@ inline int Array<T>::Prepend(const T &el)
return size;
}
template<class T>
inline int Array<T>::Insert(int i, const Array<T> &els)
{
MFEM_ASSERT(i < size, "Insert index is out-of-bounds.");
const int old_size = size;
SetSize(size + els.Size());
for (int j = old_size-1; j >= i; j--)
{
data[j+els.Size()] = data[j];
}
SetSubArray(i, els);
return size;
}
template <class T>
inline T &Array<T>::Last()
{
@@ -963,30 +931,6 @@ inline void Array<T>::DeleteFirst(const T &el)
}
}
template <class T>
inline void Array<T>::DeleteAt(const Array<int> &indices)
{
// Make a copy of the indices, sorted.
Array<int> sorted_indices(indices);
sorted_indices.Sort();
int rm_count = 0;
for (int i = 0; i < size; i++)
{
if (rm_count < sorted_indices.Size() && i == sorted_indices[rm_count])
{
rm_count++;
}
else
{
data[i-rm_count] = data[i]; // shift data rm_count
}
}
// Resize to remove tail
SetSize(size - rm_count);
}
template <class T>
inline void Array<T>::DeleteAll()
{
@@ -1039,18 +983,6 @@ inline void Array<T>::GetSubArray(int offset, int sa_size, Array<T> &sa) const
}
}
template<class T>
inline void Array<T>::SetSubArray(int offset, const Array<T> &sa)
{
MFEM_ASSERT(offset + sa.Size() < size,
"Sub-array with size " << sa.Size() << " is too large to set at offset " <<
offset << ", given array size " << size);
for (int i = 0; i < sa.Size(); i++)
{
data[offset + i] = sa[i];
}
}
template <class T>
inline void Array<T>::operator=(const T &a)
{
+6 -7
View File
@@ -14,7 +14,7 @@
#include "../config/config.hpp"
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#ifdef MFEM_USE_CUDA
#include <cusparse.h>
#include <library_types.h>
#include <cuda_runtime.h>
@@ -22,7 +22,7 @@
#endif
#include "cuda.hpp"
#if defined(MFEM_USE_HIP) && defined(__HIP__)
#ifdef MFEM_USE_HIP
#include <hip/hip_runtime.h>
#endif
#include "hip.hpp"
@@ -43,7 +43,7 @@
#endif
#endif
#if !defined(MFEM_USE_CUDA_OR_HIP)
#if !(defined(MFEM_USE_CUDA) || defined(MFEM_USE_HIP))
#define MFEM_DEVICE
#define MFEM_HOST
#define MFEM_LAMBDA
@@ -55,18 +55,17 @@
#endif
#if !((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
#define MFEM_SHARED
#define MFEM_SYNC_THREAD
#define MFEM_BLOCK_ID(k) 0
#define MFEM_THREAD_ID(k) 0
#define MFEM_THREAD_SIZE(k) 1
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=0; i<N; i++)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) MFEM_FOREACH_THREAD(i,k,N)
#endif
// 'double' and 'float' atomicAdd implementation for previous versions of CUDA
#if defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__) && (__CUDA_ARCH__ < 600)
#if defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ < 600
MFEM_DEVICE inline mfem::real_t atomicAdd(mfem::real_t *add, mfem::real_t val)
{
unsigned long long int *ptr = (unsigned long long int *) add;
@@ -94,7 +93,7 @@ template <typename T>
MFEM_HOST_DEVICE T AtomicAdd(T &add, const T val)
{
#if ((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
return atomicAdd(&add,val);
#else
T old = add;
+8 -16
View File
@@ -425,24 +425,16 @@ public:
~GroupCommunicator();
};
/// General MPI message tags used by MFEM
enum MessageTag
{
DEREFINEMENT_MATRIX_CONSTRUCTION_DATA =
291, /// ParFiniteElementSpace ParallelDerefinementMatrix and
/// ParDerefineMatrixOp
};
enum VarMessageTag
{
NEIGHBOR_ELEMENT_RANK_VM, ///< NeighborElementRankMessage
NEIGHBOR_ORDER_VM, ///< NeighborOrderMessage
NEIGHBOR_DEREFINEMENT_VM, ///< NeighborDerefinementMessage
NEIGHBOR_REFINEMENT_VM, ///< NeighborRefinementMessage
NEIGHBOR_PREFINEMENT_VM, ///< NeighborPRefinementMessage
NEIGHBOR_ROW_VM, ///< NeighborRowMessage
REBALANCE_VM, ///< RebalanceMessage
REBALANCE_DOF_VM, ///< RebalanceDofMessage
NEIGHBOR_ELEMENT_RANK_VM, ///< NeighborElementRankMessage
NEIGHBOR_ORDER_VM, ///< NeighborOrderMessage
NEIGHBOR_DEREFINEMENT_VM, ///< NeighborDerefinementMessage
NEIGHBOR_REFINEMENT_VM, ///< NeighborRefinementMessage
NEIGHBOR_PREFINEMENT_VM, ///< NeighborPRefinementMessage
NEIGHBOR_ROW_VM, ///< NeighborRowMessage
REBALANCE_VM, ///< RebalanceMessage
REBALANCE_DOF_VM ///< RebalanceDofMessage
};
/// \brief Variable-length MPI message containing unspecific binary data.
+1 -1
View File
@@ -24,7 +24,7 @@ void mfem_cuda_error(cudaError_t err, const char *expr, const char *func,
const char *file, int line)
{
mfem::err << "\n\nCUDA error: (" << expr << ") failed with error:\n --> "
<< cudaGetErrorString(err) << " [code: " << (int)err << ']'
<< cudaGetErrorString(err)
<< "\n ... in function: " << func
<< "\n ... in file: " << file << ':' << line << '\n';
mfem_error();
+5 -6
View File
@@ -18,7 +18,7 @@
// CUDA block size used by MFEM.
#define MFEM_CUDA_BLOCKS 256
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#ifdef MFEM_USE_CUDA
#define MFEM_USE_CUDA_OR_HIP
#define MFEM_DEVICE __device__
#define MFEM_HOST __host__
@@ -37,23 +37,22 @@
__FILE__, __LINE__); \
} \
} while (0)
#endif // MFEM_USE_CUDA
// Define the MFEM inner threading macros
#if defined(__CUDA_ARCH__)
#if defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)
#define MFEM_SHARED __shared__
#define MFEM_SYNC_THREAD __syncthreads()
#define MFEM_BLOCK_ID(k) blockIdx.k
#define MFEM_THREAD_ID(k) threadIdx.k
#define MFEM_THREAD_SIZE(k) blockDim.k
#define MFEM_FOREACH_THREAD(i,k,N) for(int i=threadIdx.k; i<N; i+=blockDim.k)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) if(const int i=threadIdx.k; i<N)
#endif // defined(__CUDA_ARCH__)
#endif // defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#endif
namespace mfem
{
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#ifdef MFEM_USE_CUDA
// Function used by the macro MFEM_GPU_CHECK.
void mfem_cuda_error(cudaError_t err, const char *expr, const char *func,
const char *file, int line);
+1 -37
View File
@@ -16,7 +16,6 @@
#include "../fem/ceed/interface/util.hpp"
#endif
#ifdef MFEM_USE_MPI
#include "communication.hpp"
#include "../linalg/hypre.hpp"
#endif
@@ -146,11 +145,6 @@ Device::Device()
Configure(device);
device_env = true;
}
if (GetEnv("MFEM_GPU_AWARE_MPI"))
{
SetGPUAwareMPI(true);
}
}
Device::~Device()
@@ -202,29 +196,6 @@ void Device::Configure(const std::string &device, const int device_id)
{
bmap[internal::backend_name[i]] = internal::backend_list[i];
}
// auto-detect GPU configurations
// assumes only one of HIP or CUDA are available
#ifdef MFEM_USE_HIP
bmap["gpu"] = Backend::HIP;
#ifdef MFEM_USE_RAJA
bmap["raja-gpu"] = Backend::RAJA_HIP;
#endif
#ifdef MFEM_USE_CEED
bmap["ceed-gpu"] = Backend::CEED_HIP;
#endif
// no OCCA+HIP?
#elif defined(MFEM_USE_CUDA)
bmap["gpu"] = Backend::CUDA;
#ifdef MFEM_USE_RAJA
bmap["raja-gpu"] = Backend::RAJA_CUDA;
#endif
#ifdef MFEM_USE_CEED
bmap["ceed-gpu"] = Backend::CEED_CUDA;
#endif
#ifdef MFEM_USE_OCCA
bmap["occa-gpu"] = Backend::OCCA_CUDA;
#endif
#endif
std::string device_option;
std::string::size_type beg = 0, end;
while (1)
@@ -342,13 +313,6 @@ void Device::Print(std::ostream &os)
{
os << ',' << MemoryTypeName[static_cast<int>(device_mem_type)];
}
#ifdef MFEM_USE_MPI
if (Allows(Backend::DEVICE_MASK) &&
Mpi::IsInitialized() && !Mpi::IsFinalized())
{
os << "\nUse GPU-aware MPI: " << (GetGPUAwareMPI() ? "yes" : "no");
}
#endif
os << std::endl;
}
@@ -615,7 +579,7 @@ void Device::Setup(const std::string &device_option, const int device_id)
if (Allows(Backend::DEBUG_DEVICE)) { ngpu = 1; }
}
MemoryType Device::QueryMemoryType(const void* ptr)
MemoryType Device::QueryMemoryType(void *ptr)
{
// from HYPRE's hypre_GetPointerLocation
MemoryType res = MemoryType::HOST;
+3 -7
View File
@@ -198,10 +198,6 @@ public:
'ceed-hip', 'hip', 'debug',
'occa-omp', 'raja-omp', 'omp',
'ceed-cpu', 'occa-cpu', 'raja-cpu', 'cpu'.
- The following backend aliases are also available: 'ceed-gpu',
'occa-gpu', 'raja-gpu', and 'gpu' where they alias their respective
'*-cuda' or '*-hip' backends depending on the MFEM build-time
configuration.
- Multiple backends can be configured at the same time.
- Only one 'occa-*' backend can be configured at a time.
- The backend 'occa-cuda' enables the 'cuda' backend unless 'raja-cuda'
@@ -297,9 +293,9 @@ public:
/// Get the status of GPU-aware MPI flag.
static bool GetGPUAwareMPI() { return Get().mpi_gpu_aware; }
/** Query the device driver for what memory type a given @a ptr is allocated
* with. */
static MemoryType QueryMemoryType(const void* ptr);
/** @brief Query the device driver for what memory type a given @a ptr is
allocated with. */
static MemoryType QueryMemoryType(void *ptr);
/** @brief The number of hardware compute units/streaming multiprocessors
available on a given compute device @a device_id. */
+1 -1
View File
@@ -176,7 +176,7 @@ __device__ void abort_msg(T & msg)
printf(__VA_ARGS__); \
asm("trap;"); \
}
#elif defined(__HIP_DEVICE_COMPILE__)
#elif defined(MFEM_USE_HIP)
#define MFEM_ABORT_KERNEL(...) \
{ \
printf(__VA_ARGS__); \
+12 -12
View File
@@ -158,8 +158,8 @@ private:
#define MFEM_PRAGMA(X) _Pragma(#X)
// MFEM_UNROLL pragma macro that can be used inside MFEM_FORALL macros.
#if defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__) // Clang cuda or nvcc
#ifdef __NVCC__ // nvcc specifically
#if defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)
#ifdef __NVCC__
#define MFEM_UNROLL(N) MFEM_PRAGMA(unroll(N))
#else // Assuming Clang CUDA
#define MFEM_UNROLL(N) MFEM_PRAGMA(unroll N)
@@ -169,12 +169,12 @@ private:
#endif
// MFEM_GPU_FORALL: "parallel for" executed with CUDA or HIP based on the MFEM
// build-time configuration (MFEM_USE_CUDA or MFEM_USE_HIP), and if compiling
// with CUDA/HIP language. Otherwise, this macro is a no-op.
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
// build-time configuration (MFEM_USE_CUDA or MFEM_USE_HIP). If neither CUDA nor
// HIP is enabled, this macro is a no-op.
#if defined(MFEM_USE_CUDA)
#define MFEM_GPU_FORALL(i, N,...) CuWrap1D(N, [=] MFEM_DEVICE \
(int i) {__VA_ARGS__})
#elif defined(MFEM_USE_HIP) && defined(__HIP__)
#elif defined(MFEM_USE_HIP)
#define MFEM_GPU_FORALL(i, N,...) HipWrap1D(N, [=] MFEM_DEVICE \
(int i) {__VA_ARGS__})
#else
@@ -481,7 +481,7 @@ void RajaSeqWrap(const int N, HBODY &&h_body)
/// CUDA backend
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#ifdef MFEM_USE_CUDA
template <typename BODY> __global__ static
void CuKernel1D(const int N, BODY body)
@@ -573,11 +573,11 @@ struct CuWrap<3>
}
};
#endif // defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#endif // MFEM_USE_CUDA
/// HIP backend
#if defined(MFEM_USE_HIP) && defined(__HIP__)
#ifdef MFEM_USE_HIP
template <typename BODY> __global__ static
void HipKernel1D(const int N, BODY body)
@@ -668,7 +668,7 @@ struct HipWrap<3>
}
};
#endif // defined(MFEM_USE_HIP) && defined(__HIP__)
#endif // MFEM_USE_HIP
/// The forall kernel body wrapper
@@ -701,7 +701,7 @@ inline void ForallWrap(const bool use_dev, const int N,
}
#endif
#if defined(MFEM_USE_CUDA) && defined(__CUDACC__)
#ifdef MFEM_USE_CUDA
// If Backend::CUDA is allowed, use it
if (Device::Allows(Backend::CUDA))
{
@@ -709,7 +709,7 @@ inline void ForallWrap(const bool use_dev, const int N,
}
#endif
#if defined(MFEM_USE_HIP) && defined(__HIP__)
#ifdef MFEM_USE_HIP
// If Backend::HIP is allowed, use it
if (Device::Allows(Backend::HIP))
{
+1 -1
View File
@@ -24,7 +24,7 @@ void mfem_hip_error(hipError_t err, const char *expr, const char *func,
const char *file, int line)
{
mfem::err << "\n\nHIP error: (" << expr << ") failed with error:\n --> "
<< hipGetErrorString(err) << " [code: " << (int)err << ']'
<< hipGetErrorString(err)
<< "\n ... in function: " << func
<< "\n ... in file: " << file << ':' << line << '\n';
mfem_error();
+5 -7
View File
@@ -18,7 +18,7 @@
// HIP block size used by MFEM.
#define MFEM_HIP_BLOCKS 256
#if defined(MFEM_USE_HIP) && defined(__HIP__)
#ifdef MFEM_USE_HIP
#define MFEM_USE_CUDA_OR_HIP
#define MFEM_DEVICE __device__
#define MFEM_HOST __host__
@@ -37,20 +37,18 @@
__FILE__, __LINE__); \
} \
} while (0)
#endif // MFEM_USE_HIP
// Define the MFEM inner threading macros
#if defined(__HIP_DEVICE_COMPILE__)
#if defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)
#define MFEM_SHARED __shared__
#define MFEM_SYNC_THREAD __syncthreads()
#define MFEM_BLOCK_ID(k) hipBlockIdx_ ##k
#define MFEM_THREAD_ID(k) hipThreadIdx_ ##k
#define MFEM_THREAD_SIZE(k) hipBlockDim_ ##k
#define MFEM_FOREACH_THREAD(i,k,N) \
for(int i=hipThreadIdx_ ##k; i<N; i+=hipBlockDim_ ##k)
#define MFEM_FOREACH_THREAD_DIRECT(i,k,N) \
if(const int i=hipThreadIdx_ ##k; i<N)
#endif // defined(__HIP_DEVICE_COMPILE__)
#endif // defined(MFEM_USE_HIP) && defined(__HIP__)
for(int i=hipThreadIdx_ ##k; i<N; i+=hipBlockDim_ ##k)
#endif
namespace mfem
{
+8 -2
View File
@@ -513,7 +513,10 @@ public:
void *HtoD(void *dst, const void *src, size_t bytes) override
{ return HipMemcpyHtoD(dst, src, bytes); }
void *DtoD(void* dst, const void* src, size_t bytes) override
{ return HipMemcpyDtoD(dst, src, bytes); }
// Unlike cudaMemcpy(DtoD), hipMemcpy(DtoD) causes a host-side synchronization so
// instead we use hipMemcpyAsync to get similar behavior.
// for more info see: https://github.com/mfem/mfem/pull/2780
{ return HipMemcpyDtoDAsync(dst, src, bytes); }
void *DtoH(void *dst, const void *src, size_t bytes) override
{ return HipMemcpyDtoH(dst, src, bytes); }
};
@@ -655,7 +658,10 @@ public:
return CuMemcpyDtoD(dst, src, bytes);
#endif
#ifdef MFEM_USE_HIP
return HipMemcpyDtoD(dst, src, bytes);
// Unlike cudaMemcpy(DtoD), hipMemcpy(DtoD) causes a host-side synchronization so
// instead we use hipMemcpyAsync to get similar behavior.
// for more info see: https://github.com/mfem/mfem/pull/2780
return HipMemcpyDtoDAsync(dst, src, bytes);
#endif
// rm.copy(dst, const_cast<void*>(src), bytes); return dst;
}
+1 -3
View File
@@ -896,7 +896,6 @@ inline HYPRE_MemoryLocation GetHypreMemoryLocation()
#elif MFEM_HYPRE_VERSION < 23100
return HYPRE_MEMORY_DEVICE;
#else // HYPRE_USING_GPU is defined and MFEM_HYPRE_VERSION >= 23100
if (!HYPRE_Initialized()) { return HYPRE_MEMORY_HOST; }
HYPRE_MemoryLocation loc;
HYPRE_GetMemoryLocation(&loc);
return loc;
@@ -1058,8 +1057,7 @@ inline void Memory<T>::MakeAlias(const Memory &base, int offset, int size)
// register the 'base' if the MemoryManager::Exists():
MemoryManager::Exists()
#else // HYPRE_USING_GPU is defined and MFEM_HYPRE_VERSION >= 23100
IsDeviceMemory(MemoryManager::GetDeviceMemoryType()) ||
(MemoryManager::Exists() && HypreUsingGPU())
MemoryManager::Exists() && HypreUsingGPU()
#endif
)
{
+1 -1
View File
@@ -537,7 +537,7 @@ void reduce(int N, T &res, B &&body, const R &reducer, bool use_dev,
return;
}
#if defined(MFEM_USE_CUDA_OR_HIP)
#if defined(MFEM_USE_HIP) || defined(MFEM_USE_CUDA)
if (use_dev &&
mfem::Device::Allows(Backend::CUDA | Backend::HIP | Backend::RAJA_CUDA |
Backend::RAJA_HIP))
-28
View File
@@ -4405,32 +4405,4 @@ void BatchLUSolve(const DenseTensor &Mlu, const Array<int> &P, Vector &X)
BatchedLinAlg::LUSolve(Mlu, P, X);
}
#ifdef MFEM_USE_LAPACK
void BandedSolve(int KL, int KU, DenseMatrix &AB, DenseMatrix &B,
Array<int> &ipiv)
{
int LDAB = (2*KL) + KU + 1;
int N = AB.NumCols();
int NRHS = B.NumCols();
int info;
ipiv.SetSize(N);
MFEM_LAPACK_PREFIX(gbsv_)(&N, &KL, &KU, &NRHS, AB.GetData(), &LDAB,
ipiv.GetData(), B.GetData(), &N, &info);
MFEM_ASSERT(info == 0, "BandedSolve failed in LAPACK");
}
void BandedFactorizedSolve(int KL, int KU, DenseMatrix &AB, DenseMatrix &B,
bool transpose, Array<int> &ipiv)
{
int LDAB = (2*KL) + KU + 1;
int N = AB.NumCols();
int NRHS = B.NumCols();
char trans = transpose ? 'T' : 'N';
int info;
MFEM_LAPACK_PREFIX(gbtrs_)(&trans, &N, &KL, &KU, &NRHS, AB.GetData(), &LDAB,
ipiv.GetData(), B.GetData(), &N, &info);
MFEM_ASSERT(info == 0, "BandedFactorizedSolve failed in LAPACK");
}
#endif
} // namespace mfem
-7
View File
@@ -1329,13 +1329,6 @@ void BatchLUFactor(DenseTensor &Mlu, Array<int> &P, const real_t TOL = 0.0);
dimension m x n. */
void BatchLUSolve(const DenseTensor &Mlu, const Array<int> &P, Vector &X);
#ifdef MFEM_USE_LAPACK
void BandedSolve(int KL, int KU, DenseMatrix &AB, DenseMatrix &B,
Array<int> &ipiv);
void BandedFactorizedSolve(int KL, int KU, DenseMatrix &AB, DenseMatrix &B,
bool transpose, Array<int> &ipiv);
#endif
// Inline methods
inline real_t &DenseMatrix::operator()(int i, int j)
-12
View File
@@ -2574,18 +2574,6 @@ void HypreParMatrix::EliminateBC(const Array<int> &ess_dofs,
#if defined(HYPRE_USING_GPU)
if (HypreUsingGPU())
{
#if defined(HYPRE_WITH_GPU_AWARE_MPI) || defined(HYPRE_USING_GPU_AWARE_MPI)
// hypre_GetGpuAwareMPI() was introduced in v2.31.0, however, its value
// is not checked in hypre_ParCSRCommHandleCreate_v2() before v2.33.0,
// instead only HYPRE_WITH_GPU_AWARE_MPI is checked.
#if MFEM_HYPRE_VERSION >= 23300
if (hypre_GetGpuAwareMPI())
#endif
{
// ensure int_buf_data has been computed before sending it
MFEM_STREAM_SYNC;
}
#endif
// Try to use device-aware MPI for the communication if available
comm_handle = hypre_ParCSRCommHandleCreate_v2(
11, comm_pkg, HYPRE_MEMORY_DEVICE, int_buf_data,
-7
View File
@@ -42,13 +42,6 @@ extern "C" void
MFEM_LAPACK_PREFIX(getri_)(int *N, real_t *A, int *LDA, int *IPIV, real_t *WORK,
int *LWORK, int *INFO);
extern "C" void
MFEM_LAPACK_PREFIX(gbsv_)(int *, int *, int *, int *, real_t *, int *, int *,
real_t *, int *, int *);
extern "C" void
MFEM_LAPACK_PREFIX(gbtrs_)(char *, int *, int *, int *, int *, real_t *, int *,
int *, real_t *, int *, int *);
extern "C" void
MFEM_LAPACK_PREFIX(syevr_)(char *JOBZ, char *RANGE, char *UPLO, int *N,
real_t *A, int *LDA, real_t *VL, real_t *VU, int *IL,
int *IU, real_t *ABSTOL, int *M, real_t *W,
-22
View File
@@ -668,28 +668,6 @@ void Vector::median(const Vector &lo, const Vector &hi)
});
}
void Vector::Insert(int offset, const Vector &sv)
{
const int old_size = size;
if (sv.Size() + old_size > Capacity())
{
Vector copy = *this;
SetSize(size + sv.Size());
SetVector(copy, 0);
}
else
{
SetSize(size + sv.Size());
}
for (int j = old_size-1; j >= offset; j--)
{
data[j+sv.Size()] = data[j];
}
SetVector(sv, offset);
}
void Vector::GetSubVector(const Array<int> &dofs, Vector &elemvect) const
{
const int n = dofs.Size();
-32
View File
@@ -171,9 +171,6 @@ public:
/// Resize the vector to size @a s using the MemoryType of @a v.
void SetSize(int s, const Vector &v) { SetSize(s, v.GetMemory().GetMemoryType()); }
/// Delete elements at @a indices and resize vector accordingly
void DeleteAt(const Array<int> &indices);
/// Set the Vector data.
/// @warning This method should be called only when OwnsData() is false.
void SetData(real_t *d) { data.Wrap(d, data.Capacity(), false); }
@@ -399,12 +396,6 @@ public:
/// v = median(v,lo,hi) entrywise. Implementation assumes lo <= hi.
void median(const Vector &lo, const Vector &hi);
/// Insert sub Vector @a sv at @a offset and resize
void Insert(int offset, const Vector &sv);
/// Insert @a value at @a offset and resize
void Insert(int offset, const real_t value) { Insert(offset, Vector({value})); }
/// Extract entries listed in @a dofs to the output Vector @a elemvect.
/** Negative dof values cause the -dof-1 position in @a elemvect to receive
the -val in from this Vector. */
@@ -630,29 +621,6 @@ inline void Vector::SetSize(int s, MemoryType mt)
data.UseDevice(use_dev);
}
inline void Vector::DeleteAt(const Array<int> &indices)
{
// Make copy of the indices, sorted.
Array<int> sorted_indices(indices);
sorted_indices.Sort();
int rm_count = 0;
for (int i = 0; i < size; i++)
{
if (rm_count < sorted_indices.Size() && i == sorted_indices[rm_count])
{
rm_count++;
}
else
{
data[i-rm_count] = data[i]; // shift data rm_count
}
}
// Resize to remove tail
SetSize(size - rm_count);
}
inline void Vector::NewMemoryAndSize(const Memory<real_t> &mem, int s,
bool own_mem)
{
+27 -35
View File
@@ -26,12 +26,6 @@
}\
}
#if defined(MFEM_USE_DOUBLE)
#define MFEM_NETCDF_REAL_T NC_DOUBLE
#elif defined(MFEM_USE_SINGLE)
#define MFEM_NETCDF_REAL_T NC_FLOAT
#endif
namespace mfem
{
@@ -141,18 +135,18 @@ public:
/// @brief Writes the mesh to an ExodusII file.
/// @param fpath The path to the file.
/// @param flags NC_CLOBBER will overwrite existing file.
void PrintExodusII(const std::string &fpath, int flags = NC_CLOBBER);
void PrintExodusII(std::string fpath, int flags = NC_CLOBBER);
/// @brief Static method for writing a mesh to an ExodusII file.
/// @param mesh The mesh to write to the file.
/// @param fpath The path to the file.
/// @param flags NetCDF file flags.
static void PrintExodusII(Mesh & mesh, const std::string &fpath,
static void PrintExodusII(Mesh & mesh, std::string fpath,
int flags = NC_CLOBBER);
protected:
/// @brief Closes any open file and creates a NetCDF file using selected flags.
void OpenExodusII(const std::string &fpath, int flags);
void OpenExodusII(std::string fpath, int flags);
/// @brief Closes any open file.
void CloseExodusII();
@@ -173,9 +167,9 @@ protected:
std::unordered_set<int> GenerateUniqueNodeIDs();
/// @brief Populates vectors with x, y, z coordinates from mesh.
void ExtractVertexCoordinates(std::vector<real_t> &coordx,
std::vector<real_t> &coordy,
std::vector<real_t> &coordz);
void ExtractVertexCoordinates(std::vector<double> & coordx,
std::vector<double> & coordy,
std::vector<double> & coordz);
/// @brief Writes node connectivity for a particular block.
/// @param block_id The block to write to the file.
@@ -193,7 +187,7 @@ protected:
/// @brief Writes the number of elements in the mesh.
void WriteNumOfElements();
/// @brief Writes the floating-point word size (sizeof(real_t)).
/// @brief Writes the floating-point word size (4 == float; 8 == double).
void WriteFloatingPointWordSize();
/// @brief Writes the API version.
@@ -297,7 +291,7 @@ private:
std::map<int, std::vector<int>> exodusII_side_ids_for_boundary_id;
};
void Mesh::PrintExodusII(const std::string &fpath)
void Mesh::PrintExodusII(const std::string fpath)
{
ExodusIIWriter::PrintExodusII(*this, fpath);
}
@@ -368,7 +362,7 @@ void ExodusIIWriter::WriteExodusIIMeshInformation()
WriteNodeSets();
}
void ExodusIIWriter::PrintExodusII(const std::string &fpath, int flags)
void ExodusIIWriter::PrintExodusII(std::string fpath, int flags)
{
OpenExodusII(fpath, flags);
@@ -380,7 +374,7 @@ void ExodusIIWriter::PrintExodusII(const std::string &fpath, int flags)
mfem::out << "Mesh successfully written to Exodus II file" << std::endl;
}
void ExodusIIWriter::PrintExodusII(Mesh &mesh, const std::string &fpath,
void ExodusIIWriter::PrintExodusII(Mesh & mesh, std::string fpath,
int flags)
{
ExodusIIWriter writer(mesh);
@@ -388,7 +382,7 @@ void ExodusIIWriter::PrintExodusII(Mesh &mesh, const std::string &fpath,
writer.PrintExodusII(fpath, flags);
}
void ExodusIIWriter::OpenExodusII(const std::string &fpath, int flags)
void ExodusIIWriter::OpenExodusII(std::string fpath, int flags)
{
CloseExodusII(); // Close any open files.
@@ -428,7 +422,7 @@ void ExodusIIWriter::WriteNumOfElements()
void ExodusIIWriter::WriteFloatingPointWordSize()
{
const int word_size = sizeof(real_t);
const int word_size = 8;
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_FLOATING_POINT_WORD_SIZE_LABEL,
NC_INT, 1,
&word_size);
@@ -436,15 +430,13 @@ void ExodusIIWriter::WriteFloatingPointWordSize()
void ExodusIIWriter::WriteAPIVersion()
{
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_API_VERSION_LABEL, MFEM_NETCDF_REAL_T,
1,
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_API_VERSION_LABEL, NC_FLOAT, 1,
&ExodusIILabels::EXODUS_API_VERSION);
}
void ExodusIIWriter::WriteDatabaseVersion()
{
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_DATABASE_VERSION_LABEL,
MFEM_NETCDF_REAL_T, 1,
PutAtt(NC_GLOBAL, ExodusIILabels::EXODUS_DATABASE_VERSION_LABEL, NC_FLOAT, 1,
&ExodusIILabels::EXODUS_DATABASE_VERSION);
}
@@ -615,25 +607,25 @@ void ExodusIIWriter::WriteNodalCoordinates()
DefineDimension("num_nodes", num_nodes, &num_nodes_id);
// 3. Extract the nodal coordinates.
// NB: writes in format real_t (double or float); ndims = 1 (vector).
// NB: assume doubles (could be floats!); ndims = 1 (vector).
// https://docs.unidata.ucar.edu/netcdf-c/current/group__variables.html#gac7e8662c51f3bb07d1fc6d6c6d9052c8
std::vector<real_t> coordx(num_nodes);
std::vector<real_t> coordy(num_nodes);
std::vector<real_t> coordz(mesh.Dimension() == 3 ? num_nodes : 0);
std::vector<double> coordx(num_nodes);
std::vector<double> coordy(num_nodes);
std::vector<double> coordz(mesh.Dimension() == 3 ? num_nodes : 0);
ExtractVertexCoordinates(coordx, coordy, coordz);
// 4. Define and put the nodal coordinates.
DefineAndPutVar(ExodusIILabels::EXODUS_COORDX_LABEL, MFEM_NETCDF_REAL_T, 1,
DefineAndPutVar(ExodusIILabels::EXODUS_COORDX_LABEL, NC_DOUBLE, 1,
&num_nodes_id,
coordx.data());
DefineAndPutVar(ExodusIILabels::EXODUS_COORDY_LABEL, MFEM_NETCDF_REAL_T, 1,
DefineAndPutVar(ExodusIILabels::EXODUS_COORDY_LABEL, NC_DOUBLE, 1,
&num_nodes_id,
coordy.data());
if (mesh.Dimension() == 3)
{
DefineAndPutVar(ExodusIILabels::EXODUS_COORDZ_LABEL, MFEM_NETCDF_REAL_T, 1,
DefineAndPutVar(ExodusIILabels::EXODUS_COORDZ_LABEL, NC_DOUBLE, 1,
&num_nodes_id,
coordz.data());
}
@@ -778,9 +770,9 @@ void ExodusIIWriter::WriteNodeConnectivityForBlock(const int block_id)
}
void ExodusIIWriter::ExtractVertexCoordinates(std::vector<real_t> & coordx,
std::vector<real_t> & coordy,
std::vector<real_t> & coordz)
void ExodusIIWriter::ExtractVertexCoordinates(std::vector<double> & coordx,
std::vector<double> & coordy,
std::vector<double> & coordz)
{
if (mesh.GetNodes()) // Higher-order.
{
@@ -790,7 +782,7 @@ void ExodusIIWriter::ExtractVertexCoordinates(std::vector<real_t> & coordx,
sorted_node_ids.assign(unordered_node_ids.begin(), unordered_node_ids.end());
std::sort(sorted_node_ids.begin(), sorted_node_ids.end());
real_t coordinates[3];
double coordinates[3];
for (size_t i = 0; i < sorted_node_ids.size(); i++)
{
int node_id = sorted_node_ids[i];
@@ -810,7 +802,7 @@ void ExodusIIWriter::ExtractVertexCoordinates(std::vector<real_t> & coordx,
{
for (int ivertex = 0; ivertex < mesh.GetNV(); ivertex++)
{
real_t *coordinates = mesh.GetVertex(ivertex);
double * coordinates = mesh.GetVertex(ivertex);
coordx[ivertex] = coordinates[0];
coordy[ivertex] = coordinates[1];
@@ -1088,4 +1080,4 @@ void ExodusIIWriter::CheckNodalFESpaceIsSecondOrderH1() const
#endif
}
}
+2 -3
View File
@@ -2994,7 +2994,6 @@ void Mesh::DoNodeReorder(DSTable *old_v_to_v, Table *old_elem_vert)
const int num_edge_dofs = old_dofs.Size();
// Save the original nodes
Nodes->HostReadWrite(); // for "(*Nodes)() = "
const Vector onodes = *Nodes;
// vertex dofs do not need to be moved
@@ -13257,7 +13256,7 @@ void Mesh::ScaleElements(real_t sf)
delete [] vn;
}
void Mesh::Transform(std::function<void(const Vector &, Vector&)> f)
void Mesh::Transform(void (*f)(const Vector&, Vector&))
{
// TODO: support for different new spaceDim.
if (Nodes == NULL)
@@ -13270,7 +13269,7 @@ void Mesh::Transform(std::function<void(const Vector &, Vector&)> f)
vold(j) = vertices[i](j);
}
vnew.SetData(vertices[i]());
f(vold, vnew);
(*f)(vold, vnew);
}
}
else
+9 -12
View File
@@ -588,10 +588,9 @@ protected:
void Loader(std::istream &input, int generate_edges = 0,
std::string parse_tag = "");
/** @brief If NURBS mesh, write NURBS format. If NCMesh, write mfem v1.1
format. If section_delimiter is empty, write mfem v1.0 format. Otherwise,
write mfem v1.2 format with the given section_delimiter at the end.
/** If NURBS mesh, write NURBS format. If NCMesh, write mfem v1.1 format.
If section_delimiter is empty, write mfem v1.0 format. Otherwise, write
mfem v1.2 format with the given section_delimiter at the end.
If @a comments is non-empty, it will be printed after the first line of
the file, and each line should begin with '#'. */
void Printer(std::ostream &os = mfem::out,
@@ -2254,7 +2253,7 @@ public:
void ScaleSubdomains (real_t sf);
void ScaleElements (real_t sf);
void Transform(std::function<void(const Vector &, Vector&)> f);
void Transform(void (*f)(const Vector&, Vector&));
void Transform(VectorCoefficient &deformation);
/** @brief This function should be called after the mesh node coordinates
@@ -2483,12 +2482,10 @@ public:
/// Print the mesh to the given stream using Netgen/Truegrid format.
virtual void PrintXG(std::ostream &os = mfem::out) const;
/** @brief Print the mesh to the given stream using the default MFEM mesh
format.
\see mfem::ofgzstream() for on-the-fly compression of ascii outputs. If
@a comments is non-empty, it will be printed after the first line of the
file, and each line should begin with '#'. */
/// Print the mesh to the given stream using the default MFEM mesh format.
/// \see mfem::ofgzstream() for on-the-fly compression of ascii outputs. If
/// @a comments is non-empty, it will be printed after the first line of the
/// file, and each line should begin with '#'.
virtual void Print(std::ostream &os = mfem::out,
const std::string &comments = "") const
{ Printer(os, "", comments); }
@@ -2540,7 +2537,7 @@ public:
#ifdef MFEM_USE_NETCDF
/// @brief Export a mesh to an Exodus II file.
void PrintExodusII(const std::string &fpath);
void PrintExodusII(const std::string fpath);
#endif
/** @brief Prints the mesh with boundary elements given by the boundary of
+13 -3
View File
@@ -802,11 +802,21 @@ struct BufferReader : BufferReaderBase
{
// Each "data block" is preceded by a header that is either UInt32 or
// UInt64. The rest of the data follows.
MFEM_VERIFY(sizeof(F)*n == ReadHeaderEntry(header_buf),
"AppendedData: wrong data size");
uint64_t data_size;
if (header_type == UINT32_HEADER)
{
uint32_t *data_size_32 = (uint32_t *)header_buf;
data_size = *data_size_32;
}
else
{
uint64_t *data_size_64 = (uint64_t *)header_buf;
data_size = *data_size_64;
}
MFEM_VERIFY(sizeof(F)*n == data_size, "AppendedData: wrong data size");
}
if (std::is_same_v<T, F>)
if (std::is_same<T, F>::value)
{
// Special case: no type conversions necessary, so can just memcpy
memcpy(dest, buf, sizeof(T)*n);
+27 -99
View File
@@ -9,13 +9,8 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "nurbs.hpp"
#include "point.hpp"
#include "segment.hpp"
#include "quadrilateral.hpp"
#include "hexahedron.hpp"
#include "../fem/gridfunc.hpp"
#include "mesh_headers.hpp"
#include "../fem/fem.hpp"
#include "../general/text.hpp"
#include <fstream>
@@ -38,7 +33,6 @@ KnotVector::KnotVector(istream &input)
knot.Load(input, NumOfControlPoints + Order + 1);
GetElements();
coarse = false;
}
KnotVector::KnotVector(int order, int NCP)
@@ -47,13 +41,12 @@ KnotVector::KnotVector(int order, int NCP)
NumOfControlPoints = NCP;
knot.SetSize(NumOfControlPoints + Order + 1);
NumOfElements = 0;
coarse = false;
knot = -1.;
}
KnotVector::KnotVector(int order, const Vector& intervals,
const Array<int>& continuity)
const Array<int>& continuity )
{
// NOTE: This may need to be generalized to support periodicity
// in the future.
@@ -93,7 +86,6 @@ KnotVector::KnotVector(int order, const Vector& intervals,
++NumOfElements;
}
}
coarse = false;
}
KnotVector &KnotVector::operator=(const KnotVector &kv)
@@ -151,7 +143,7 @@ void KnotVector::UniformRefinement(Vector &newknots, int rf) const
{
for (int m = 1; m < rf; ++m)
{
newknots(j) = ((1.0 - (m * h)) * knot(i)) + (m * h * knot(i+1));
newknots(j) = m * h * (knot(i) + knot(i+1));
j++;
}
}
@@ -340,7 +332,7 @@ void KnotVector::PrintFunctions(std::ostream &os, int samples) const
}
}
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Algorithm A2.2 p. 70
void KnotVector::CalcShape(Vector &shape, int i, real_t xi) const
{
@@ -367,7 +359,7 @@ void KnotVector::CalcShape(Vector &shape, int i, real_t xi) const
}
}
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Algorithm A2.3 p. 72
void KnotVector::CalcDShape(Vector &grad, int i, real_t xi) const
{
@@ -425,7 +417,7 @@ void KnotVector::CalcDShape(Vector &grad, int i, real_t xi) const
}
}
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Algorithm A2.3 p. 72
void KnotVector::CalcDnShape(Vector &gradn, int n, int i, real_t xi) const
{
@@ -545,11 +537,11 @@ void KnotVector::FindMaxima(Array<int> &ks, Vector &xi, Vector &u) const
int i = j - d;
if (isElement(i))
{
arg1 = std::numeric_limits<real_t>::epsilon() / 2_r;
arg1 = 1e-16;
CalcShape(shape, i, arg1);
max1 = shape[d];
arg2 = 1_r - arg1;
arg2 = 1-(1e-16);
CalcShape(shape, i, arg2);
max2 = shape[d];
@@ -587,9 +579,9 @@ void KnotVector::FindMaxima(Array<int> &ks, Vector &xi, Vector &u) const
}
}
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
// Algorithm A9.1 p. 369
void KnotVector::FindInterpolant(Array<Vector*> &x, bool reuse_inverse)
void KnotVector::FindInterpolant(Array<Vector*> &x)
{
int order = GetOrder();
int ncp = GetNCP();
@@ -597,93 +589,29 @@ void KnotVector::FindInterpolant(Array<Vector*> &x, bool reuse_inverse)
// Find interpolation points
Vector xi_args, u_args;
Array<int> i_args;
FindMaxima(i_args, xi_args, u_args);
FindMaxima(i_args,xi_args, u_args);
// Assemble collocation matrix
#ifdef MFEM_USE_LAPACK
// If using LAPACK, we use banded matrix storage (order + 1 nonzeros per row).
// Find banded structure of matrix.
int KL = 0; // Number of subdiagonals
int KU = 0; // Number of superdiagonals
Vector shape(order+1);
DenseMatrix A(ncp,ncp);
A = 0.0;
for (int i = 0; i < ncp; i++)
{
CalcShape(shape, i_args[i], xi_args[i]);
for (int p = 0; p < order+1; p++)
{
const int col = i_args[i] + p;
if (col < i)
{
KL = std::max(KL, i - col);
}
else if (i < col)
{
KU = std::max(KU, col - i);
}
A(i,i_args[i] + p) = shape[p];
}
}
const int LDAB = (2*KL) + KU + 1;
const int N = ncp;
fact_AB.SetSize(LDAB, N);
#else
// Without LAPACK, we store and invert a DenseMatrix (inefficient).
if (!reuse_inverse)
{
A_coll_inv.SetSize(ncp, ncp);
A_coll_inv = 0.0;
}
#endif
Vector shape(order+1);
if (!reuse_inverse) // Set collocation matrix entries
{
for (int i = 0; i < ncp; i++)
{
CalcShape(shape, i_args[i], xi_args[i]);
for (int p = 0; p < order+1; p++)
{
const int j = i_args[i] + p;
#ifdef MFEM_USE_LAPACK
fact_AB(KL+KU+i-j,j) = shape[p];
#else
A_coll_inv(i,j) = shape[p];
#endif
}
}
}
// Solve the system
#ifdef MFEM_USE_LAPACK
const int NRHS = x.Size();
DenseMatrix B(N, NRHS);
for (int j=0; j<NRHS; ++j)
{
for (int i=0; i<N; ++i) { B(i, j) = (*x[j])[i]; }
}
if (reuse_inverse)
{
BandedFactorizedSolve(KL, KU, fact_AB, B, false, fact_ipiv);
}
else
{
BandedSolve(KL, KU, fact_AB, B, fact_ipiv);
}
for (int j=0; j<NRHS; ++j)
{
for (int i=0; i<N; ++i) { (*x[j])[i] = B(i, j); }
}
#else
if (!reuse_inverse) { A_coll_inv.Invert(); }
// Solve problems
A.Invert();
Vector tmp;
for (int i = 0; i < x.Size(); i++)
for (int i= 0; i < x.Size(); i++)
{
tmp = *x[i];
A_coll_inv.Mult(tmp, *x[i]);
A.Mult(tmp,*x[i]);
}
#endif
}
int KnotVector::findKnotSpan(real_t u) const
@@ -1485,7 +1413,7 @@ void NURBSPatch::DegreeElevate(int t)
}
}
// Routine from "The NURBS Book" - 2nd ed - Piegl and Tiller
// Routine from "The NURBS book" - 2nd ed - Piegl and Tiller
void NURBSPatch::DegreeElevate(int dir, int t)
{
if (dir >= kv.Size() || dir < 0)
@@ -1503,8 +1431,8 @@ void NURBSPatch::DegreeElevate(int dir, int t)
KnotVector &oldkv = *kv[dir];
oldkv.GetElements();
auto *newpatch = new NURBSPatch(this, dir, oldkv.GetOrder() + t,
oldkv.GetNCP() + oldkv.GetNE()*t);
NURBSPatch *newpatch = new NURBSPatch(this, dir, oldkv.GetOrder() + t,
oldkv.GetNCP() + oldkv.GetNE()*t);
NURBSPatch &newp = *newpatch;
KnotVector &newkv = *newp.GetKV(dir);
@@ -2449,7 +2377,7 @@ NURBSExtension::NURBSExtension(Mesh *mesh_array[], int num_pieces)
}
NURBSExtension::NURBSExtension(const Mesh *patch_topology,
const Array<const NURBSPatch*> &patches_)
const Array<const NURBSPatch*> patches_)
{
// Basic topology checks
MFEM_VERIFY(patches_.Size() > 0, "Must have at least one patch");
@@ -4659,7 +4587,7 @@ void NURBSExtension::KnotInsert(Array<Vector *> &kv)
// Flip vector
int size = pkvc[d]->Size();
int ns = static_cast<int>(ceil(size/2.0));
int ns = ceil(size/2.0);
for (int j = 0; j < ns; j++)
{
real_t tmp = apb - pkvc[d]->Elem(j);
@@ -4719,7 +4647,7 @@ void NURBSExtension::KnotRemove(Array<Vector *> &kv, real_t tol)
// Flip vector
int size = pkvc[d]->Size();
int ns = static_cast<int>(ceil(size/2.0));
int ns = ceil(size/2.0);
for (int j = 0; j < ns; j++)
{
real_t tmp = apb - pkvc[d]->Elem(j);
+7 -20
View File
@@ -22,6 +22,7 @@
#include "../general/communication.hpp"
#endif
#include <iostream>
#include <set>
namespace mfem
{
@@ -54,7 +55,7 @@ protected:
public:
/// Create an empty KnotVector.
KnotVector() = default;
KnotVector() { }
/** @brief Create a KnotVector by reading data from stream @a input. Two
integers are read, for order and number of control points. */
@@ -73,7 +74,7 @@ public:
polynomial degree). Periodicity is not supported.
*/
KnotVector(int order, const Vector& intervals,
const Array<int>& continuity);
const Array<int>& continuity );
/// Copy constructor.
KnotVector(const KnotVector &kv) { (*this) = kv; }
@@ -143,13 +144,8 @@ public:
/** @brief Global curve interpolation through the points @a x (overwritten).
@a x is an array with the length of the spatial dimension containing
vectors with spatial coordinates. The control points of the interpolated
curve are returned in @a x in the same form.
The inverse of the collocation matrix, used in the interpolation, is
stored for repeated calls and used if @a reuse_inverse is true. Reuse is
valid only if this KnotVector has not changed since the initial call with
@a reuse_inverse false. */
void FindInterpolant(Array<Vector*> &x, bool reuse_inverse = false);
curve are returned in @a x in the same form. */
void FindInterpolant(Array<Vector*> &x);
/** Set @a diff, comprised of knots in @a kv not contained in this KnotVector.
@a kv must be of the same order as this KnotVector. The current
@@ -207,14 +203,6 @@ public:
/** Flag to indicate whether the KnotVector has been coarsened, which means
it is ready for non-nested refinement. */
bool coarse;
#ifdef MFEM_USE_LAPACK
// Data for reusing banded matrix factorization in FindInterpolant().
DenseMatrix fact_AB; /// Banded matrix factorization
Array<int> fact_ipiv; /// Row pivot indices
#else
DenseMatrix A_coll_inv; /// Collocation matrix inverse
#endif
};
@@ -298,7 +286,7 @@ public:
includes the weight. The array of control point coordinates stores each
point's coordinates contiguously, and points are ordered in a standard
ijk grid ordering. */
NURBSPatch(Array<const KnotVector *> &kv_, int dim_,
NURBSPatch(Array<const KnotVector *> &kv_, int dim_,
const real_t* control_points);
/// Constructor for a patch of dimension equal to the size of @a kv.
@@ -713,8 +701,7 @@ public:
NURBSExtension(Mesh *mesh_array[], int num_pieces);
NURBSExtension(const Mesh *patch_topology,
const Array<const NURBSPatch*> &patches_);
NURBSExtension(const Mesh *patch_topology, const Array<const NURBSPatch*> p);
/// Copy assignment not supported.
NURBSExtension& operator=(const NURBSExtension&) = delete;
+1 -2
View File
@@ -3132,12 +3132,11 @@ void ParMesh::GetFaceNbrElementTransformation(
pNodes->ParFESpace()->GetFaceNbrElementVDofs(FaceNo, vdofs);
int n = vdofs.Size()/spaceDim;
pointmat.SetSize(spaceDim, n);
pNodes->FaceNbrData().HostRead();
for (int k = 0; k < spaceDim; k++)
{
for (int j = 0; j < n; j++)
{
pointmat(k,j) = AsConst(pNodes->FaceNbrData())(vdofs[n*k+j]);
pointmat(k,j) = (pNodes->FaceNbrData())(vdofs[n*k+j]);
}
}
+9 -1
View File
@@ -257,7 +257,15 @@ template <typename SubMeshT>
void AddBoundaryElements(SubMeshT &mesh,
const std::unordered_map<int,int> &lface_to_boundary_attribute)
{
const int num_codim_1 = mesh.GetNumFaces();
mesh.Dimension();
const int num_codim_1 = [&mesh]()
{
auto Dim = mesh.Dimension();
if (Dim == 1) { return mesh.GetNV(); }
else if (Dim == 2) { return mesh.GetNEdges(); }
else if (Dim == 3) { return mesh.GetNFaces(); }
else { MFEM_ABORT("Invalid dimension."); return -1; }
}();
if (mesh.Dimension() == 3)
{
+42 -44
View File
@@ -84,7 +84,7 @@ void VTKHDF::EnsureSteps()
}
hid_t VTKHDF::EnsureDataset(hid_t f, const std::string &name, hid_t type,
Dims &dims)
int ndims)
{
const char *name_c = name.c_str();
@@ -94,23 +94,20 @@ hid_t VTKHDF::EnsureDataset(hid_t f, const std::string &name, hid_t type,
if (status == 0)
{
// Dataset does not exist, create it.
const int ndims = dims.ndims;
// The dataset is allowed to grow in the first dimension, but is fixed
// in size in all other dimesions; the maximum dataset size is same as
// dims, but unlimited in first dimension.
Dims max_dims = dims;
max_dims[0] = H5S_UNLIMITED;
const hid_t fspace = H5Screate_simple(ndims, dims, max_dims);
Dims dims(ndims);
Dims maxdims(ndims, H5S_UNLIMITED);
const hid_t fspace = H5Screate_simple(ndims, dims, maxdims);
Dims chunk(ndims);
size_t chunk_size_bytes = 1024 * 1024 / 2; // 0.5 MB
const size_t t_bytes = H5Tget_size(type);
for (int i = 1; i < ndims; ++i)
{
chunk[i] = dims[i];
chunk_size_bytes /= dims[i];
chunk[i] = 16;
chunk_size_bytes /= 16;
}
chunk[0] = chunk_size_bytes / t_bytes;
for (int i = 1; i < ndims; ++i) { chunk[i] = 16; }
const hid_t dcpl = H5Pcreate(H5P_DATASET_CREATE);
H5Pset_chunk(dcpl, ndims, chunk);
if (compression_level >= 0)
@@ -127,19 +124,7 @@ hid_t VTKHDF::EnsureDataset(hid_t f, const std::string &name, hid_t type,
else if (status > 0)
{
// Dataset exists, open it.
const hid_t d = H5Dopen2(f, name_c, H5P_DEFAULT);
// Resize the dataset, set dims to its new size.
Dims old_dims(dims.ndims);
const hid_t dspace = H5Dget_space(d);
const int ndims_dset = H5Sget_simple_extent_ndims(dspace);
MFEM_VERIFY(ndims_dset == dims.ndims, "");
H5Sget_simple_extent_dims(dspace, old_dims, NULL);
H5Sclose(dspace);
dims[0] += old_dims[0];
H5Dset_extent(d, dims);
return d;
return H5Dopen2(f, name_c, H5P_DEFAULT);
}
else
{
@@ -175,13 +160,27 @@ void VTKHDF::AppendParData(hid_t f, const std::string &name, hsize_t locsize,
hsize_t offset, Dims globsize, T *data)
{
const int ndims = globsize.ndims;
Dims dims = globsize;
const hid_t d = EnsureDataset(f, name, GetTypeID<T>(), dims);
const hid_t d = EnsureDataset(f, name, GetTypeID<T>(), ndims);
// Resize the dataset, set dims to its new size.
hsize_t old_size;
Dims dims(ndims);
{
const hid_t dspace = H5Dget_space(d);
const int ndims_dset = H5Sget_simple_extent_ndims(dspace);
MFEM_VERIFY(ndims_dset == ndims, "");
H5Sget_simple_extent_dims(dspace, dims, NULL);
H5Sclose(dspace);
old_size = dims[0];
dims[0] += globsize[0];
for (int i = 1; i < ndims; ++i) { dims[i] = globsize[i]; }
H5Dset_extent(d, dims);
}
// Write the new entry.
const hid_t dspace = H5Dget_space(d);
Dims start(ndims);
start[0] = dims[0] - globsize[0] + offset;
start[0] = old_size + offset;
Dims count(ndims);
count[0] = locsize;
for (int i = 1; i < ndims; ++i) { count[i] = globsize[i]; }
@@ -335,14 +334,14 @@ void VTKHDF::Truncate(const real_t t)
}
// Index of found time index (may be 'one-past-the-end' if not found)
const ptrdiff_t i = std::distance(tvals.begin(), it);
const int i = std::distance(tvals.begin(), it);
// Only truncate if needed
const bool truncate = it != tvals.end();
// Number of steps we are keeping
nsteps = i;
H5LTset_attribute_ulong(vtk, "Steps", "NSteps", &nsteps, 1);
H5LTset_attribute_int(vtk, "Steps", "NSteps", &nsteps, 1);
// We want to continue writing immediately after step 'i - 1'. If i = 0,
// then this is at the beginning of the file, and the offsets do not need
@@ -510,7 +509,7 @@ void VTKHDF::UpdateSteps(real_t t)
// Set the NSteps attribute
++nsteps;
H5LTset_attribute_ulong(steps, ".", "NSteps", &nsteps, 1);
H5LTset_attribute_int(steps, ".", "NSteps", &nsteps, 1);
AppendValue(steps, "Values", t);
AppendValue(steps, "PartOffsets", part_offset);
@@ -619,16 +618,16 @@ void VTKHDF::SaveMesh(const Mesh &mesh, bool high_order, int ref)
for (int i = 0; i < pmat.Width(); i++)
{
points.push_back(FP_T(pmat(0,i)));
if (pmat.Height() > 1) { points.push_back(FP_T(pmat(1,i))); }
points.push_back(pmat(0,i));
if (pmat.Height() > 1) { points.push_back(pmat(1,i)); }
else { points.push_back(0.0); }
if (pmat.Height() > 2) { points.push_back(FP_T(pmat(2,i))); }
if (pmat.Height() > 2) { points.push_back(pmat(2,i)); }
else { points.push_back(0.0); }
}
}
}
const int ne_0 = mesh.GetNE();
const hsize_t ne_0 = mesh.GetNE();
const hsize_t ne = high_order ? ne_0 : ne_ref;
AppendParData(vtk, "NumberOfPoints", 1, mpi_rank, mpi_dims, &np);
@@ -658,7 +657,7 @@ void VTKHDF::SaveMesh(const Mesh &mesh, bool high_order, int ref)
if (high_order)
{
Array<int> local_connectivity;
for (int e = 0; e < int(ne); ++e)
for (size_t e = 0; e < ne; ++e)
{
offsets[e] = off;
const Geometry::Type geom = mesh.GetElementGeometry(e);
@@ -676,7 +675,7 @@ void VTKHDF::SaveMesh(const Mesh &mesh, bool high_order, int ref)
{
int off_0 = 0;
int e_ref = 0;
for (int e = 0; e < ne_0; ++e)
for (hsize_t e = 0; e < ne_0; ++e)
{
const Geometry::Type geom = mesh.GetElementGeometry(e);
const int nv = get_nv(e);
@@ -715,13 +714,12 @@ void VTKHDF::SaveMesh(const Mesh &mesh, bool high_order, int ref)
const int *vtk_geom_map =
high_order ? VTKGeometry::HighOrderMap : VTKGeometry::Map;
int e_ref = 0;
for (int e = 0; e < ne_0; ++e)
for (hsize_t e = 0; e < ne_0; ++e)
{
const int ne_ref_e = get_ne_ref(e, ref_0);
for (int i = 0; i < ne_ref_e; ++i, ++e_ref)
const int ne_ref = get_ne_ref(e, ref_0);
for (int i = 0; i < ne_ref; ++i, ++e_ref)
{
cell_types[e_ref] = static_cast<unsigned char>(
vtk_geom_map[mesh.GetElementGeometry(e)]);
cell_types[e_ref] = vtk_geom_map[mesh.GetElementGeometry(e)];
}
}
AppendParData(vtk, "Types", ne, e_offset, Dims({ne_total}),
@@ -734,11 +732,11 @@ void VTKHDF::SaveMesh(const Mesh &mesh, bool high_order, int ref)
EnsureGroup("CellData", cell_data);
std::vector<int> attributes(ne);
hsize_t e_ref = 0;
for (int e = 0; e < ne_0; ++e)
for (hsize_t e = 0; e < ne_0; ++e)
{
const int attr = mesh.GetAttribute(e);
const int ne_ref_e = get_ne_ref(e, ref_0);
for (int i = 0; i < ne_ref_e; ++i, ++e_ref)
const int ne_ref = get_ne_ref(e, ref_0);
for (int i = 0; i < ne_ref; ++i, ++e_ref)
{
attributes[e_ref] = attr;
}
@@ -774,7 +772,7 @@ void VTKHDF::SaveGridFunction(const GridFunction &gf, const std::string &name)
{
for (int vd = 0; vd < vdim; ++vd)
{
point_values[off] = FP_T(vec_val(vd, i));
point_values[off] = vec_val(vd, i);
++off;
}
}
+7 -9
View File
@@ -76,14 +76,14 @@ private:
/// Wrapper for storing dataset dimensions (max ndims is 2D in VTKHDF).
struct Dims
{
static constexpr size_t MAX_NDIMS = 2;
static constexpr int MAX_NDIMS = 2;
std::array<hsize_t, MAX_NDIMS> data = { }; // Zero initialized
int ndims = 0;
Dims() = default;
Dims(int ndims_) : ndims(ndims_) { MFEM_ASSERT(ndims <= MAX_NDIMS, ""); }
Dims(int ndims_, hsize_t val) : Dims(ndims_) { data.fill(val); }
template <typename T>
Dims(std::initializer_list<T> data_) : Dims(int(data_.size()))
Dims(std::initializer_list<T> data_) : Dims(data_.size())
{ std::copy(data_.begin(), data_.end(), data.begin()); }
operator hsize_t*() { return data.data(); }
hsize_t &operator[](int i) { return data[i]; }
@@ -97,7 +97,7 @@ private:
hid_t steps = H5I_INVALID_HID;
/// Number of time steps saved.
unsigned long nsteps = 0;
int nsteps = 0;
/// Keep track of the offsets into the data arrays at each time step.
struct Offsets
@@ -123,8 +123,8 @@ private:
class MeshId
{
const Mesh *mesh_ptr = nullptr;
long sequence = -1;
long nodes_sequence = -1;
int sequence = -1;
int nodes_sequence = -1;
bool high_order = true;
int ref = -1;
public:
@@ -187,10 +187,8 @@ private:
/// The rank (number of dimensions) of the dataset is given by @a ndims and
/// its data type is given by @a type.
///
/// If the dataset does not exist, it will initially have size @a dims.
/// Otherwise, it will be resized to append data of size @a dims, and @a dims
/// will be set to the new total size.
hid_t EnsureDataset(hid_t f, const std::string &name, hid_t type, Dims &dims);
/// The dataset will initially have zero size and unlimited maximum size.
hid_t EnsureDataset(hid_t f, const std::string &name, hid_t type, int ndims);
/// @brief Ensure the named group is open, creating it if needed. Set @a
/// group to the ID.
-1
View File
@@ -224,6 +224,5 @@ int main (int argc, char *argv[])
}
delete metric;
delete fec_mesh;
return 0;
}
+15 -1
View File
@@ -52,6 +52,20 @@ if (MFEM_USE_MPI)
${NAVIER_COMMON_FILES}
LIBRARIES mfem)
add_mfem_miniapp(incompressible_navier_dfem
MAIN incompressible_navier_dfem.cpp
EXTRA_HEADERS incompressible_navier_nvtx.hpp
LIBRARIES mfem)
add_mfem_miniapp(incompNS_2Dtest
MAIN incompNS_2Dtest.cpp
EXTRA_SOURCES incompressible_navier_solver.cpp
incompressible_navier_tests.cpp
EXTRA_HEADERS incompressible_navier_solver.hpp
incompressible_navier_nvtx.hpp
EXTRA_DEFINES MFEM_USE_CMAKE_TESTS
LIBRARIES mfem)
add_mfem_miniapp(navier_turbchan
MAIN navier_turbchan.cpp
${NAVIER_COMMON_FILES}
@@ -84,4 +98,4 @@ if (MFEM_USE_MPI)
${MPIEXEC_POSTFLAGS})
endforeach()
endif()
endif ()
endif()
+205
View File
@@ -0,0 +1,205 @@
// Copyright (c) 2010-2024, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
// 3D flow over a cylinder benchmark example
#include "incompressible_navier_solver.hpp"
#define NVTX_COLOR ::gpu::nvtx::kLawnGreen
#include "incompressible_navier_nvtx.hpp"
using namespace mfem;
using namespace incompressible_navier;
void vel(const Vector &x, real_t t, Vector &u)
{
// real_t xi = x(0), yi = x(1);
u = 0.0;
}
void vel_inlet(const Vector &x, real_t t, Vector &u)
{
u = 0.0;
if (x(0) < 0.001) { u(0) = -0.001 * (std::pow(x(1) - 0.5, 2.0) - 0.25); }
}
MFEM_EXPORT int navier(int argc, char *argv[], double &u, double &p, double &Ψ)
{
dbg();
static mfem::MPI_Session mpi(argc, argv);
const int myid = mpi.WorldRank();
Hypre::Init();
const char *device_config = "cpu";
int serial_refinements = 1;
int nx = 90, ny = 30;
int v_order = 2;
int p_order = 1;
int t_order = 1;
real_t kin_vis = 20.0;
real_t dt = 1e-2;
real_t t = 0.0;
real_t t_final = 1.0;
bool last_step = false;
bool visualization = true;
bool use_paraview = false;
bool pa = false;
int vis_steps = 100;
int max_tsteps = -1;
constexpr int precision = 8;
std::cout.precision(precision);
OptionsParser args(argc, argv);
args.AddOption(&serial_refinements, "-sr", "--serial-refinements",
"Number serial refinements.");
args.AddOption(&nx, "-nx", "--nx", "Number of elements in X.");
args.AddOption(&ny, "-ny", "--ny", "Number of elements in Y.");
args.AddOption(&v_order, "-vo", "--vo", "Order.");
args.AddOption(&p_order, "-po", "--po", "Order.");
args.AddOption(&t_order, "-to", "--to", "Order.");
args.AddOption(&kin_vis, "-kv", "--kin-vis",
"Kineic viscosity coefficient.");
args.AddOption(&dt, "-dt", "--time-step", "Initial time step size.");
args.AddOption(&t_final, "-tf", "--t-final", "Final time; start time is 0.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly",
"Enable or disable partial assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.AddOption(&use_paraview, "-pv", "--paraview", "-no-pv", "--no-paraview",
"Use ParaView.");
args.AddOption(&vis_steps, "-vs", "--visualization-steps",
"Visualize every n-th timestep.");
args.AddOption(&max_tsteps, "-ms", "--max-steps",
"Maximum number of steps (negative means no restriction).");
args.Parse();
if (!args.Good())
{
if (myid == 0) { args.PrintUsage(mfem::out); }
return EXIT_FAILURE;
}
if (myid == 0) { args.PrintOptions(mfem::out); }
// Mesh *mesh = new Mesh("box-cylinder.mesh");
const real_t sx = 3.0, sy = 1.0;
const bool generate_edges = true;
const auto QUAD = Element::QUADRILATERAL;
Mesh mesh = Mesh::MakeCartesian2D(nx, ny, QUAD, generate_edges, sx, sy);
for (int i = 0; i < serial_refinements; ++i) { mesh.UniformRefinement(); }
if (Mpi::Root())
{
std::cout << "Number of elements: " << mesh.GetNE() << std::endl;
}
auto *pmesh = new ParMesh(MPI_COMM_WORLD, mesh);
// Create the flow solver.
IncompressibleNavierSolver flowsolver(pmesh, v_order, p_order, t_order,
kin_vis);
flowsolver.EnablePA(pa);
// // Set the initial condition.
// ParGridFunction *u_ic = flowsolver.GetCurrentVelocity();
// VectorFunctionCoefficient u_excoeff(pmesh->Dimension(), vel);
// u_ic->ProjectCoefficient(u_excoeff);
// Add Dirichlet boundary conditions to velocity space restricted to
// selected attributes on the mesh.
Array<int> attr(pmesh->bdr_attributes.Max());
attr = 0;
Array<int> attr_inlet(pmesh->bdr_attributes.Max());
attr_inlet = 0;
// Inlet is attribute 1.
attr[0] = 1;
// Walls is attribute 3.
attr[2] = 1;
flowsolver.AddVelDirichletBC(vel, attr);
attr_inlet[3] = 1;
flowsolver.AddVelDirichletBC(vel_inlet, attr_inlet);
flowsolver.Setup(dt);
ParGridFunction *u_gf = flowsolver.GetCurrentVelocity();
ParGridFunction *p_gf = flowsolver.GetCurrentPressure();
ParGridFunction *psi_gf = flowsolver.GetCurrentPsi();
ParaViewDataCollection pvdc("3dfoc", pmesh);
if (use_paraview)
{
pvdc.SetDataFormat(VTKFormat::BINARY32);
// pvdc.SetHighOrderOutput(true);
pvdc.SetCycle(0);
pvdc.SetTime(t);
pvdc.RegisterField("velocity", u_gf);
pvdc.RegisterField("pressure", p_gf);
pvdc.RegisterField("psi", psi_gf);
pvdc.Save();
}
for (int step = 0; !last_step; ++step)
{
if (step == max_tsteps) { last_step = true; }
if (t + dt >= t_final - dt / 2) { last_step = true; }
const bool vis_step = last_step || (step % vis_steps) == 0;
flowsolver.Step(t, dt, step, vis_step);
if (vis_step)
{
if (Mpi::Root() && vis_steps)
{
printf("%11s %11s\n", "Time", "dt");
printf("%.5E %.5E\n", t, dt);
}
if (use_paraview)
{
pvdc.SetCycle(step);
pvdc.SetTime(t);
pvdc.Save();
}
}
}
// flowsolver.PrintTimingData();
auto reduce = [](ParGridFunction *gf) -> real_t { return (*gf) * (*gf); };
// auto reduce = [](ParGridFunction *gf) -> real_t { return gf->Norml2(); };
u = reduce(u_gf), p = reduce(p_gf), Ψ = reduce(psi_gf);
fflush(stdout);
delete pmesh;
return EXIT_SUCCESS;
}
///////////////////////////////////////////////////////////////////////////////
#ifndef MFEM_USE_CMAKE_TESTS
int main(int argc, char *argv[])
try
{
dbg();
double u, p, Ψ; // unused
return navier(argc, argv, u, p, Ψ);
}
catch (std::exception &e)
{
std::cerr << "\033[31m..xxxXXX[ERROR]XXXxxx.." << std::endl;
std::cerr << "\033[31m{}" << e.what() << std::endl;
return EXIT_FAILURE;
}
#endif // MFEM_USE_CMAKE_TESTS
@@ -0,0 +1,397 @@
// Copyright (c) 2010-2024, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "mfem.hpp"
using namespace mfem;
using namespace mfem::future;
#include "linalg/tensor.hpp"
using mfem::future::tensor;
#define NVTX_COLOR ::gpu::nvtx::kOrchid
#include "incompressible_navier_nvtx.hpp"
///////////////////////////////////////////////////////////////////////////////
template <int DIM>
void DiffusionSetup(const ParFiniteElementSpace &sfes,
const IntegrationRule &ir,
const Array<int> &domain_attributes,
ParameterFunction &qdata)
{
NVTX_MARK_FUNCTION;
auto pmesh = sfes.GetParMesh();
auto nodes = static_cast<ParGridFunction *>(pmesh->GetNodes());
auto mfes = nodes->ParFESpace();
constexpr int U = 0, Ξ = 1, Δ = 2;
DifferentiableOperator dop(
{{ U, &sfes }},
{
{ { Ξ, mfes },
{ Δ, &qdata.GetParameterSpace() }
}
},
*pmesh);
const auto qfunc =
[] MFEM_HOST_DEVICE(const tensor<real_t, DIM, DIM> &J,
const real_t &w)
{
auto invJ = inv(J);
tensor<real_t, DIM, DIM> C{};
C(0, 0) = M_PI;
C(0, 1) = 0, C(1, 0) = 0;
assert(C(0, 1) == 0 && C(1, 0) == 0); // diff otherwise
C(1, 1) = 1.0 / M_PI;
return tuple{ C * invJ * transpose(invJ) * det(J) * w };
};
dop.AddDomainIntegrator(qfunc,
tuple{ Gradient<Ξ>{}, Weight{} }, // inputs
tuple{ Identity<Δ>{} }, // outputs
ir, domain_attributes);
dop.SetParameters({ nodes, &qdata });
Vector unused(sfes.GetTrueVSize());
dop.Mult(unused, qdata);
qdata.HostRead();
}
///////////////////////////////////////////////////////////////////////////////
template <int DIM>
void DiffusionApply(const ParFiniteElementSpace &sfes,
const IntegrationRule &ir,
const Array<int> &domain_attributes,
ParameterFunction &qdata,
const Vector &x, Vector &y)
{
NVTX_MARK_FUNCTION;
auto pmesh = sfes.GetParMesh();
constexpr int U = 0, Q = 1;
auto qd_ps = &qdata.GetParameterSpace();
DifferentiableOperator dop({ { U, &sfes } }, { { Q, qd_ps } }, *pmesh);
const auto qfunc =[] MFEM_HOST_DEVICE(const tensor<real_t, DIM> &u,
const tensor<real_t, DIM, DIM> &Q)
{
return tuple{ Q * u };
};
dop.AddDomainIntegrator(
qfunc,
tuple{ Gradient<U>{}, Identity<Q>{} },
tuple{ Gradient<U>{} },
ir, domain_attributes);
dop.SetParameters({ &qdata });
dop.Mult(x, y);
y.HostRead();
}
///////////////////////////////////////////////////////////////////////////////
template <int DIM>
int DiffVerification(ParFiniteElementSpace &h1fes,
const IntegrationRule &ir,
const Vector &qdata,
const Vector &x, const Vector &y)
{
NVTX_MARK_FUNCTION;
constexpr real_t ϵ = 1e-12;
MatrixFunctionCoefficient matrix_coeff(
DIM, [](const Vector &, DenseMatrix &C)
{
C.SetSize(DIM);
C(0, 0) = M_PI;
C(0, 1) = 0, C(1, 0) = 0;
assert(C(0, 1) == 0 && C(1, 0) == 0); // diff otherwise
C(1, 1) = 1.0 / M_PI;
});
ParBilinearForm a(&h1fes);
auto diff_integ = new DiffusionIntegrator(matrix_coeff);
diff_integ->SetIntRule(&ir);
a.AddDomainIntegrator(diff_integ);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
OperatorPtr A;
a.Assemble(), a.Finalize();
a.FormSystemMatrix(Array<int> {}, A);
Vector y2(h1fes.TrueVSize());
y2 = 0.0;
A->Mult(x, y2);
y2.HostRead();
Vector diff(y2);
diff -= y;
const auto diff_norm = diff.Norml2();
if (diff_norm > ϵ)
{
dbg("\x1B[31m||dFdu_FD u^* - ex||_l2 = {}", diff_norm);
return EXIT_FAILURE;
}
dbg("\x1B[32m||dFdu_FD u^* - ex||_l2 = {}", diff_norm);
return EXIT_SUCCESS;
}
///////////////////////////////////////////////////////////////////////////////
template <int DIM>
void MassApply(const ParFiniteElementSpace &sfes,
const IntegrationRule &ir,
const Array<int> &domain_attributes,
const Vector &x, Vector &y)
{
dbg();
auto &pmesh = *sfes.GetParMesh();
auto *nodes = static_cast<ParGridFunction *>(pmesh.GetNodes());
auto *mfes = nodes->ParFESpace();
constexpr int U = 0, Coords = 1;
DifferentiableOperator dop({{ U, &sfes }}, {{ Coords, mfes }}, pmesh);
const auto mf_mass_qf =
[](const real_t &dudxi,
const tensor<real_t, DIM, DIM> &J,
const real_t &w)
{
return tuple{ dudxi * w * det(J) };
};
dop.AddDomainIntegrator(
mf_mass_qf,
tuple{ Value<U>{}, Gradient<Coords>{}, Weight{} },
tuple{ Value<U>{} },
ir, domain_attributes);
dop.SetParameters({ nodes });
// Vector X(sfes.GetTrueVSize()), Y(sfes.GetTrueVSize());
// sfes.GetRestrictionMatrix()->Mult(x, X);
// dop.Mult(X, Y);
dop.Mult(x, y);
}
///////////////////////////////////////////////////////////////////////////////
template <int DIM>
int MassVerification(ParFiniteElementSpace &h1fes,
const IntegrationRule &ir,
const Vector &x, Vector &y)
{
ParBilinearForm a(&h1fes);
auto mass_integ = new MassIntegrator;
mass_integ->SetIntRule(&ir);
a.AddDomainIntegrator(mass_integ);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.Assemble(), a.Finalize();
Vector y2(h1fes.TrueVSize());
a.Mult(x, y2);
y2.HostRead();
Vector diff(y2);
diff -= y;
const auto diff_norm = diff.Norml2();
constexpr real_t ϵ = 1e-12;
if (diff_norm > ϵ)
{
dbg("\x1B[31m||dFdu_FD u^* - ex||_l2 = {}", diff_norm);
return EXIT_FAILURE;
}
dbg("\x1B[32m||dFdu_FD u^* - ex||_l2 = {}", diff_norm);
return EXIT_SUCCESS;
}
///////////////////////////////////////////////////////////////////////////////
template <int DIM>
void VectorDiffApply(const ParFiniteElementSpace &vfes,
const IntegrationRule &ir,
const Array<int> &domain_attributes,
const Vector &x,
Vector &y)
{
NVTX_MARK_FUNCTION;
auto pmesh = vfes.GetParMesh();
auto nodes = static_cast<ParGridFunction *>(pmesh->GetNodes());
auto mfes = nodes->ParFESpace();
constexpr int U = 0, Coords = 1;
DifferentiableOperator dop({{ U, &vfes }}, {{ Coords, mfes }}, *pmesh);
const auto qfunc =
[] MFEM_HOST_DEVICE(const tensor<real_t, DIM, DIM> &u,
const tensor<real_t, DIM, DIM> &J, const real_t &w)
{
return tuple{ u * inv(J) * det(J) * w * transpose(inv(J)) };
};
dop.AddDomainIntegrator(qfunc,
tuple{ Gradient<U>{}, Gradient<Coords>{}, Weight{} },
tuple{ Gradient<U>{} },
ir, domain_attributes);
dop.SetParameters({ nodes });
dop.Mult(x, y);
y.HostRead();
}
///////////////////////////////////////////////////////////////////////////////
template <int DIM, int VDIM = DIM>
int VectorDiffVerif(ParFiniteElementSpace &h1fes,
const IntegrationRule &ir, const Vector &x,
Vector &y)
{
NVTX_MARK_FUNCTION;
ParBilinearForm a(&h1fes);
auto A_integ = new VectorDiffusionIntegrator(VDIM);
A_integ->SetIntRule(&ir);
a.AddDomainIntegrator(A_integ);
a.SetAssemblyLevel(AssemblyLevel::PARTIAL);
a.Assemble(), a.Finalize();
Vector y2(h1fes.TrueVSize());
a.Mult(x, y2);
y2.HostRead();
Vector diff(y2);
diff -= y;
const auto diff_norm = diff.Norml2();
constexpr real_t ϵ = 1e-12;
if (diff_norm > ϵ)
{
dbg("\x1B[31m||dFdu_FD u^* - ex||_l2 = {}", diff_norm);
return EXIT_FAILURE;
}
dbg("\x1B[32m||dFdu_FD u^* - ex||_l2 = {}", diff_norm);
return EXIT_SUCCESS;
}
///////////////////////////////////////////////////////////////////////////////
int main(int argc, char *argv[]) try
{
NVTX_MARK_FUNCTION;
constexpr int DIM = 2, VDIM = DIM;
static mfem::MPI_Session mpi(argc, argv);
const int myid = mpi.WorldRank();
Hypre::Init();
const char *device_config = "cpu";
const char *mesh_file = "none";
int serial_refinements = 0;
int nx = 1, ny = 1;
int p = 1;
bool visualization = false;
bool pa = false;
std::cout.precision(8);
OptionsParser args(argc, argv);
args.AddOption(&mesh_file, "-m", "--mesh", "Mesh file to use.");
args.AddOption(&p, "-o", "--order", "Finite element order.");
args.AddOption(&serial_refinements, "-sr", "--serial-refinements",
"Number serial refinements.");
args.AddOption(&nx, "-nx", "--nx", "Number of elements in X.");
args.AddOption(&ny, "-ny", "--ny", "Number of elements in Y.");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&pa, "-pa", "--partial-assembly", "-no-pa",
"--no-partial-assembly",
"Enable or disable partial assembly.");
args.AddOption(&device_config, "-d", "--device",
"Device configuration string, see Device::Configure().");
args.Parse();
if (!args.Good())
{
if (myid == 0) { args.PrintUsage(mfem::out); }
return EXIT_FAILURE;
}
if (myid == 0) { args.PrintOptions(mfem::out); }
Mesh smesh;
if (std::string(mesh_file) != "none")
{
smesh = Mesh(mesh_file);
}
else
{
const real_t sx = 3.0, sy = 1.0;
const bool generate_edges = true;
const auto QUAD = Element::QUADRILATERAL;
smesh = Mesh::MakeCartesian2D(nx, ny, QUAD, generate_edges, sx, sy);
}
MFEM_ASSERT(smesh.Dimension() == 2, "2D mesh required!");
for (int i = 0; i < serial_refinements; ++i) { smesh.UniformRefinement(); }
dbg("Number of elements: {}", smesh.GetNE());
ParMesh pmesh(MPI_COMM_WORLD, smesh);
smesh.Clear();
pmesh.EnsureNodes();
pmesh.SetCurvature(p);
assert(DIM == pmesh.Dimension());
Array<int> domain_attributes;
if (pmesh.attributes.Size() > 0)
{
domain_attributes.SetSize(pmesh.attributes.Max());
domain_attributes = 1;
}
H1_FECollection fec(p, DIM);
ParFiniteElementSpace fes(&pmesh, &fec), vfes(&pmesh, &fec, VDIM);
dbg("#dofs:{} ", fes.GetTrueVSize());
const auto &fe = *fes.GetFE(0);
const auto &ir =
IntRules.Get(fe.GetGeomType(), fe.GetOrder() + fe.GetOrder() + fe.GetDim() - 1);
dbg("#ndof per el = {}", fe.GetDof());
dbg("#nqp = {}", ir.GetNPoints());
dbg("#q1d = {}", (int)floor(pow(ir.GetNPoints(), 1.0 / DIM) + 0.5));
ParGridFunction f1_gf(&fes);
auto f1 = [](const Vector &coords)
{
assert(DIM == 2);
const double x = coords(0), y = coords(1);
return M_PI + x + x * x + x * y + y;
};
FunctionCoefficient f1_c(f1);
f1_gf.ProjectCoefficient(f1_c);
Vector x(f1_gf), y(fes.GetTrueVSize());
UniformParameterSpace qd_ps(pmesh, ir, DIM * DIM);
ParameterFunction qdata(qd_ps);
dbg("Diffusion setup, apply & verification");
DiffusionSetup<DIM>(fes, ir, domain_attributes, qdata);
DiffusionApply<DIM>(fes, ir, domain_attributes, qdata, x, y);
if (DiffVerification<DIM>(fes, ir, qdata, x, y) != EXIT_SUCCESS) { return EXIT_FAILURE; }
dbg("Mass apply & verification");
MassApply<DIM>(fes, ir, domain_attributes, x, y);
if (MassVerification<DIM>(fes, ir, x, y) != EXIT_SUCCESS) { return EXIT_FAILURE; }
dbg("Vector diffusion apply");
VectorFunctionCoefficient vf1_c(VDIM, [](const Vector &coords, Vector &u)
{
assert(DIM == 2);
const double x = coords(0), y = coords(1);
u(0) = M_PI + 0.25 * x * x * y + y * y * x;
u(1) = M_PI - 0.25 * x * y * y + y * x * x;
});
ParGridFunction vf1_gf(&vfes);
vf1_gf.ProjectCoefficient(vf1_c);
Vector vx(vf1_gf), vy(vfes.GetTrueVSize());
VectorDiffApply<DIM>(vfes, ir, domain_attributes, vx, vy);
if (VectorDiffVerif<DIM>(vfes, ir, vx, vy) != EXIT_SUCCESS) { return EXIT_FAILURE; }
return EXIT_SUCCESS;
}
catch (std::exception &e)
{
std::cerr << "\033[31m..xxxXXX[ERROR]XXXxxx.." << std::endl;
std::cerr << "\033[31m{}" << e.what() << std::endl;
return EXIT_FAILURE;
}
@@ -0,0 +1,478 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#define FMT_HEADER_ONLY
#include <fmt/format.h>
#include <array>
#include <cassert>
#include <cstddef>
#include <cstdint>
#include <iomanip>
#include <iostream>
#include <memory>
#include <mutex>
#include <stack>
#include <string>
#ifdef MFEM_USE_CALIPER
#include <caliper/cali.h>
#endif
#ifdef MFEM_USE_CUDA
#include <cudaProfiler.h>
#include <cuda_runtime_api.h>
#include <nvToolsExt.h>
#else
struct nvtxEventAttributes_t
{
int version;
int size;
int category;
int colorType;
uint32_t color;
int payloadType;
uint64_t payload;
int messageType;
struct
{
std::string ascii;
} message;
};
#define NVTX_VERSION 1
#define NVTX_EVENT_ATTRIB_STRUCT_SIZE 256
#define NVTX_COLOR_ARGB 0
#define NVTX_MESSAGE_TYPE_ASCII 0
#define nvtxRangePushEx(...)
#define nvtxRangePop(...)
#define cudaStreamSynchronize(...)
#endif
namespace gpu::nvtx
{
///////////////////////////////////////////////////////////////////////////////
// https://en.wikipedia.org/wiki/Web_colors#Extended_colors
// http://www.calmar.ws/vim/256-xterm-24bit-rgb-color-chart.html
// clang-format off
enum color_names
{
kBlack = 0, kNavyBlue, kDarkBlue, kMediumBlue, kBlue, kDarkGreen, kWebGreen, kTeal,
kDarkCyan, kDeepSkyBlue, kDarkTurquoise, kMediumSpringGreen, kGreen, kLime,
kSpringGreen, kAqua, kCyan, kMidnightBlue, kDodgerBlue, kLightSeaGreen, kForestGreen,
kSeaGreen, kDarkSlateGray, kLimeGreen, kMediumSeaGreen, kTurquoise, kRoyalBlue,
kSteelBlue, kDarkSlateBlue, kMediumTurquoise, kIndigo, kDarkOliveGreen, kCadetBlue,
kCornflower, kRebeccaPurple, kMediumAquamarine, kDimGray, kSlateBlue, kOliveDrab,
kSlateGray, kLightSlateGray, kMediumSlateBlue, kLawnGreen, kWebMaroon, kWebPurple,
kChartreuse, kAquamarine, kOlive, kWebGray, kSkyBlue, kLightSkyBlue, kBlueViolet,
kDarkRed, kDarkMagenta, kSaddleBrown, kDarkSeaGreen, kLightGreen, kMediumPurple,
kDarkViolet, kPaleGreen, kDarkOrchid, kYellowGreen, kPurple, kSienna, kBrown,
kDarkGray, kLightBlue, kGreenYellow, kPaleTurquoise, kMaroon, kLightSteelBlue,
kPowderBlue, kFirebrick, kDarkGoldenrod, kMediumOrchid, kRosyBrown, kDarkKhaki,
kGray, kSilver, kMediumVioletRed, kIndianRed, kPeru, kChocolate, kTan, kLightGray,
kThistle, kOrchid, kGoldenrod, kPaleVioletRed, kCrimson, kGainsboro, kPlum, kBurlywood,
kLightCyan, kLavender, kDarkSalmon, kViolet, kPaleGoldenrod, kLightCoral, kKhaki,
kAliceBlue, kHoneydew, kAzure, kSandyBrown, kWheat, kBeige, kWhiteSmoke, kMintCream,
kGhostWhite, kSalmon, kAntiqueWhite, kLinen, kLightGoldenrod, kOldLace, kRed,
kFuchsia, kMagenta, kDeepPink, kOrangeRed, kTomato, kHotPink, kCoral, kDarkOrange,
kLightSalmon, kOrange, kLightPink, kPink, kGold, kPeachPuff, kNavajoWhite, kMoccasin,
kBisque, kMistyRose, kBlanchedAlmond, kPapayaWhip, kLavenderBlush, kSeashell,
kCornsilk, kLemonChiffon, kFloralWhite, kSnow, kYellow, kLightYellow, kIvory, kWhite,
kNvidia
};
// clang-format on
static constexpr int kNumHexColors = 146;
static constexpr std::array<uint32_t, kNumHexColors> kHexColors =
{
{
0x000000, 0x000080, 0x00008B, 0x0000CD, 0x0000FF, 0x006400, 0x008000,
0x008080, 0x008B8B, 0x00BFFF, 0x00CED1, 0x00FA9A, 0x00FF00, 0x00FF00,
0x00FF7F, 0x00FFFF, 0x00FFFF, 0x191970, 0x1E90FF, 0x20B2AA, 0x228B22,
0x2E8B57, 0x2F4F4F, 0x32CD32, 0x3CB371, 0x40E0D0, 0x4169E1, 0x4682B4,
0x483D8B, 0x48D1CC, 0x4B0082, 0x556B2F, 0x5F9EA0, 0x6495ED, 0x663399,
0x66CDAA, 0x696969, 0x6A5ACD, 0x6B8E23, 0x708090, 0x778899, 0x7B68EE,
0x7CFC00, 0x7F0000, 0x7F007F, 0x7FFF00, 0x7FFFD4, 0x808000, 0x808080,
0x87CEEB, 0x87CEFA, 0x8A2BE2, 0x8B0000, 0x8B008B, 0x8B4513, 0x8FBC8F,
0x90EE90, 0x9370DB, 0x9400D3, 0x98FB98, 0x9932CC, 0x9ACD32, 0xA020F0,
0xA0522D, 0xA52A2A, 0xA9A9A9, 0xADD8E6, 0xADFF2F, 0xAFEEEE, 0xB03060,
0xB0C4DE, 0xB0E0E6, 0xB22222, 0xB8860B, 0xBA55D3, 0xBC8F8F, 0xBDB76B,
0xBEBEBE, 0xC0C0C0, 0xC71585, 0xCD5C5C, 0xCD853F, 0xD2691E, 0xD2B48C,
0xD3D3D3, 0xD8BFD8, 0xDA70D6, 0xDAA520, 0xDB7093, 0xDC143C, 0xDCDCDC,
0xDDA0DD, 0xDEB887, 0xE0FFFF, 0xE6E6FA, 0xE9967A, 0xEE82EE, 0xEEE8AA,
0xF08080, 0xF0E68C, 0xF0F8FF, 0xF0FFF0, 0xF0FFFF, 0xF4A460, 0xF5DEB3,
0xF5F5DC, 0xF5F5F5, 0xF5FFFA, 0xF8F8FF, 0xFA8072, 0xFAEBD7, 0xFAF0E6,
0xFAFAD2, 0xFDF5E6, 0xFF0000, 0xFF00FF, 0xFF00FF, 0xFF1493, 0xFF4500,
0xFF6347, 0xFF69B4, 0xFF7F50, 0xFF8C00, 0xFFA07A, 0xFFA500, 0xFFB6C1,
0xFFC0CB, 0xFFD700, 0xFFDAB9, 0xFFDEAD, 0xFFE4B5, 0xFFE4C4, 0xFFE4E1,
0xFFEBCD, 0xFFEFD5, 0xFFF0F5, 0xFFF5EE, 0xFFF8DC, 0xFFFACD, 0xFFFAF0,
0xFFFAFA, 0xFFFF00, 0xFFFFE0, 0xFFFFF0, 0xFFFFFF, 0x76B900
}
};
///////////////////////////////////////////////////////////////////////////////
constexpr size_t static_strlen(const char *str)
{
return *str == '\0' ? 0 : static_strlen(str + 1) + 1;
}
constexpr uint8_t static_checksum8(const char *bfr)
{
unsigned int chk = 0;
size_t len = static_strlen(bfr);
for (; len; len--, bfr++) { chk += static_cast<unsigned int>(*bfr); }
return static_cast<uint8_t>(chk);
}
constexpr char *static_strrnchr(const char *str, const char c, int n)
{
size_t len = static_strlen(str);
char *p = const_cast<char *>(str) + len - 1;
for (; n; n--, p--, len--)
{
for (; len; p--, len--)
{
if (*p == c) { break; }
}
if (!len) { return nullptr; }
if (n == 1) { return p; }
}
return nullptr;
}
inline uint32_t static_color(const uint8_t COLOR, const int RANK,
const char *FILE)
{
constexpr auto kMpiColorShift = 1;
const auto rank_shift = kMpiColorShift * RANK;
if (COLOR > 0) { return kHexColors[COLOR + rank_shift]; }
const auto file_color = static_checksum8(FILE);
return kHexColors[(file_color + rank_shift) % kNumHexColors];
}
///////////////////////////////////////////////////////////////////////////////
// Helpers to generate unique variable names
#define NVTX_FLF __FILE__, __LINE__, __FUNCTION__
#define NVTX_PRIVATE_NAME(prefix) NVTX_PRIVATE_CONCAT(prefix, __LINE__)
#define NVTX_PRIVATE_CONCAT(a, b) NVTX_PRIVATE_CONCAT2(a, b)
#define NVTX_PRIVATE_CONCAT2(a, b) a##b
#ifndef NVTX_COLOR
#define NVTX_COLOR ::gpu::nvtx::kBlack
#endif
///////////////////////////////////////////////////////////////////////////////
struct Debug
{
const bool debug = false, end = true;
inline Debug() = default;
inline Debug(const int RANK, const char *FILE, const int LINE,
const char *FUNC, uint8_t COLOR, bool ini = true,
bool END = true): debug(true), end(END)
{
const char *base = static_strrnchr(FILE, '/', 2);
const char *file = base ? base + 1 : FILE;
const uint32_t rgb = static_color(COLOR, RANK, FILE);
const uint8_t r = (rgb >> 16) & 0xFF, g = (rgb >> 8) & 0xFF,
b = rgb & 0xFF;
std::cout << "\033[38;2;";
std::cout << std::to_string(r) << ";";
std::cout << std::to_string(g) << ";";
std::cout << std::to_string(b) << "m";
if (ini)
{
std::cout << RANK << std::setw(64) << file << ":";
std::cout << "\033[2m" << std::setw(4) << std::left << LINE
<< "\033[22m: ";
if (FUNC) { std::cout << "[" << FUNC << "] "; }
}
std::cout << std::right << "\033[1m";
}
inline ~Debug()
{
if (debug) { std::cout << "\033[m" << (end ? "\n" : "") << std::flush; }
}
template <typename T>
inline void operator<<(const T &arg) const noexcept
{
if (debug) { std::cout << arg; }
}
template <typename T>
inline void operator()(const T &arg) const noexcept
{
if (debug) { this->operator<<(arg); }
}
template <typename... Args>
inline void operator()(const char *fmt, Args &&...args) const noexcept
{
if (debug) { std::cout << fmt::format(fmt, std::forward<Args>(args)...); }
}
inline void operator()() const noexcept {}
static Debug Set(const char *FILE, const int LINE, const char *FUNC,
uint8_t COLOR, bool INI = true, bool END = true)
{
static int mpi_rank = 0, dbg_mpi_rank = 0;
static bool env_mpi = false, env_dbg = false;
if (static bool ini = false; !std::exchange(ini, true))
{
env_dbg = (getenv("MFEM_DEBUG") != nullptr);
env_mpi = getenv("MFEM_DEBUG_MPI") != nullptr;
// int mpi_flag = 0;
// MPI_Initialized(&mpi_flag);
// if (mpi_flag) { MPI_Comm_rank(MPI_COMM_WORLD, &mpi_rank); }
dbg_mpi_rank = atoi(env_mpi ? getenv("MFEM_DEBUG_MPI") : "0");
}
const bool debug = (env_dbg && (!env_mpi || (dbg_mpi_rank == mpi_rank)));
return debug ? Debug(mpi_rank, FILE, LINE, FUNC, COLOR, INI, END)
: Debug();
}
};
// Debug console traces, unnamed
#define NVTX_DEBUG(...) \
::gpu::nvtx::Debug::Set(NVTX_FLF, NVTX_COLOR).operator()(__VA_ARGS__)
#define NVTX_DEBUG_NO_INI(...) \
::gpu::nvtx::Debug::Set(NVTX_FLF, NVTX_COLOR, false, true) \
.operator()(__VA_ARGS__)
#define NVTX_DEBUG_APPEND(...) \
::gpu::nvtx::Debug::Set(NVTX_FLF, NVTX_COLOR, false, false) \
.operator()(__VA_ARGS__)
#define NVTX_DEBUG_NO_END(...) \
::gpu::nvtx::Debug::Set(NVTX_FLF, NVTX_COLOR, true, false) \
.operator()(__VA_ARGS__)
///////////////////////////////////////////////////////////////////////////////
struct Nvtx
{
const bool nvtx = false, enforce_kernel_sync = false;
const char *base, *file;
const uint32_t color = kBlack;
mutable std::string ascii;
mutable nvtxEventAttributes_t event;
mutable bool pushed = false;
inline Nvtx() = default;
Nvtx(bool enforce_kernel_sync, const char *FILE, const int LINE,
const char *FUNC, uint8_t COLOR):
nvtx(true), enforce_kernel_sync(enforce_kernel_sync),
base(static_strrnchr(FILE, '/', 2)), file(base ? base + 1 : FILE),
color(COLOR), ascii(file), event({})
{
event.version = NVTX_VERSION;
event.size = NVTX_EVENT_ATTRIB_STRUCT_SIZE;
event.colorType = NVTX_COLOR_ARGB;
event.color = static_color(COLOR, 0, FILE);
event.messageType = NVTX_MESSAGE_TYPE_ASCII;
ascii += ":";
ascii += std::to_string(LINE);
ascii += ":[";
ascii += FUNC;
ascii += "] ";
pushed = false;
}
explicit Nvtx(const char *title, uint8_t color = kWheat,
bool enforce_kernel_sync = true):
nvtx(true), enforce_kernel_sync(enforce_kernel_sync), color(color),
ascii(title), event({})
{
event.version = NVTX_VERSION;
event.size = NVTX_EVENT_ATTRIB_STRUCT_SIZE;
event.colorType = NVTX_COLOR_ARGB;
event.color = static_color(color, 0, "");
event.messageType = NVTX_MESSAGE_TYPE_ASCII;
event.message.ascii = ascii.c_str();
nvtxRangePushEx(&event);
pushed = true;
}
inline void operator()() const
{
if (!nvtx) { return; }
event.message.ascii = ascii.c_str();
assert(!pushed);
nvtxRangePushEx(&event);
pushed = true;
}
template <typename T>
inline void operator()(const T &arg) const
{
if (!nvtx) { return; }
this->operator<<(arg);
event.message.ascii = ascii.c_str();
assert(!pushed);
nvtxRangePushEx(&event);
pushed = true;
}
template <typename... Args>
inline void operator()(fmt::format_string<Args...> fmt, Args &&...args) const
{
if (!nvtx) { return; }
ascii += fmt::format(fmt, std::forward<Args>(args)...);
event.message.ascii = ascii.c_str();
assert(!pushed);
nvtxRangePushEx(&event);
pushed = true;
}
template <typename T>
inline void operator<<(const T &arg) const
{
if (nvtx) { ascii += arg; }
}
inline ~Nvtx()
{
if (!nvtx) { return; }
if (enforce_kernel_sync)
{
nvtxEventAttributes_t eks = {};
eks.version = NVTX_VERSION;
eks.size = NVTX_EVENT_ATTRIB_STRUCT_SIZE;
eks.category = 0; // user value
eks.colorType = NVTX_COLOR_ARGB;
eks.messageType = NVTX_MESSAGE_TYPE_ASCII;
eks.message.ascii = "!"; // enforce kernel synchronization
eks.color = kHexColors[kYellow];
nvtxRangePushEx(&eks);
cudaStreamSynchronize(nullptr);
nvtxRangePop(/*eks*/);
}
assert(pushed);
nvtxRangePop(/*event*/);
}
using nvtx_ptr = std::unique_ptr<Nvtx>;
using nvtx_stack_t = std::stack<nvtx_ptr>;
static nvtx_ptr Set(const char *FILE, const int LINE, const char *FUNC,
uint8_t COLOR)
{
static bool nvtx = false, eks = false;
if (static bool ini = false; !std::exchange(ini, true))
{
eks = getenv("MFEM_EKS") != nullptr;
nvtx = getenv("MFEM_NVTX") != nullptr;
Nvtx force_first_eks("Init EKS", kYellow, true);
}
return nvtx_ptr(nvtx ? new Nvtx(eks, FILE, LINE, FUNC, COLOR)
: new Nvtx());
}
static nvtx_stack_t &Stack()
{
auto nvtx_events = []() -> nvtx_stack_t &
{
static nvtx_stack_t events;
return events;
};
static std::once_flag ready;
// one touch to guarantee the object is ready
std::call_once(ready, [&] { nvtx_events(); });
return nvtx_events();
}
};
// Temporary object only alive for the current statement
#define NVTX_(COLOR, ...) \
NVTX_DEBUG(__VA_ARGS__); \
std::unique_ptr<::gpu::nvtx::Nvtx> NVTX_PRIVATE_NAME(nvtx) = \
::gpu::nvtx::Nvtx::Set(NVTX_FLF, COLOR); \
NVTX_PRIVATE_NAME(nvtx)->operator()(__VA_ARGS__)
// Temporary object only alive for the current statement
#define NVTX(...) NVTX_(NVTX_COLOR, __VA_ARGS__)
// Begin(with color)/End NVTX event traces
#define NVTX_BEGIN_(COLOR, ...) \
NVTX_DEBUG(__VA_ARGS__); \
::gpu::nvtx::Nvtx::Stack().push(::gpu::nvtx::Nvtx::Set(NVTX_FLF, COLOR)); \
::gpu::nvtx::Nvtx::Stack().top()->operator()(__VA_ARGS__)
// Begin/End NVTX event traces
#define NVTX_BEGIN(...) NVTX_BEGIN_(NVTX_COLOR, __VA_ARGS__);
#define NVTX_END(...) \
::gpu::nvtx::Nvtx::Stack().top().reset(); \
::gpu::nvtx::Nvtx::Stack().pop()
#ifdef USE_CALIPER
// CALIPER & NVTX marks
#define NVTX_MARK_FUNCTION \
NVTX(); \
std::unique_ptr<cali::Function> __cali_ann##__func__; \
__cali_ann##__func__ = std::make_unique<cali::Function>(__func__);
#define NVTX_MARK(...) \
NVTX(__VA_ARGS__); \
std::unique_ptr<cali::Function> __cali_ann##__func__; \
__cali_ann##__func__ = std::make_unique<cali::Function>(__VA_ARGS__);
#define NVTX_MARK_FUNCTION_NAME(STR_NAME) \
NVTX(STR_NAME); \
std::unique_ptr<cali::Function> __cali_ann##__func__; \
if (g_caliper) \
{ \
__cali_ann##__func__ = std::make_unique<cali::Function>(STR_NAME); \
}
#define NVTX_MARK_BEGIN(...) \
CALI_MARK_BEGIN(__VA_ARGS__); \
NVTX_BEGIN(__VA_ARGS__);
#define NVTX_MARK_END(...) \
NVTX_END(__VA_ARGS__); \
CALI_MARK_END(__VA_ARGS__);
#else
#define NVTX_MARK_FUNCTION NVTX()
#define NVTX_MARK(...) NVTX(__VA_ARGS__)
#define NVTX_MARK_FUNCTION_NAME(...) NVTX(__VA_ARGS__)
#define NVTX_MARK_BEGIN(...) NVTX_BEGIN(__VA_ARGS__)
#define NVTX_MARK_END(...) NVTX_END(__VA_ARGS__)
#endif
} // namespace gpu::nvtx
// Debug console traces, unnamed
#if 1
#define dbg(...) NVTX_DEBUG(__VA_ARGS__)
#define dbl(...) NVTX_DEBUG_NO_END(__VA_ARGS__)
#define dba(...) NVTX_DEBUG_APPEND(__VA_ARGS__)
#define dbc(...) NVTX_DEBUG_NO_INI(__VA_ARGS__)
#else
#define dbg(...)
#define dbl(...) (void)0
#define dba(...)
#define dbc(...)
#endif
inline bool ClearScreen()
{
dbg("\x1B[2J\x1B[3J\x1B[H");
return true;
}
@@ -0,0 +1,490 @@
// Copyright (c) 2010-2024, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "incompressible_navier_solver.hpp"
#define NVTX_COLOR ::gpu::nvtx::kCyan
#include "incompressible_navier_nvtx.hpp"
using namespace mfem;
using namespace incompressible_navier;
IncompressibleNavierSolver::IncompressibleNavierSolver(ParMesh *mesh,
int velorder, int porder,
int torder,
real_t kin_vis):
pmesh(mesh), velorder(velorder), porder(porder), torder(torder),
kin_vis(kin_vis), gll_rules(0, Quadrature1D::GaussLobatto),
vfec(new H1_FECollection(velorder, pmesh->Dimension())),
psifec(new H1_FECollection(porder)), pfec(new H1_FECollection(porder)),
vfes(new ParFiniteElementSpace(pmesh, vfec, pmesh->Dimension())),
psifes(new ParFiniteElementSpace(pmesh, pfec)),
pfes(new ParFiniteElementSpace(pmesh, pfec)),
velGF(torder + 1, nullptr), pGF(torder + 1, nullptr)
{
NVTX();
// Check if fully periodic mesh
if (!(pmesh->bdr_attributes.Size() == 0))
{
vel_ess_attr.SetSize(pmesh->bdr_attributes.Max());
vel_ess_attr = 0;
pres_ess_attr.SetSize(pmesh->bdr_attributes.Max());
pres_ess_attr = 0;
}
for (int i = 0; i < torder + 1; i++)
{
velGF[i] = new ParGridFunction(vfes);
*velGF[i] = 0.0;
pGF[i] = new ParGridFunction(pfes);
*pGF[i] = 0.0;
}
psiGF.SetSpace(psifes);
DvGF.SetSpace(vfes);
divVelGF.SetSpace(pfes);
pRHS.SetSpace(pfes);
}
void IncompressibleNavierSolver::Setup(real_t dt)
{
if (verbose && pmesh->GetMyRank() == 0)
{
mfem::out << "Setup" << std::endl;
if (partial_assembly)
{
mfem::out << "Using Partial Assembly" << std::endl;
}
else { mfem::out << "Using Full Assembly" << std::endl; }
}
this->Setup_velocity(dt);
this->Setup_auxiliary(dt);
this->Setup_pressure(dt);
}
void IncompressibleNavierSolver::Setup_velocity(real_t dt)
{
// GLL integration rule (Numerical Integration)
const IntegrationRule &ir_ni =
gll_rules.Get(vfes->GetFE(0)->GetGeomType(), 2 * velorder - 1);
vfes->GetEssentialTrueDofs(vel_ess_attr, vel_ess_tdof);
//-------------------------------------------------------------------------
// Setup of coefficient for mass term of Eq(13)
dtCoeff = new ConstantCoefficient(1.0 / dt);
auto *vmass_blfi = new VectorMassIntegrator(*dtCoeff);
// Setup of coefficient for stiffness term of Eq(13)
kinvisCoeff = new ConstantCoefficient(kin_vis);
auto *vdiff_blfi = new VectorDiffusionIntegrator(*kinvisCoeff);
// setup of Bilinear form of Eq(13)
velBForm = new ParBilinearForm(vfes);
if (numerical_integ)
{
vmass_blfi->SetIntRule(&ir_ni);
vdiff_blfi->SetIntRule(&ir_ni);
}
velBForm->AddDomainIntegrator(vmass_blfi);
velBForm->AddDomainIntegrator(vdiff_blfi);
if (partial_assembly) { velBForm->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
velBForm->Assemble();
velBForm->FormSystemMatrix(vel_ess_tdof, vOp);
//-------------------------------------------------------------------------
// Setup of coefficient for Eq(18)
pUnitVectorCoeff = new UnitVectorGridFunctionCoeff(pmesh->Dimension());
auto *pvel_lfi = new VectorDomainLFGradIntegrator(*pUnitVectorCoeff);
// Setup of coefficient for Eq(20)
nonlinTermCoeff = new NonLinTermVectorGridFunctionCoeff(pmesh->Dimension());
auto *p_nonlintermlfi = new VectorDomainLFIntegrator(*nonlinTermCoeff);
// Setup of coefficient for Eq(21)
prevVelLoadCoeff = new PrevVelVectorGridFunctionCoeff(pmesh->Dimension());
auto *prevVelLoadLFi = new VectorDomainLFIntegrator(*prevVelLoadCoeff);
// Setup of linear form of Eq(13)
velLForm = new ParLinearForm(vfes);
if (numerical_integ)
{
prevVelLoadLFi->SetIntRule(&ir_ni);
pvel_lfi->SetIntRule(&ir_ni);
p_nonlintermlfi->SetIntRule(&ir_ni);
}
velLForm->AddDomainIntegrator(prevVelLoadLFi);
velLForm->AddDomainIntegrator(pvel_lfi);
velLForm->AddDomainIntegrator(p_nonlintermlfi);
//-------------------------------------------------------------------------
if (partial_assembly)
{
Vector diag_pa(vfes->GetTrueVSize());
velBForm->AssembleDiagonal(diag_pa);
velInvPC = new OperatorJacobiSmoother(diag_pa, vel_ess_tdof);
}
else
{
velInvPC = new HypreSmoother(*vOp.As<HypreParMatrix>());
dynamic_cast<HypreSmoother *>(velInvPC)->SetType(HypreSmoother::Jacobi,
1);
}
velInv = new CGSolver(vfes->GetComm());
velInv->iterative_mode = true;
velInv->SetOperator(*vOp);
velInv->SetPreconditioner(*velInvPC);
velInv->SetPrintLevel(pl_velsolve);
velInv->SetRelTol(rtol_velsolve);
velInv->SetAbsTol(0.0);
velInv->SetMaxIter(1200);
}
void IncompressibleNavierSolver::Setup_auxiliary(real_t dt)
{
// GLL integration rule (Numerical Integration)
const IntegrationRule &ir_ni =
gll_rules.Get(vfes->GetFE(0)->GetGeomType(), 2 * velorder - 1);
Array<int> empty;
// setup of Bilinear form of Eq(14)
psiBForm = new ParBilinearForm(psifes);
auto *psidiff_blfi = new DiffusionIntegrator;
if (numerical_integ) { psidiff_blfi->SetIntRule(&ir_ni); }
psiBForm->AddDomainIntegrator(psidiff_blfi);
if (partial_assembly) { psiBForm->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
psiBForm->Assemble();
psiBForm->FormSystemMatrix(empty, psiOp);
//-------------------------------------------------------------------------
// Setup of coefficient for linear form in Eq(14)
DvelCoeff = new VectorGridFunctionCoefficient;
auto *Dvel_lfi = new DomainLFGradIntegrator(*DvelCoeff);
// Setup of linear form of Eq(14)
psiLForm = new ParLinearForm(psifes);
if (numerical_integ) { Dvel_lfi->SetIntRule(&ir_ni); }
psiLForm->AddDomainIntegrator(Dvel_lfi);
//-------------------------------------------------------------------------
if (partial_assembly)
{
int psifes_truevsize = psifes->GetTrueVSize();
mfem::Vector psin(psifes_truevsize);
psin = 0.0;
mfem::Vector respsi(psifes_truevsize);
respsi = 0.0;
lor = new ParLORDiscretization(*psiBForm, empty);
psiInvPC = new HypreBoomerAMG(lor->GetAssembledMatrix());
psiInvPC->SetPrintLevel(0);
psiInvPC->Mult(respsi, psin);
SpInvOrthoPC = new OrthoSolver(psifes->GetComm());
SpInvOrthoPC->SetSolver(*psiInvPC);
}
else
{
psiInvPC = new HypreBoomerAMG(*psiOp.As<HypreParMatrix>());
psiInvPC->SetPrintLevel(0);
SpInvOrthoPC = new OrthoSolver(psifes->GetComm());
SpInvOrthoPC->SetSolver(*psiInvPC);
}
psiInv = new CGSolver(psifes->GetComm());
psiInv->iterative_mode = true;
psiInv->SetOperator(*psiOp);
psiInv->SetPreconditioner(*SpInvOrthoPC);
psiInv->SetPrintLevel(pl_psisolve);
psiInv->SetRelTol(rtol_psisolve);
psiInv->SetAbsTol(0.0);
psiInv->SetMaxIter(1000);
}
void IncompressibleNavierSolver::Setup_pressure(real_t dt)
{
// GLL integration rule (Numerical Integration)
const IntegrationRule &ir_ni =
gll_rules.Get(vfes->GetFE(0)->GetGeomType(), 2 * velorder - 1);
Array<int> empty;
//-------------------------------------------------------------------------
// setup of Bilinear form of Eq(15)
pBForm = new ParBilinearForm(pfes);
auto *pmass_blfi = new MassIntegrator;
if (numerical_integ) { pmass_blfi->SetIntRule(&ir_ni); }
pBForm->AddDomainIntegrator(pmass_blfi);
if (partial_assembly) { pBForm->SetAssemblyLevel(AssemblyLevel::PARTIAL); }
pBForm->Assemble();
pBForm->FormSystemMatrix(empty, pOp);
//-------------------------------------------------------------------------
// Setup of divergence of velocity coefficient for linear form in Eq(15)
divVelCoeff = new DivergenceGridFunctionCoefficient(velGF[0]);
// Setup of coefficient for linear form in Eq(14)
pRHSCoeff = new GridFunctionCoefficient(&pRHS);
auto *p_lfi = new DomainLFIntegrator(*pRHSCoeff);
// Setup of linear form of Eq(15)
pLForm = new ParLinearForm(pfes);
if (numerical_integ) { p_lfi->SetIntRule(&ir_ni); }
pLForm->AddDomainIntegrator(p_lfi);
//-------------------------------------------------------------------------
if (partial_assembly)
{
Vector diag_pa(pfes->GetTrueVSize());
pBForm->AssembleDiagonal(diag_pa);
pInvPC = new OperatorJacobiSmoother(diag_pa, empty);
}
else
{
pInvPC = new HypreSmoother(*pOp.As<HypreParMatrix>());
dynamic_cast<HypreSmoother *>(pInvPC)->SetType(HypreSmoother::Jacobi, 1);
}
pInv = new CGSolver(pfes->GetComm());
pInv->iterative_mode = true;
pInv->SetOperator(*pOp);
pInv->SetPreconditioner(*pInvPC);
pInv->SetPrintLevel(pl_psolve);
pInv->SetRelTol(rtol_psolve);
pInv->SetAbsTol(0.0);
pInv->SetMaxIter(1000);
}
void IncompressibleNavierSolver::UpdateTimestepHistory(real_t dt) {}
void IncompressibleNavierSolver::Step(real_t &time, real_t dt, int current_step,
const bool vis_step)
{
this->Step_velocity(time, dt, current_step);
this->Step_auxiliary(time, dt, current_step);
this->Step_pressure(time, dt, current_step);
*velGF[1] = *velGF[0];
*pGF[1] = *pGF[0];
if (vis_step)
{
mfem::out << "It: " << iter << " | Iter_U: " << iter_vsolve
<< " | Iter_Psi: " << iter_psisolve
<< " | Iter_P: " << iter_psolve << "\n";
mfem::out << "It: " << iter << " | Resid_U: " << res_vsolve
<< " | Resid_Psi: " << res_psisolve
<< " | Resid_P: " << res_psisolve << "\n";
}
time += dt;
iter++;
}
void IncompressibleNavierSolver::Step_velocity(real_t &time, real_t dt,
int current_step)
{
for (auto &vel_dbc : vel_dbcs)
{
velGF[0]->ProjectBdrCoefficient(*vel_dbc.coeff, vel_dbc.attr);
velGF[1]->ProjectBdrCoefficient(*vel_dbc.coeff, vel_dbc.attr);
}
// Update state in coefficient for Eq(18)
pUnitVectorCoeff->SetGridFunction(pGF[1]);
// Update state in coefficient for Eq(20)
nonlinTermCoeff->SetGridFunction(velGF[1]);
// Update state in coefficient for Eq(21)
prevVelLoadCoeff->SetGridFunction(velGF[1], dt);
velLForm->Assemble();
velLForm->ParallelAssemble(velLF);
Vector X1, B1;
if (partial_assembly)
{
auto *vpC = vOp.As<ConstrainedOperator>();
EliminateRHS(*velBForm, *vpC, vel_ess_tdof, *velGF[0], velLF, X1, B1, 1);
}
else
{
velBForm->FormLinearSystem(vel_ess_tdof, *velGF[0], velLF, vOp, X1, B1,
1);
}
velInv->Mult(B1, X1);
iter_vsolve = velInv->GetNumIterations();
res_vsolve = velInv->GetFinalNorm();
velBForm->RecoverFEMSolution(X1, velLF, *velGF[0]);
}
void IncompressibleNavierSolver::Step_auxiliary(real_t &time, real_t dt,
int current_step)
{
// Compute new increment GF for LF of Eq(14) and update state in coefficient
subtract(1.0 / dt, *velGF[0], *velGF[1], DvGF);
DvelCoeff->SetGridFunction(&DvGF);
psiLForm->Assemble();
psiLForm->ParallelAssemble(psiLF);
Vector X2, B2;
Array<int> empty;
if (partial_assembly)
{
auto *psipC = psiOp.As<ConstrainedOperator>();
EliminateRHS(*psiBForm, *psipC, empty, psiGF, psiLF, X2, B2, 1);
}
else { psiBForm->FormLinearSystem(empty, psiGF, psiLF, psiOp, X2, B2, 1); }
psiInv->Mult(B2, X2);
iter_psisolve = psiInv->GetNumIterations();
res_psisolve = psiInv->GetFinalNorm();
psiBForm->RecoverFEMSolution(X2, psiLF, psiGF);
}
void IncompressibleNavierSolver::Step_pressure(real_t &time, real_t dt,
int current_step)
{
Array<int> empty;
// Compute new GF for LF of Eq(15) and update state in coefficient
divVelCoeff->SetGridFunction(velGF[0]);
divVelGF.ProjectCoefficient(*divVelCoeff);
add(*pGF[1], psiGF, pRHS);
add(pRHS, -1.0 * kin_vis, divVelGF, pRHS);
pRHSCoeff->SetGridFunction(&pRHS);
pLForm->Assemble();
pLForm->ParallelAssemble(pLF);
Vector X3, B3;
if (partial_assembly)
{
auto *ppC = pOp.As<ConstrainedOperator>();
EliminateRHS(*pBForm, *ppC, empty, *pGF[0], pLF, X3, B3, 1);
}
else { pBForm->FormLinearSystem(empty, *pGF[0], pLF, pOp, X3, B3, 1); }
pInv->Mult(B3, X3);
iter_psolve = pInv->GetNumIterations();
res_psisolve = pInv->GetFinalNorm();
pBForm->RecoverFEMSolution(X3, pLF, *pGF[0]);
}
void IncompressibleNavierSolver::EliminateRHS(Operator &A,
ConstrainedOperator &constrainedA,
const Array<int> &ess_tdof_list,
Vector &x, Vector &b, Vector &X,
Vector &B, int copy_interior)
{
const Operator *Po = A.GetOutputProlongation();
const Operator *Pi = A.GetProlongation();
const Operator *Ri = A.GetRestriction();
A.InitTVectors(Po, Ri, Pi, x, b, X, B);
if (!copy_interior) { X.SetSubVectorComplement(ess_tdof_list, 0.0); }
constrainedA.EliminateRHS(X, B);
}
real_t IncompressibleNavierSolver::ComputeCFL(ParGridFunction &u, real_t dt)
{
return 0;
}
void IncompressibleNavierSolver::AddVelDirichletBC(VectorCoefficient *coeff,
Array<int> &attr)
{
vel_dbcs.emplace_back(attr, coeff);
if (verbose && pmesh->GetMyRank() == 0)
{
mfem::out << "Adding Velocity Dirichlet BC to attributes ";
for (int i = 0; i < attr.Size(); ++i)
{
if (attr[i] == 1) { mfem::out << i << " "; }
}
mfem::out << std::endl;
}
for (int i = 0; i < attr.Size(); ++i)
{
MFEM_ASSERT((vel_ess_attr[i] && attr[i]) == 0,
"Duplicate boundary definition deteceted.");
if (attr[i] == 1) { vel_ess_attr[i] = 1; }
}
}
void IncompressibleNavierSolver::AddVelDirichletBC(VecFuncT *f,
Array<int> &attr)
{
AddVelDirichletBC(new VectorFunctionCoefficient(pmesh->Dimension(), f),
attr);
}
IncompressibleNavierSolver::~IncompressibleNavierSolver()
{
delete velBForm;
delete psiBForm;
delete pBForm;
delete kinvisCoeff;
delete dtCoeff;
for (int i = 0; i < torder + 1; i++)
{
delete velGF[i];
delete pGF[i];
}
delete DvelCoeff;
delete divVelCoeff;
delete pRHSCoeff;
delete pUnitVectorCoeff;
delete velInv;
delete velInvPC;
delete psiInv;
delete SpInvOrthoPC;
delete psiInvPC;
delete lor;
delete pInv;
delete pInvPC;
delete vfec;
delete psifec;
delete pfec;
delete vfes;
delete psifes;
delete pfes;
}
@@ -0,0 +1,338 @@
// Copyright (c) 2010-2024, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#define INCOMP_NAVIER_VERSION 0.1
#include "mfem.hpp"
namespace mfem::incompressible_navier
{
using VecFuncT = void(const Vector &x, real_t t, Vector &u);
using ScalarFuncT = real_t(const Vector &x, real_t t);
// Coefficient which computed contribution of Eq(18)
class UnitVectorGridFunctionCoeff : public VectorCoefficient
{
public:
UnitVectorGridFunctionCoeff(int dim): VectorCoefficient(dim * dim) {}
void Eval(Vector &V, ElementTransformation &T,
const IntegrationPoint &ip) override
{
real_t coeffVal = gridfunc_->GetValue(T, ip);
V.SetSize(vdim);
V = 0.0; // FIXME
V[0] = coeffVal;
V[3] = coeffVal;
}
void SetGridFunction(GridFunction *gridfunc) { gridfunc_ = gridfunc; }
GridFunction *gridfunc_ = nullptr;
};
// Coefficient which computed contribution of Eq(21)
class PrevVelVectorGridFunctionCoeff : public VectorCoefficient
{
public:
PrevVelVectorGridFunctionCoeff(int dim): VectorCoefficient(dim) {}
void Eval(Vector &V, ElementTransformation &T, const IntegrationPoint &ip)
{
V.SetSize(vdim);
gridFuncCoeff->Eval(V, T, ip);
V *= 1.0 / dt_;
}
void SetGridFunction(GridFunction *gridfunc, real_t dt)
{
gridfunc_ = gridfunc;
dt_ = dt;
delete gridFuncCoeff;
gridFuncCoeff = new VectorGridFunctionCoefficient(gridfunc);
}
GridFunction *gridfunc_ = nullptr;
VectorGridFunctionCoefficient *gridFuncCoeff = nullptr;
real_t dt_;
};
// Coefficient which computed contribution of Eq(20)
class NonLinTermVectorGridFunctionCoeff : public VectorCoefficient
{
public:
NonLinTermVectorGridFunctionCoeff(int dim): VectorCoefficient(dim) {}
void Eval(Vector &V, ElementTransformation &T, const IntegrationPoint &ip)
{
Vector val(vdim);
Vector resultVal(vdim);
DenseMatrix vecGrad;
V.SetSize(vdim);
gridFuncCoeff->Eval(val, T, ip);
gridfunc_->GetVectorGradient(T, vecGrad);
vecGrad.MultTranspose(val, V);
V *= -1.0;
}
void SetGridFunction(ParGridFunction *gridfunc)
{
delete gridFuncCoeff;
gridfunc_ = gridfunc;
gridFuncCoeff = new VectorGridFunctionCoefficient(gridfunc);
}
VectorGridFunctionCoefficient *gridFuncCoeff = nullptr;
ParGridFunction *gridfunc_ = nullptr;
};
/// Container for a Dirichlet boundary condition of the velocity field.
class VelDirichletBC_T
{
public:
VelDirichletBC_T(Array<int> attr, VectorCoefficient *coeff):
attr(attr), coeff(coeff)
{
}
VelDirichletBC_T(VelDirichletBC_T &&obj)
{
// Deep copy the attribute array
this->attr = obj.attr;
// Move the coefficient pointer
this->coeff = obj.coeff;
obj.coeff = nullptr;
}
~VelDirichletBC_T() { delete coeff; }
Array<int> attr;
VectorCoefficient *coeff;
};
/// Transient incompressible Navier Stokes solver in a split scheme formulation.
/**
* This implementation of a transient incompressible Navier Stokes solver uses
* the non-dimensionalized formulation. The coupled momentum and
* incompressibility equations are decoupled using the split scheme described in
* [1]. This leads to three solving steps.
*
*/
class IncompressibleNavierSolver
{
public:
/// Initialize data structures, set FE space order and kinematic viscosity.
/**
* The ParMesh @a mesh can be a linear or curved parallel mesh. The @a order
* of the finite element spaces is
*/
IncompressibleNavierSolver(ParMesh *mesh, int velorder, int porder,
int tOrder, real_t kin_vis);
/// Initialize forms, solvers and preconditioners.
void Setup(real_t dt);
void Setup_velocity(real_t dt);
void Setup_auxiliary(real_t dt);
void Setup_pressure(real_t dt);
/// Compute solution at the next time step t+dt.
/**
* This method can
*/
void Step(real_t &time, real_t dt, int cur_step, bool vis_step);
void Step_velocity(real_t &time, real_t dt, int cur_step);
void Step_auxiliary(real_t &time, real_t dt, int cur_step);
void Step_pressure(real_t &time, real_t dt, int cur_step);
/// Return a pointer to the provisional velocity ParGridFunction.
ParGridFunction *GetProvisionalVelocity() { return velGF[1]; }
/// Return a pointer to the current velocity ParGridFunction.
ParGridFunction *GetCurrentVelocity() { return velGF[0]; }
/// Return a pointer to the current pressure ParGridFunction.
ParGridFunction *GetCurrentPressure() { return pGF[0]; }
/// Return a pointer to the current pressure ParGridFunction.
ParGridFunction *GetCurrentPsi() { return &psiGF; }
/// Add a Dirichlet boundary condition to the velocity field.
void AddVelDirichletBC(VectorCoefficient *coeff, Array<int> &attr);
void AddVelDirichletBC(VecFuncT *f, Array<int> &attr);
/// Add a Dirichlet boundary condition to the pressure field.
// void AddPresDirichletBC(Coefficient *coeff, Array<int> &attr);
// void AddPresDirichletBC(ScalarFuncT *f, Array<int> &attr);
/// Enable partial assembly for every operator.
void EnablePA(bool pa) { partial_assembly = pa; }
/// Enable numerical integration rules. This means collocated quadrature at
/// the nodal points.
void EnableNI(bool ni) { numerical_integ = ni; }
/// Print timing summary of the solving routine.
void PrintTimingData();
~IncompressibleNavierSolver();
/// Rotate entries in the time step and solution history arrays.
void UpdateTimestepHistory(real_t dt);
/// Compute CFL
real_t ComputeCFL(ParGridFunction &u, real_t dt);
protected:
/// Eliminate essential BCs in an Operator and apply to RHS.
void EliminateRHS(Operator &A, ConstrainedOperator &constrainedA,
const Array<int> &ess_tdof_list, Vector &x, Vector &b,
Vector &X, Vector &B, int copy_interior = 0);
/// Enable/disable debug output.
bool debug = false;
/// Enable/disable verbose output.
bool verbose = true;
/// Enable/disable partial assembly of forms.
bool partial_assembly = false;
/// Enable/disable numerical integration rules of forms.
bool numerical_integ = false;
/// The parallel mesh.
ParMesh *pmesh = nullptr;
/// The order of the velocity and pressure space.
const int velorder;
const int porder;
const int torder;
/// Kinematic viscosity (dimensionless).
const real_t kin_vis;
Coefficient *kinvisCoeff = nullptr;
Coefficient *dtCoeff = nullptr;
IntegrationRules gll_rules;
/// Velocity $H^1$ finite element collection.
FiniteElementCollection *vfec = nullptr;
/// Psi $H^1$ finite element collection.
FiniteElementCollection *psifec = nullptr;
/// Pressure $H^1$ finite element collection.
FiniteElementCollection *pfec = nullptr;
/// Velocity $(H^1)^d$ finite element space.
ParFiniteElementSpace *vfes = nullptr;
/// Psi $(H^1)^d$ finite element space.
ParFiniteElementSpace *psifes = nullptr;
/// Pressure $H^1$ finite element space.
ParFiniteElementSpace *pfes = nullptr;
ParBilinearForm *velBForm = nullptr; // vmass + vdiff
ParBilinearForm *psiBForm = nullptr; // diffusion
ParBilinearForm *pBForm = nullptr; // mass
ParLinearForm *velLForm = nullptr; // vLF + vLFGrad + vLF
ParLinearForm *psiLForm = nullptr; // LFGrad
ParLinearForm *pLForm = nullptr; // LF
// current (0) and provisional (1) velocity
std::vector<ParGridFunction *> velGF;
// current (0) pressure ParGridFunction.
std::vector<ParGridFunction *> pGF;
ParGridFunction psiGF;
ParGridFunction DvGF, divVelGF, pRHS;
VectorGridFunctionCoefficient *DvelCoeff = nullptr;
DivergenceGridFunctionCoefficient *divVelCoeff = nullptr;
GridFunctionCoefficient *pRHSCoeff = nullptr;
UnitVectorGridFunctionCoeff *pUnitVectorCoeff = nullptr;
NonLinTermVectorGridFunctionCoeff *nonlinTermCoeff = nullptr;
PrevVelVectorGridFunctionCoeff *prevVelLoadCoeff = nullptr;
OperatorHandle vOp;
OperatorHandle psiOp;
OperatorHandle pOp;
Solver *velInvPC = nullptr;
CGSolver *velInv = nullptr;
ParLORDiscretization *lor = nullptr;
HypreBoomerAMG *psiInvPC = nullptr;
OrthoSolver *SpInvOrthoPC = nullptr;
CGSolver *psiInv = nullptr;
Solver *pInvPC = nullptr;
CGSolver *pInv = nullptr;
Vector velLF, psiLF, pLF;
// All essential attributes.
Array<int> vel_ess_attr;
Array<int> pres_ess_attr;
// All essential true dofs.
Array<int> vel_ess_tdof;
Array<int> pres_ess_tdof;
// Bookkeeping for velocity dirichlet bcs.
std::vector<VelDirichletBC_T> vel_dbcs;
// Print levels.
int pl_psolve = 0;
int pl_psisolve = 0;
int pl_velsolve = 0;
int pl_amg = 0;
#if defined(MFEM_USE_DOUBLE)
real_t rtol_psolve = 1e-12;
real_t rtol_psisolve = 1e-12;
real_t rtol_velsolve = 1e-12;
#elif defined(MFEM_USE_SINGLE)
real_t rtol_psolve = 1e-9;
real_t rtol_psisolve = 1e-5;
real_t rtol_velsolve = 1e-7;
#else
#error "Only single and double precision are supported!"
#endif
// Iteration counts.
int iter = 1, iter_vsolve = 0, iter_psolve = 0, iter_psisolve = 0;
// Residuals.
real_t res_vsolve = 0.0, res_psolve = 0.0, res_psisolve = 0.0;
};
} // namespace mfem::incompressible_navier
@@ -0,0 +1,206 @@
#include <algorithm>
#include <iostream>
#include <memory>
#include <sstream>
#include <unistd.h>
#define NVTX_COLOR ::gpu::nvtx::kMagenta
#include "incompressible_navier_nvtx.hpp"
///////////////////////////////////////////////////////////////////////////////
int navier(int argc, char *argv[], double &u, double &p, double &Ψ);
///////////////////////////////////////////////////////////////////////////////
template <class T>
std::enable_if_t<!std::numeric_limits<T>::is_integer, bool>
AlmostEq(T x, T y, T tolerance = 100.0 * std::numeric_limits<T>::epsilon())
{
const T neg = std::abs(x - y);
constexpr T min = std::numeric_limits<T>::min();
constexpr T eps = std::numeric_limits<T>::epsilon();
const T min_abs = std::min(std::abs(x), std::abs(y));
if (std::abs(min_abs) == 0.0) { return neg < eps; }
return (neg / (1.0 + std::max(min, min_abs))) < tolerance;
}
///////////////////////////////////////////////////////////////////////////////
using char_uptr = std::unique_ptr<char[]>;
using args_ptr_t = std::vector<char_uptr>;
using args_t = std::vector<char *>;
///////////////////////////////////////////////////////////////////////////////
struct Results
{
double u{}, p{}, Ψ {};
};
///////////////////////////////////////////////////////////////////////////////
struct Test
{
static constexpr const char *binary = "incompNS_2Dtest ";
static constexpr const char *common = "-no-vis -no-pv";
const std::string options;
const Results results;
Test(const char *args, const Results &res):
options(std::string(args) + " " + common), results(res)
{
dbg("options: {}", options.c_str());
dbg("results: U={:.15e}, P={:.15e}, Ψ={:.15e}",
results.u, results.p, results.Ψ);
}
std::string Command() const { return binary + options; }
};
///////////////////////////////////////////////////////////////////////////////
#if 1 // dot product reduction (miniapps/navier/incompNS_2Dtest.cpp#L182)
static const Test gold[] =
{
{
"-nx 9 -ny 3 -sr 0",
{ 3.056430866716070e-06, 1.504027632950462e-01, 7.132601171183242e-08 }
},
{
"-nx 16 -ny 8 -sr 0",
{ 1.409287729554512e-05, 5.718053801962010e-01, 1.904938419441012e-07 }
},
// {
// "-nx 9 -ny 3 -sr 1",
// { 1.23258426138828e-05, 5.18207619597952e-01, 1.956175418199867e-07 }
// },
// {
// "-nx 9 -ny 3 -sr 2",
// { 4.74381190869396e-05, 1.80653923294262e+00, 5.90778654333642e-07 }
// },
};
#else // Norml2 reduction
///////////////////////////////////////////////////////////////////////////////
const Test runs[] =
{
{
"-nx 9 -ny 3 -sr 0",
{ 1.746844586767688e-03, 3.874421956807944e-01, 2.670296315544763e-04 }
},
// 1.748265101955712e-03, 3.878179512284757e-01, 2.670693013280680e-04 //
// Release { "-nx 9 -ny 3 -sr 1",
// { 1.232584261388279e-05, 5.182076195979519e-01, 1.956175418199866e-07 }
// },
// { "-nx 9 -ny 3 -sr 2",
// { 4.743811908693962e-05, 1.806539232942622e+00, 5.907786543336419e-07 }
// },
};
#endif
///////////////////////////////////////////////////////////////////////////////
int NavierTest(const int k, const Test &run)
{
dbg();
static args_ptr_t args_ptr;
args_t args;
std::istringstream iss(run.Command());
auto add_arg = [&](std::string token) -> char_uptr
{
auto arg_ptr = std::make_unique<char[]>(token.size() + 1);
std::memcpy(arg_ptr.get(), token.c_str(), token.size() + 1);
arg_ptr[token.size()] = '\0';
return arg_ptr;
};
std::string token;
while (iss >> token)
{
auto arg_ptr = add_arg(token);
args.push_back(arg_ptr.get());
args_ptr.emplace_back(std::move(arg_ptr));
}
args.push_back(nullptr);
auto launch = [&args, &run, &k]() -> int
{
// dbg("Launching test #{}: \x1B[33m{}\x1B[m", k, gold.Command().c_str());
Results res{};
navier(args.size() - 1, args.data(), res.u, res.p, res.Ψ);
// dbg("Results: U={:.15e}, P={:.15e}, Ψ={:.15e}", res.u, res.p, res.Ψ);
const bool u = AlmostEq(res.u, run.results.u);
const bool p = AlmostEq(res.p, run.results.p);
const bool Ψ = AlmostEq(res.Ψ, run.results.Ψ);
constexpr auto ok = [](bool ok) -> int { return ok ? 32 : 31; };
constexpr auto to_string = [](args_t &args) -> std::string
{
std::string args_str;
for (auto &arg : args)
{
if (!arg) { break; }
args_str += std::string(arg) + " ";
}
return args_str;
};
dbg("#{} \x1B[33m{}\x1B[m", k, to_string(args).c_str());
dbg("U: \x1B[33m{:.15e} \x1B[{}m{:.15e}", run.results.u, ok(u), res.u );
dbg("P: \x1B[33m{:.15e} \x1B[{}m{:.15e}", run.results.p, ok(p), res.p);
dbg("Ψ: \x1B[33m{:.15e} \x1B[{}m{:.15e}", run.results.Ψ, ok(Ψ), res.Ψ);
if (u && p && Ψ) { return std::cout << "" << std::endl, EXIT_SUCCESS; }
else { return std::cout << "" << std::endl, EXIT_FAILURE; }
};
// first launch with default arguments
if (launch() != EXIT_SUCCESS) { return EXIT_FAILURE; }
// second launch with the same arguments, but with -pa
args.pop_back(); // nullptr
auto arg_pa_ptr = add_arg("-pa");
args.push_back(arg_pa_ptr.get());
args_ptr.emplace_back(std::move(arg_pa_ptr));
args.push_back(nullptr);
if (launch() != EXIT_SUCCESS) { return EXIT_FAILURE; }
return EXIT_SUCCESS;
}
///////////////////////////////////////////////////////////////////////////////
int main(int argc, char *argv[])
try
{
dbg();
int opt;
int test = -1;
auto show_usage = [](const int ret = EXIT_FAILURE)
{
printf("Usage: program [-a <arg>] [-b <arg>] [-h]\n");
printf(" -t <test> Optional test number \n");
printf(" -h Show this help message\n");
exit(ret);
};
while ((opt = getopt(argc, argv, "t:h")) != -1)
{
switch (opt)
{
case 't': test = std::atoi(optarg); break;
case 'h': show_usage(EXIT_SUCCESS);
default: show_usage(EXIT_FAILURE);
}
}
constexpr int N_TESTS = sizeof(gold) / sizeof(Test);
if (test >= 0 && test < N_TESTS) { return NavierTest(test, gold[test]); }
int k = 0;
for (auto &run : gold)
{
if (NavierTest(k++, run) != EXIT_SUCCESS) { return EXIT_FAILURE; }
}
return EXIT_SUCCESS;
}
catch (std::exception &e)
{
std::cerr << "\033[31m..xxxXXX[ERROR]XXXxxx.." << std::endl;
std::cerr << "\033[31m{}" << e.what() << std::endl;
return EXIT_FAILURE;
}
-12
View File
@@ -80,10 +80,6 @@ add_mfem_miniapp(nurbs_solenoidal
LIBRARIES mfem)
add_dependencies(nurbs_solenoidal copy_miniapps_nurbs_data)
add_mfem_miniapp(nurbs_surface
MAIN nurbs_surface.cpp
LIBRARIES mfem)
if (MFEM_ENABLE_TESTING)
add_test(NAME nurbs_ex1_1d_r1_o2_ser
COMMAND $<TARGET_FILE:nurbs_ex1> -no-vis
@@ -251,14 +247,6 @@ if (MFEM_ENABLE_TESTING)
COMMAND $<TARGET_FILE:nurbs_solenoidal> -no-vis
-m ${PROJECT_SOURCE_DIR}/data/cube-nurbs.mesh -r 1 -o 2)
add_test(NAME nurbs_surface_10_10_10_10_ex1_o3_ser
COMMAND $<TARGET_FILE:nurbs_surface> -no-vis
-o 3 -nx 10 -ny 10 -fnx 10 -fny 10 -ex 1 -orig)
add_test(NAME nurbs_surface_10_10_40_40_ex1_o3_ser
COMMAND $<TARGET_FILE:nurbs_surface> -no-vis
-o 3 -nx 10 -ny 10 -fnx 40 -fny 14 -ex 1)
endif()
if (MFEM_USE_MPI)
+2 -9
View File
@@ -21,7 +21,7 @@ MFEM_LIB_FILE = mfem_is_not_built
-include $(CONFIG_MK)
SEQ_MINIAPPS = nurbs_ex1 nurbs_patch_ex1 nurbs_ex3 nurbs_ex5 nurbs_ex24 \
nurbs_curveint nurbs_printfunc nurbs_solenoidal nurbs_naca_cmesh nurbs_surface
nurbs_curveint nurbs_printfunc nurbs_solenoidal nurbs_naca_cmesh
PAR_MINIAPPS = nurbs_ex1p nurbs_ex11p
ifeq ($(MFEM_USE_MPI),NO)
MINIAPPS = $(SEQ_MINIAPPS)
@@ -158,13 +158,6 @@ nurbs_naca_cmesh-test-seq: nurbs_naca_cmesh
nurbs_printfunc-test-seq: nurbs_printfunc
@$(call mfem-test,$<,, NURBS miniapp)
SURF_ARGS_1 := -o 3 -nx 10 -ny 10 -fnx 10 -fny 10 -ex 1 -orig
SURF_ARGS_2 := -o 3 -nx 10 -ny 10 -fnx 40 -fny 40 -ex 1
nurbs_surface-test-seq: nurbs_surface
@$(call mfem-test,$<,, NURBS miniapp,$(SURF_ARGS_1))
@$(call mfem-test,$<,, NURBS miniapp,$(SURF_ARGS_2))
EX1P_ARGS_1 :=
EX1P_ARGS_2 := -m ../../data/pipe-nurbs-2d.mesh -o 2 -no-ibp
EX1P_ARGS_3 := -m ../../data/ball-nurbs.mesh -o 2 --weak-bc -r 0
@@ -199,6 +192,6 @@ clean-build:
clean-exec:
@rm -f refined.mesh sin-fit.mesh ex5.mesh exsol.mesh mesh.* sol.* mode_*
@rm -f naca-cmesh.mesh sol_?.gf *-Surface.mesh
@rm -f naca-cmesh.mesh sol_?.gf
@rm -rf Example1* Example3* Example5* Solenoidal_* ParaView
@rm -rf CurveInt Naca_cmesh glvis_naca-cmesh.mesh solution.dat
-655
View File
@@ -1,655 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
//
// --------------------------------------------------------
// NURBS Surface: Interpolate a 3D Surface in a NURBS Patch
// --------------------------------------------------------
//
// Compile with: make nurbs_surface
//
// Sample runs: nurbs_surface -o 3 -nx 10 -ny 10 -fnx 10 -fny 10 -ex 1 -orig
// nurbs_surface -o 3 -nx 10 -ny 10 -fnx 40 -fny 40 -ex 1
// nurbs_surface -o 3 -nx 20 -ny 20 -fnx 10 -fny 10 -ex 1
// nurbs_surface -o 3 -nx 20 -ny 20 -fnx 40 -fny 40 -ex 1 -j 0.5
// nurbs_surface -o 3 -nx 10 -ny 10 -fnx 10 -fny 10 -ex 2 -orig
// nurbs_surface -o 3 -nx 10 -ny 10 -fnx 40 -fny 40 -ex 2
// nurbs_surface -o 3 -nx 20 -ny 20 -fnx 10 -fny 10 -ex 2
// nurbs_surface -o 3 -nx 10 -ny 10 -fnx 10 -fny 10 -ex 3 -orig
// nurbs_surface -o 3 -nx 10 -ny 10 -fnx 40 -fny 40 -ex 3
// nurbs_surface -o 3 -nx 20 -ny 20 -fnx 10 -fny 10 -ex 3
// nurbs_surface -o 3 -nx 20 -ny 10 -fnx 20 -fny 10 -ex 4 -orig
// * nurbs_surface -o 3 -nx 20 -ny 10 -fnx 80 -fny 40 -ex 4
// * nurbs_surface -o 3 -nx 40 -ny 20 -fnx 20 -fny 10 -ex 4
// * nurbs_surface -o 3 -nx 100 -ny 100 -fnx 100 -fny 100 -ex 5 -orig
// * nurbs_surface -o 3 -nx 100 -ny 100 -fnx 400 -fny 400 -ex 5
// * nurbs_surface -o 3 -nx 200 -ny 200 -fnx 100 -fny 100 -ex 5
//
// Description: This example demonstrates the use of MFEM to interpolate an
// input surface point grid in 3D using a NURBS surface. The NURBS
// surface can then be sampled to generate an output mesh of
// arbitrary resolution while staying close to the input geometry.
#include "mfem.hpp"
#include <fstream>
#include <iostream>
using namespace std;
using namespace mfem;
// Example data for 3D point grid on surface, given by an analytic function.
void SurfaceGridExample(int example, int nx, int ny, Array3D<real_t> &vertices,
real_t jitter);
// Write a linear surface mesh with given vertex positions in v.
void WriteLinearMesh(int nx, int ny, const Array3D<real_t> &v,
const std::string &basename, bool visualization = false,
int x = 0, int y = 0, int w = 500, int h = 500);
// Given an input grid of 3D points on a surface, this class computes a NURBS
// surface of given order that interpolates the vertices of the input grid.
class SurfaceInterpolator
{
public:
/// Constructor for a given 2D point grid size and NURBS order.
SurfaceInterpolator(int num_elem_x, int num_elem_y, int order);
/// Create a surface interpolating the 2D grid of 3D points in @a input3D.
void CreateSurface(const Array3D<real_t> &input3D);
/// Sample the surface with the given grid size, storing points in
/// @a output3D.
void SampleSurface(int num_elem_x, int num_elem_y, bool compareOriginal,
Array3D<real_t> &output3D);
/** @brief Write the NURBS surface mesh to file, defined coordinate-wise by
the entries of @a cmesh. */
void WriteNURBSMesh(const std::string &basename, bool visualization = false,
int x = 0, int y = 0, int w = 500, int h = 500);
protected:
/** @brief Compute the NURBS mesh interpolating the given coordinate of the
grid of 3D points in @a input3D. */
void ComputeNURBS(int coordinate, const Array3D<real_t> &input3D);
private:
int nx, ny; // Number of elements in two directions of the surface grid
int orderNURBS; // NURBS degree
real_t hx, hy, hz; // Grid size in reference space
Array3D<real_t> initial3D; // Initial grid of points
static constexpr int dim = 3;
Array<int> ncp; // Number of control points in each direction
Array<int> nks; // Number of knot-spans in each direction
std::vector<Vector> ugrid; // Parameter space [0,1]^2 grid point coordinates
std::vector<KnotVector> kv; // KnotVectors in each direction
std::unique_ptr<NURBSPatch> patch; // Pointer to the only patch in the mesh
Mesh mesh; // NURBS mesh representing the surface
std::vector<Mesh> cmesh; // NURBS meshes representing point components
};
int main(int argc, char *argv[])
{
// Parse command-line options
int nx = 4;
int ny = 4;
int fnx = 40;
int fny = 40;
int order = 3;
int example = 1;
bool visualization = true;
bool compareOriginal = false;
real_t jitter = 0.0;
OptionsParser args(argc, argv);
args.AddOption(&example, "-ex", "--example",
"Example data");
args.AddOption(&nx, "-nx", "--nx",
"Number of elements in x");
args.AddOption(&ny, "-ny", "--ny",
"Number of elements in y");
args.AddOption(&fnx, "-fnx", "--fnx",
"Number of resampled elements in x");
args.AddOption(&fny, "-fny", "--fny",
"Number of resampled elements in y");
args.AddOption(&order, "-o", "--order",
"NURBS finite element order (polynomial degree)");
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
"--no-visualization",
"Enable or disable GLVis visualization.");
args.AddOption(&compareOriginal, "-orig", "--compare-original", "-no-orig",
"--no-compare-original",
"Compare to the original mesh?");
args.AddOption(&jitter, "-j", "--jitter",
"Relative jittering in (0,1) to add to the input point "
"coordinates on a uniform nx x ny grid (0 by default)");
args.Parse();
if (!args.Good())
{
args.PrintUsage(cout);
return 1;
}
args.PrintOptions(cout);
if (compareOriginal && (fnx != nx || fny != ny))
{
cout << "Comparing to the original mesh requires the same number of "
<< "samples!\n";
return 1;
}
// Dimensions of the 3 surfaces (Input, NURBS, Output)
cout << "Input Surface: " << nx << " x " << ny << " linear elements\n";
cout << "NURBS Surface: " << nx + 1 - order << " x " << ny + 1 - order
<< " knot elements of order " << order << "\n";
cout << "Output Surface: " << fnx << " x " << fny << " linear elements\n";
// Set the vertex coordinates of the initial linear mesh
constexpr int dim = 3;
Array3D<real_t> input3D(nx + 1, ny + 1, dim);
SurfaceGridExample(example, nx, ny, input3D, jitter);
// Create a NURBS surface for the given nx, ny and order parameters that
// interpolates the input vertex coordinates
SurfaceInterpolator surf(nx, ny, order);
surf.CreateSurface(input3D);
// Compute the vertex coordinates of the output linear mesh by sampling the
// values from the NURBS surface
Array3D<real_t> output3D(fnx + 1, fny + 1, dim);
surf.SampleSurface(fnx, fny, compareOriginal, output3D);
// Save and optionally visualize the 3 surfaces (Input, NURBS, Output)
WriteLinearMesh(nx, ny, input3D, "Input-Surface", visualization, 0, 0);
surf.WriteNURBSMesh("NURBS-Surface", visualization, 502, 0);
WriteLinearMesh(fnx, fny, output3D, "Output-Surface", visualization, 1004, 0);
return 0;
}
// f(x,y) = sin(2 * pi * x) * sin(2 * pi * y)
void Function1(real_t u, real_t v, real_t &x, real_t &y, real_t &z)
{
x = u;
y = v;
z = sin(2.0 * M_PI * u) * sin(2.0 * M_PI * v);
}
// Part of the parametric surface of a sphere, using spherical coordinates.
void Function2(real_t u, real_t v, real_t &x, real_t &y, real_t &z)
{
constexpr real_t r = 1.0;
constexpr real_t pi_4 = M_PI * 0.25;
constexpr real_t phi0 = -3*pi_4;
constexpr real_t phi1 = 3*pi_4;
constexpr real_t theta0 = pi_4;
constexpr real_t theta1 = 3 * pi_4;
const real_t phi = (phi0 * (1.0 - v)) + (phi1 * v);
const real_t theta = (theta0 * (1.0 - u)) + (theta1 * u);
x = r * sin(theta) * cos(phi);
y = r * sin(theta) * sin(phi);
z = r * cos(theta);
}
// Helicoid surface
void Function3(real_t u, real_t v, real_t &x, real_t &y, real_t &z)
{
x = u * cos(2.0 * M_PI * v);
y = u * sin(2.0 * M_PI * v);
z = v;
}
// Mobius strip
void Function4(real_t u, real_t v, real_t &x, real_t &y, real_t &z)
{
constexpr int twists = 1;
const real_t a = 1.0 + 0.5 * ((2.0 * v) - 1.0) * cos(2.0 * M_PI * twists * u);
x = a * cos(2.0 * M_PI * u);
y = a * sin(2.0 * M_PI * u);
z = 0.5 * (2.0 * v - 1.0) * sin(2.0 * M_PI * twists * u);
}
// Breather surface
void Function5(real_t u, real_t v, real_t &x, real_t &y, real_t &z)
{
const real_t m = 13.2 * ((2.0 * u) - 1.0);
const real_t n = 37.4 * ((2.0 * v) - 1.0);
constexpr real_t b = 0.4;
constexpr real_t r = 1.0 - (b*b);
const real_t w = sqrt(r);
const real_t denom = b * (pow(w*cosh(b*m),2) + pow(b*sin(w*n),2));
x = -m + (2*r*cosh(b*m)*sinh(b*m)) / denom;
y = (2*w*cosh(b*m)*(-(w*cos(n)*cos(w*n)) - sin(n)*sin(w*n))) / denom;
z = (2*w*cosh(b*m)*(-(w*sin(n)*cos(w*n)) + cos(n)*sin(w*n))) / denom;
}
void SurfaceFunction(int example, real_t u, real_t v,
real_t &x, real_t &y, real_t &z)
{
switch (example)
{
case 1:
Function1(u, v, x, y, z);
break;
case 2:
Function2(u, v, x, y, z);
break;
case 3:
Function3(u, v, x, y, z);
break;
case 4:
Function4(u, v, x, y, z);
break;
default:
Function5(u, v, x, y, z);
};
}
// Example data for 3D point grid on surface, given by an analytic function.
void SurfaceExample(int example, const std::vector<Vector> &grid,
Array3D<real_t> &v3D, real_t jitter)
{
int seed = (int)time(0);
srand((unsigned)seed);
real_t h0 = grid[0][1]-grid[0][0], h1 = grid[1][1]-grid[1][0];
for (int i = 0; i < grid[0].Size(); i++)
{
for (int j = 0; j < grid[1].Size(); j++)
{
if (i != 0 && i != grid[0].Size()-1 && j != 0 && j != grid[1].Size()-1)
{
SurfaceFunction(example, grid[0][i] + rand_real()*h0*jitter,
grid[1][j] + rand_real()*h1*jitter,
v3D(i, j, 0), v3D(i, j, 1), v3D(i, j, 2));
}
else
{
SurfaceFunction(example, grid[0][i], grid[1][j],
v3D(i, j, 0), v3D(i, j, 1), v3D(i, j, 2));
}
}
}
}
void SurfaceGridExample(int example, int nx, int ny, Array3D<real_t> &vertices,
real_t jitter = 0)
{
// Define a uniform grid of the reference parameter space [0,1]^2
std::vector<Vector> uniformGrid(2);
for (int i = 0; i < 2; ++i)
{
const int n = (i == 0) ? nx : ny;
const real_t h = 1.0 / n;
uniformGrid[i].SetSize(n + 1);
for (int j = 0; j <= n; ++j) { uniformGrid[i][j] = j * h; }
}
SurfaceExample(example, uniformGrid, vertices, jitter);
}
// Write a linear surface mesh with given vertex positions in v.
void WriteLinearMesh(int nx, int ny, const Array3D<real_t> &v,
const std::string &basename, bool visualization,
int x, int y, int w, int h)
{
const int nv = (nx + 1) * (ny + 1);
const int nelem = nx * ny;
constexpr int dim = 3; // Spatial dimension
Mesh lmesh(2, nv, nelem, 0, dim);
Vector vertex(dim);
for (int i = 0; i <= nx; ++i)
{
for (int j = 0; j <= ny; ++j)
{
for (int k = 0; k < dim; ++k) { vertex[k] = v(i, j, k); }
lmesh.AddVertex(vertex);
}
}
Array<int> verts(4);
auto vID = [&](int i, int j)
{
return j + (i * (ny + 1));
};
for (int i = 0; i < nx; ++i)
{
for (int j = 0; j < ny; ++j)
{
verts[0] = vID(i, j);
verts[1] = vID(i+1, j);
verts[2] = vID(i+1, j+1);
verts[3] = vID(i, j+1);
Element* el = lmesh.NewElement(Element::QUADRILATERAL);
el->SetVertices(verts);
lmesh.AddElement(el);
}
}
lmesh.FinalizeTopology();
ofstream mesh_ofs(basename + ".mesh");
mesh_ofs.precision(8);
lmesh.Print(mesh_ofs);
if (visualization)
{
char vishost[] = "localhost";
constexpr int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "mesh\n" << lmesh
<< "window_title '" << basename << "'"
<< "window_geometry "
<< x << " " << y << " " << w << " " << h << "\n"
<< "keys PPPPPPPPAattttt******\n"
<< flush;
}
}
// Compute error of interpolation with respect to an input grid of point data.
void CheckError(const Array3D<real_t> &a, const Array3D<real_t> &b, int c,
int nx, int ny)
{
real_t maxErr = 0.0;
for (int i = 0; i <= nx; ++i)
{
for (int j = 0; j <= ny; ++j)
{
const real_t err_ij = std::abs(a(i, j, c) - b(i, j, 2));
maxErr = std::max(maxErr, err_ij);
}
}
cout << "Max error: " << maxErr << " for coordinate " << c << endl;
}
// Sample a NURBS mesh to generate a first-order mesh.
void SampleNURBS(bool uniform, int nx, int ny, const Mesh &mesh,
const Array<int> &nks, const std::vector<Vector> &ugrid,
Array3D<real_t> &vpos)
{
const GridFunction *nodes = mesh.GetNodes();
const real_t hx = 1.0 / (real_t) nx;
const real_t hy = 1.0 / (real_t) ny;
const real_t hxks = 1.0 / (real_t) nks[0];
const real_t hyks = 1.0 / (real_t) nks[1];
Vector vertex;
IntegrationPoint ip;
ip.z = 1.0;
for (int i = 0; i <= nx; ++i)
{
const real_t xref = uniform ? i * hx : ugrid[0][i];
const int nurbsElem0 = std::min((int) (xref / hxks), nks[0] - 1);
const real_t ipx = (xref - (nurbsElem0 * hxks)) / hxks;
ip.x = ipx;
for (int j = 0; j <= ny; ++j)
{
const real_t yref = uniform ? j * hy : ugrid[1][j];
const int nurbsElem1 = std::min((int) (yref / hyks), nks[1] - 1);
const real_t ipy = (yref - (nurbsElem1 * hyks)) / hyks;
ip.y = ipy;
const int nurbsElem = nurbsElem0 + (nurbsElem1 * nks[0]);
nodes->GetVectorValue(nurbsElem, ip, vertex);
for (int k = 0; k < 3; ++k)
{
vpos(i, j, k) = vertex[k];
}
}
}
}
SurfaceInterpolator::SurfaceInterpolator(int num_elem_x, int num_elem_y,
int order) :
nx(num_elem_x), ny(num_elem_y), orderNURBS(order),
ncp(dim), nks(dim), ugrid(dim - 1)
{
ncp[0] = nx + 1;
ncp[1] = ny + 1;
ncp[2] = order + 1;
for (int i = 0; i < dim; ++i)
{
nks[i] = ncp[i] - order;
Vector intervals(nks[i]);
Array<int> continuity(nks[i] + 1);
intervals = 1.0 / (real_t) nks[i];
continuity = order - 1;
continuity[0] = -1;
continuity[nks[i]] = -1;
kv.emplace_back(order, intervals, continuity);
}
patch.reset(new NURBSPatch(&kv[0], &kv[1], &kv[2], dim + 1));
hx = 1.0 / (real_t) (ncp[0] - 1);
hy = 1.0 / (real_t) (ncp[1] - 1);
hz = 1.0 / (real_t) (ncp[2] - 1);
Vector xi_args;
Array<int> i_args;
for (int i = 0; i < 2; ++i)
{
kv[i].FindMaxima(i_args, xi_args, ugrid[i]);
}
}
void SurfaceInterpolator::CreateSurface(const Array3D<real_t> &input3D)
{
cmesh.clear();
for (int c = 0; c < dim; ++c) // Loop over coordinates
{
ComputeNURBS(c, input3D);
cmesh.emplace_back(mesh);
}
initial3D = input3D;
}
void SurfaceInterpolator::SampleSurface(int num_elem_x, int num_elem_y,
bool compareOriginal,
Array3D<real_t> &output3D)
{
Array3D<real_t> vpos(num_elem_x + 1, num_elem_y + 1, dim);
for (int c = 0; c < dim; ++c) // Loop over coordinates
{
SampleNURBS(true, num_elem_x, num_elem_y, cmesh[c], nks, ugrid, vpos);
if (compareOriginal)
{
SampleNURBS(false, num_elem_x, num_elem_y, cmesh[c], nks, ugrid, vpos);
CheckError(initial3D, vpos, c, nx, ny);
}
for (int i = 0; i <= num_elem_x; ++i)
{
for (int j = 0; j <= num_elem_y; ++j)
{
output3D(i,j,c) = vpos(i,j,2);
}
}
}
}
void SurfaceInterpolator::ComputeNURBS(int coordinate,
const Array3D<real_t> &input3D)
{
Array<Vector*> x;
for (int i = 0; i < dim; ++i) { x.Append(new Vector(ncp[0])); }
for (int k = 0; k < ncp[2]; ++k)
{
const real_t z = k * hz;
// For each horizontal slice (fixed k), interpolate a 2D surface by
// sweeping curve interpolations in each direction. See Algorithm A9.4 of
// "The NURBS Book" - 2nd ed - Piegl and Tiller.
// Resize for sweep in first direction
for (int i = 0; i < dim; ++i) { x[i]->SetSize(ncp[0]); }
// Sweep in the first direction
for (int j = 0; j < ncp[1]; ++j)
{
for (int i = 0; i < ncp[0]; i++)
{
(*x[0])[i] = ugrid[0][i];
(*x[1])[i] = ugrid[1][j];
const real_t s_ij = input3D(i, j, coordinate);
(*x[2])[i] = -1.0 + z + s_ij;
}
const bool reuse_factorization = j > 0;
kv[0].FindInterpolant(x, reuse_factorization);
for (int i = 0; i < ncp[0]; i++)
{
(*patch)(i,j,k,0) = (*x[0])[i];
(*patch)(i,j,k,1) = (*x[1])[i];
(*patch)(i,j,k,2) = (*x[2])[i];
(*patch)(i,j,k,3) = 1.0; // weight
}
}
// Resize for sweep in second direction
for (int i = 0; i < dim; ++i) { x[i]->SetSize(ncp[1]); }
// Do another sweep in the second direction
for (int i = 0; i < ncp[0]; i++)
{
for (int j = 0; j < ncp[1]; ++j)
{
(*x[0])[j] = (*patch)(i,j,k,0);
(*x[1])[j] = (*patch)(i,j,k,1);
(*x[2])[j] = (*patch)(i,j,k,2);
}
const bool reuse_factorization = i > 0;
kv[1].FindInterpolant(x, reuse_factorization);
for (int j = 0; j < ncp[1]; ++j)
{
(*patch)(i,j,k,0) = (*x[0])[j];
(*patch)(i,j,k,1) = (*x[1])[j];
(*patch)(i,j,k,2) = (*x[2])[j];
}
}
}
for (auto p : x) { delete p; }
Array<const NURBSPatch*> patches(1);
patches[0] = patch.get();
Mesh patch_topology = Mesh::MakeCartesian3D(1, 1, 1, Element::HEXAHEDRON);
NURBSExtension nurbsExt(&patch_topology, patches);
mesh = Mesh(nurbsExt);
}
void SurfaceInterpolator::WriteNURBSMesh(const std::string &basename,
bool visualization,
int x, int y, int w, int h)
{
GridFunction *nodes = cmesh[0].GetNodes();
NURBSPatch patch2D(&kv[0], &kv[1], dim);
Array<const NURBSPatch*> patches(1);
patches[0] = &patch2D;
Mesh patch_topology = Mesh::MakeCartesian2D(1, 1, Element::QUADRILATERAL);
Array<int> dofs;
cmesh[0].NURBSext->GetPatchDofs(0, dofs);
MFEM_VERIFY(dofs.Size() == (nx + 1) * (ny + 1) * (orderNURBS + 1), "");
for (int j = 0; j < ncp[1]; ++j)
{
for (int i = 0; i < ncp[0]; i++)
{
const int dof = dofs[i + (ncp[0] * (j + (ncp[1] * orderNURBS)))];
for (int k = 0; k < 2; ++k) { patch2D(i,j,k) = (*nodes)[dim*dof + k]; }
patch2D(i,j,2) = 1.0; // weight
}
}
NURBSExtension nurbsExt(&patch_topology, patches);
Mesh mesh2D(nurbsExt);
FiniteElementCollection *fec = nodes->OwnFEC();
FiniteElementSpace fespace(&mesh2D, fec, dim, Ordering::byVDIM);
GridFunction nodes2D(&fespace);
const int n = mesh2D.GetNodes()->Size() / (dim - 1);
MFEM_VERIFY((dim - 1) * n == mesh2D.GetNodes()->Size(), "");
MFEM_VERIFY(dim * n == nodes2D.Size(), "");
Array<int> dofs2D;
mesh2D.NURBSext->GetPatchDofs(0, dofs2D);
for (int k = 0; k < dim; ++k)
{
const GridFunction &nodes_k = *cmesh[k].GetNodes();
for (int j = 0; j < ncp[1]; ++j)
{
for (int i = 0; i < ncp[0]; i++)
{
const int dof = dofs[i + (ncp[0] * (j + (ncp[1] * orderNURBS)))];
const int dof2D = dofs2D[i + (ncp[0] * j)];
nodes2D[(dim*dof2D) + k] = nodes_k[dim*dof + 2];
}
}
}
// Make mesh2D into a surface mesh with nodes given by nodes2D
mesh2D.NewNodes(nodes2D);
ofstream mesh_ofs(basename + ".mesh");
mesh_ofs.precision(8);
mesh2D.Print(mesh_ofs);
if (visualization)
{
char vishost[] = "localhost";
constexpr int visport = 19916;
socketstream sol_sock(vishost, visport);
sol_sock.precision(8);
sol_sock << "mesh\n" << mesh2D
<< "window_title '" << basename << "'"
<< "window_geometry "
<< x << " " << y << " " << w << " " << h << "\n"
<< "keys PPPPPPPPAattttt******\n"
<< flush;
}
}
+2 -5
View File
@@ -105,11 +105,8 @@ MFEM_PERF_CXXFLAGS_xlc = -mcpu=native
# - Clang extra options:
ifeq ($(MFEM_MACHINE),riscv64)
MFEM_PERF_CXXFLAGS_clang += -march=rv64gc
else ifneq (,$(findstring ppc,$(MFEM_MACHINE)))
MFEM_PERF_CXXFLAGS_clang += -mcpu=native -mtune=native
else ifeq ($(MFEM_MACHINE),arm64)
MFEM_PERF_CXXFLAGS_clang += -mcpu=native -mtune=native
else
else ifneq ($(MFEM_MACHINE),arm64)
# -march=native is unavailable on clang/ARM64 as of 05/2021: support could be added later.
MFEM_PERF_CXXFLAGS_clang += -march=native
endif
MFEM_PERF_CXXFLAGS_clang += $(PEDANTIC_FLAG) -Wall
+42 -45
View File
@@ -52,13 +52,12 @@
// (respectively 0), essential (respectively natural) boundary condition
// will be imposed on boundary with the i-th attribute.
#include <fstream>
#include <iostream>
#include <functional>
#include "mfem.hpp"
#include "bramble_pasciak.hpp"
#include "div_free_solver.hpp"
#include <fstream>
#include <iostream>
#include <memory>
using namespace std;
using namespace mfem;
@@ -84,54 +83,48 @@ real_t natural_bc(const Vector & x);
D: subset of the boundary where natural boundary condition is imposed. */
class DarcyProblem
{
OperatorPtr M_, B_;
Vector rhs_, ess_data_;
ParGridFunction u_, p_;
OperatorPtr M_;
OperatorPtr B_;
Vector rhs_;
Vector ess_data_;
ParGridFunction u_;
ParGridFunction p_;
ParMesh mesh_;
DFSSpaces dfs_spaces_;
std::function<bool (int)> refine_fn = [&](int num_refs)
{
for (int l = 0; l < num_refs; l++)
{
mesh_.UniformRefinement();
dfs_spaces_.CollectDFSData();
}
return true;
};
const bool dfs_refine_;
ParBilinearForm mVarf_;
ParMixedBilinearForm bVarf_;
ParBilinearForm *mVarf_;
ParMixedBilinearForm *bVarf_;
VectorFunctionCoefficient ucoeff_;
FunctionCoefficient pcoeff_;
DFSSpaces dfs_spaces_;
PWConstCoefficient mass_coeff;
const IntegrationRule *irs_[Geometry::NumGeom];
public:
DarcyProblem(Mesh &mesh, int num_refines, int order, const char *coef_file,
Array<int> &ess_bdr, DFSParameters param);
const HypreParMatrix& GetM() const { return *M_.As<HypreParMatrix>(); }
const HypreParMatrix& GetB() const { return *B_.As<HypreParMatrix>(); }
HypreParMatrix& GetM() { return *M_.As<HypreParMatrix>(); }
HypreParMatrix& GetB() { return *B_.As<HypreParMatrix>(); }
const Vector& GetRHS() { return rhs_; }
const Vector& GetEssentialBC() { return ess_data_; }
const DFSData& GetDFSData() const { return dfs_spaces_.GetDFSData(); }
void ShowError(const Vector &sol, bool verbose);
void VisualizeSolution(const Vector &sol, std::string tag, int visport = 19916);
ParBilinearForm& GetMform() { return mVarf_; }
ParMixedBilinearForm& GetBform() { return bVarf_; }
ParBilinearForm* GetMform() const { return mVarf_; }
ParMixedBilinearForm* GetBform() const { return bVarf_; }
};
DarcyProblem::DarcyProblem(Mesh &mesh, int num_refs, int order,
const char *coef_file, Array<int> &ess_bdr,
DFSParameters dfs_param)
: mesh_(MPI_COMM_WORLD, mesh),
dfs_spaces_(order, num_refs, &mesh_, ess_bdr, dfs_param),
dfs_refine_(refine_fn(num_refs)),
mVarf_(dfs_spaces_.GetHdivFES()),
bVarf_(dfs_spaces_.GetHdivFES(), dfs_spaces_.GetL2FES()),
ucoeff_(mesh.Dimension(), u_exact),
pcoeff_(p_exact),
: mesh_(MPI_COMM_WORLD, mesh), ucoeff_(mesh.Dimension(), u_exact),
pcoeff_(p_exact), dfs_spaces_(order, num_refs, &mesh_, ess_bdr, dfs_param),
mass_coeff()
{
for (int l = 0; l < num_refs; l++)
{
mesh_.UniformRefinement();
dfs_spaces_.CollectDFSData();
}
Vector coef_vector(mesh.GetNE());
coef_vector = 1.0;
if (std::strcmp(coef_file, ""))
@@ -160,20 +153,24 @@ DarcyProblem::DarcyProblem(Mesh &mesh, int num_refs, int order,
gform.AddDomainIntegrator(new DomainLFIntegrator(gcoeff));
gform.Assemble();
mVarf_.AddDomainIntegrator(new VectorFEMassIntegrator(mass_coeff));
mVarf_.ComputeElementMatrices();
mVarf_.Assemble();
mVarf_.EliminateEssentialBC(ess_bdr, u_, fform);
mVarf_ = new ParBilinearForm(dfs_spaces_.GetHdivFES());
bVarf_ = new ParMixedBilinearForm(dfs_spaces_.GetHdivFES(),
dfs_spaces_.GetL2FES());
mVarf_.Finalize();
M_.Reset(mVarf_.ParallelAssemble());
mVarf_->AddDomainIntegrator(new VectorFEMassIntegrator(mass_coeff));
mVarf_->ComputeElementMatrices();
mVarf_->Assemble();
mVarf_->EliminateEssentialBC(ess_bdr, u_, fform);
bVarf_.AddDomainIntegrator(new VectorFEDivergenceIntegrator);
bVarf_.Assemble();
bVarf_.SpMat() *= -1.0;
bVarf_.EliminateTrialEssentialBC(ess_bdr, u_, gform);
bVarf_.Finalize();
B_.Reset(bVarf_.ParallelAssemble());
mVarf_->Finalize();
M_.Reset(mVarf_->ParallelAssemble());
bVarf_->AddDomainIntegrator(new VectorFEDivergenceIntegrator);
bVarf_->Assemble();
bVarf_->SpMat() *= -1.0;
bVarf_->EliminateTrialEssentialBC(ess_bdr, u_, gform);
bVarf_->Finalize();
B_.Reset(bVarf_->ParallelAssemble());
rhs_.SetSize(M_->NumRows() + B_->NumRows());
Vector rhs_block0(rhs_.GetData(), M_->NumRows());
@@ -344,8 +341,8 @@ int main(int argc, char *argv[])
// Generate components of the saddle point problem
DarcyProblem darcy(*mesh, par_ref_levels, order, coef_file, ess_bdr, param);
const HypreParMatrix &M = darcy.GetM();
const HypreParMatrix &B = darcy.GetB();
HypreParMatrix& M = darcy.GetM();
HypreParMatrix& B = darcy.GetB();
const DFSData& DFS_data = darcy.GetDFSData();
delete mesh;
+15 -12
View File
@@ -14,27 +14,29 @@
namespace mfem
{
BlockFESpaceOperator::BlockFESpaceOperator(const FESVector &fespaces):
BlockFESpaceOperator::BlockFESpaceOperator(const
std::vector<const FiniteElementSpace*> &fespaces):
Operator(GetHeight(fespaces)),
offsets(GetBlockOffsets(fespaces)),
prolongColOffsets(GetProColBlockOffsets(fespaces)),
restrictRowOffsets(GetResRowBlockOffsets(fespaces)),
A(offsets),
prolongation(offsets, prolongColOffsets),
prolongation(offsets,prolongColOffsets),
restriction(restrictRowOffsets, offsets)
{
for (size_t i = 0; i <fespaces.size(); i++)
{
// Since const_cast is required here, be sure to avoid using
// BlockOperator::GetBlock on restriction or prolongation.
auto prolongation_matrix = fespaces[i]->GetProlongationMatrix();
auto restriction_matrix = fespaces[i]->GetRestrictionOperator();
prolongation.SetDiagonalBlock(i, const_cast<Operator *>(prolongation_matrix));
restriction.SetDiagonalBlock(i, const_cast<Operator *>(restriction_matrix));
prolongation.SetDiagonalBlock(i,
const_cast<Operator *>(fespaces[i]->GetProlongationMatrix()));
restriction.SetDiagonalBlock(i,
const_cast<Operator *>(fespaces[i]->GetRestrictionOperator()));
}
}
int BlockFESpaceOperator::GetHeight(const FESVector &fespaces)
int BlockFESpaceOperator::GetHeight(const std::vector<const FiniteElementSpace*>
&fespaces)
{
int height = 0;
for (size_t i = 0; i < fespaces.size(); i++)
@@ -44,7 +46,8 @@ int BlockFESpaceOperator::GetHeight(const FESVector &fespaces)
return height;
}
Array<int> BlockFESpaceOperator::GetBlockOffsets(const FESVector &fespaces)
Array<int> BlockFESpaceOperator::GetBlockOffsets(const
std::vector<const FiniteElementSpace*> &fespaces)
{
Array<int> offsets(fespaces.size()+1);
offsets[0] = 0;
@@ -57,8 +60,8 @@ Array<int> BlockFESpaceOperator::GetBlockOffsets(const FESVector &fespaces)
return offsets;
}
Array<int> BlockFESpaceOperator::GetProColBlockOffsets(const FESVector
&fespaces)
Array<int> BlockFESpaceOperator::GetProColBlockOffsets(const
std::vector<const FiniteElementSpace*> &fespaces)
{
Array<int> offsets(fespaces.size()+1);
offsets[0] = 0;
@@ -80,8 +83,8 @@ Array<int> BlockFESpaceOperator::GetProColBlockOffsets(const FESVector
return offsets;
}
Array<int> BlockFESpaceOperator::GetResRowBlockOffsets(const FESVector
&fespaces)
Array<int> BlockFESpaceOperator::GetResRowBlockOffsets(const
std::vector<const FiniteElementSpace*> &fespaces)
{
Array<int> offsets(fespaces.size()+1);
std::cout << "fespaces.size() = " << fespaces.size() << std::endl;
+15 -10
View File
@@ -25,8 +25,7 @@ namespace mfem
/// L-Vectors. For example, a block may be a BilinearForm.
class BlockFESpaceOperator : public Operator
{
using FESVector = std::vector<const FiniteElementSpace*>;
private:
/// Offsets for the square "A" operator.
Array<int> offsets;
/// Column offsets for the prolongation operator.
@@ -40,27 +39,33 @@ class BlockFESpaceOperator : public Operator
/// Maps true dofs of each block to local dofs.
BlockOperator restriction;
/// Computes height for parent operator.
static int GetHeight(const FESVector &fespaces);
static int GetHeight(const std::vector<const FiniteElementSpace*>
&fespaces);
/// Computes offsets for A BlockOperator.
static Array<int> GetBlockOffsets(const FESVector &fespaces);
static Array<int> GetBlockOffsets(const std::vector<const FiniteElementSpace*>
&fespaces);
/// Computes col_offsets for prolongation operator.
static Array<int> GetProColBlockOffsets(const FESVector &fespaces);
static Array<int> GetProColBlockOffsets(const
std::vector<const FiniteElementSpace*> &fespaces);
/// Computes row_offsets for restriction operator.
static Array<int> GetResRowBlockOffsets(const FESVector &fespaces);
static Array<int> GetResRowBlockOffsets(const
std::vector<const FiniteElementSpace*> &fespaces);
public:
/// @brief Constructor for BlockFESpaceOperator.
/// @param[in] fespaces Finite element spaces for diagonal blocks. Spaces are not owned.
BlockFESpaceOperator(const FESVector &fespaces);
BlockFESpaceOperator(const std::vector<const FiniteElementSpace*> &fespaces);
const Operator* GetProlongation () const override;
const Operator* GetRestriction () const override;
void Mult(const Vector &x, Vector &y) const override {A.Mult(x,y);};
/// @brief Wraps BlockOperator::SetBlock. Eventually would like this class to inherit
/// from BlockOperator instead, but can't easily due to ownership of offset data
/// in BlockOperator being by reference.
void SetBlock(int iRow, int iCol, Operator *op, real_t c = 1.0) { A.SetBlock(iRow, iCol, op, c); };
void SetBlock( int iRow,
int iCol,
Operator * op,
real_t c = 1.0) {A.SetBlock(iRow, iCol, op, c);};
};
} // namespace mfem
#endif // MFEM_BLOCK_FESPACE_OPERATOR
#endif
+49 -41
View File
@@ -9,65 +9,70 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "bramble_pasciak.hpp"
using namespace std;
namespace mfem::blocksolvers
namespace mfem
{
namespace blocksolvers
{
/// Bramble-Pasciak Solver
BramblePasciakSolver::BramblePasciakSolver(ParBilinearForm &mVarf,
ParMixedBilinearForm &bVarf,
const BPSParameters &param)
: DarcySolver(mVarf.ParFESpace()->GetTrueVSize(),
bVarf.TestFESpace()->GetTrueVSize())
BramblePasciakSolver::BramblePasciakSolver(
ParBilinearForm *mVarf,
ParMixedBilinearForm *bVarf,
const BPSParameters &param)
: DarcySolver(mVarf->ParFESpace()->GetTrueVSize(),
bVarf->TestFESpace()->GetTrueVSize())
{
M_.reset(mVarf.ParallelAssemble());
B_.reset(bVarf.ParallelAssemble());
Q_.reset(ConstructMassPreconditioner(mVarf, param.q_scaling));
M_.reset(mVarf->ParallelAssemble());
B_.reset(bVarf->ParallelAssemble());
Q_.reset(ConstructMassPreconditioner(*mVarf, param.q_scaling));
Vector diagM;
M_->GetDiag(diagM);
std::unique_ptr<HypreParMatrix> invDBt(B_->Transpose());
auto BT = B_->Transpose();
auto invDBt = new HypreParMatrix(*BT);
invDBt->InvScaleRows(diagM);
S_.reset(ParMult(B_.get(), invDBt.get(), true));
auto S = ParMult(B_.get(), invDBt);
M0_.Reset(new HypreDiagScale(*M_));
M1_.Reset(new HypreBoomerAMG(*S_));
M1_.Reset(new HypreBoomerAMG(*S));
M1_.As<HypreBoomerAMG>()->SetPrintLevel(0);
Init(*M_, *B_, *Q_, *M0_.As<Solver>(), *M1_.As<Solver>(), param);
}
BramblePasciakSolver::BramblePasciakSolver(HypreParMatrix &M,
HypreParMatrix &B,
HypreParMatrix &Q,
Solver &M0, Solver &M1,
const BPSParameters &param)
BramblePasciakSolver::BramblePasciakSolver(
HypreParMatrix &M, HypreParMatrix &B, HypreParMatrix &Q,
Solver &M0, Solver &M1,
const BPSParameters &param)
: DarcySolver(M.NumRows(), B.NumRows())
{
Init(M, B, Q, M0, M1, param);
}
void BramblePasciakSolver::Init(HypreParMatrix &M,
HypreParMatrix &B,
HypreParMatrix &Q,
Solver &M0, Solver &M1,
const BPSParameters &param)
void BramblePasciakSolver::Init(
HypreParMatrix &M, HypreParMatrix &B, HypreParMatrix &Q,
Solver &M0, Solver &M1,
const BPSParameters &param)
{
Bt_ = std::make_unique<TransposeOperator>(&B);
auto Bt = new TransposeOperator(&B);
auto invQ = new HypreDiagScale(Q);
use_bpcg = param.use_bpcg;
if (use_bpcg)
{
oop_ = std::make_unique<BlockOperator>(offsets_);
oop_ = new BlockOperator(offsets_);
oop_->owns_blocks = false;
oop_->SetBlock(0, 0, &M);
oop_->SetBlock(0, 1, Bt_.get());
oop_->SetBlock(0, 1, Bt);
oop_->SetBlock(1, 0, &B);
// cpc_ unused in bpcg
auto temp_cpc = new BlockDiagonalPreconditioner(offsets_);
temp_cpc->owns_blocks = true;
temp_cpc->SetDiagonalBlock(0, invQ);
temp_cpc->SetDiagonalBlock(1, &M1);
// tri(1,0) = B M0 = B invQ
@@ -76,48 +81,51 @@ void BramblePasciakSolver::Init(HypreParMatrix &M,
auto BinvQ = new ProductOperator(&B, invQ, false, false);
// tri
auto temp_tri = new BlockOperator(offsets_);
temp_tri->owns_blocks = true;
temp_tri->SetBlock(0, 0, id_m);
temp_tri->SetBlock(1, 1, id_b, -1.0);
temp_tri->SetBlock(1, 0, BinvQ);
temp_tri->owns_blocks = 1;
ppc_ = std::make_unique<ProductOperator>(temp_cpc, temp_tri, true, true);
ppc_ = new ProductOperator(temp_cpc, temp_tri, true, true);
ipc_ = std::make_unique<BlockOperator>(offsets_);
ipc_ = new BlockOperator(offsets_);
ipc_->owns_blocks = false;
ipc_->SetDiagonalBlock(0, invQ);
ipc_->owns_blocks = 1;
// bpcg
solver_ = std::make_unique<BPCGSolver>(M.GetComm(), ipc_.get(), ppc_.get());
solver_.reset(new BPCGSolver(M.GetComm(), *ipc_, *ppc_));
solver_->SetOperator(*oop_);
}
else
{
// oop_ unused in cg
auto temp_oop = new BlockOperator(offsets_);
temp_oop->owns_blocks = false;
temp_oop->SetBlock(0, 0, &M);
temp_oop->SetBlock(0, 1, Bt_.get());
temp_oop->SetBlock(0, 1, Bt);
temp_oop->SetBlock(1, 0, &B);
// ipc_ unused in cg
auto temp_ipc = new BlockOperator(offsets_);
temp_ipc->owns_blocks = false;
temp_ipc->SetDiagonalBlock(0, invQ);
temp_ipc->owns_blocks = 1;
// temp_AN = temp_oop * temp_ipc
auto temp_AN = new ProductOperator(temp_oop, temp_ipc, true, true);
// Required for updating the RHS
auto id = new IdentityOperator(M.NumRows()+B.NumRows());
map_ = std::make_unique<SumOperator>(temp_AN, 1.0, id, -1.0, true, true);
mop_ = std::make_unique<ProductOperator>(map_.get(), temp_oop, false, false);
map_ = new SumOperator(temp_AN, 1.0, id, -1.0, true, true);
cpc_ = std::make_unique<BlockDiagonalPreconditioner>(offsets_);
mop_ = new ProductOperator(map_, temp_oop, false, true);
cpc_ = new BlockDiagonalPreconditioner(offsets_);
cpc_->owns_blocks = true;
cpc_->SetDiagonalBlock(0, &M0);
cpc_->SetDiagonalBlock(1, &M1);
// (P)CG
solver_ = std::make_unique<CGSolver>(M.GetComm());
solver_.reset(new CGSolver(M.GetComm()));
solver_->SetOperator(*mop_);
solver_->SetPreconditioner(*cpc_);
}
@@ -125,7 +133,7 @@ void BramblePasciakSolver::Init(HypreParMatrix &M,
}
HypreParMatrix *BramblePasciakSolver::ConstructMassPreconditioner(
const ParBilinearForm &mVarf, real_t q_scaling)
ParBilinearForm &mVarf, real_t q_scaling)
{
MFEM_ASSERT((q_scaling > 0.0) && (q_scaling < 1.0),
"Invalid Q-scaling factor: q_scaling = " << q_scaling );
@@ -159,7 +167,7 @@ HypreParMatrix *BramblePasciakSolver::ConstructMassPreconditioner(
Vector x(M_i.Height()), Mx(M_i.Height()), diff(M_i.Height());
real_t eval_prev = 0.0;
int iter = 0;
x.Randomize(static_cast<int>(696383552LL+779345LL*i));
x.Randomize(696383552+779345*i);
#if defined(MFEM_USE_DOUBLE)
const real_t rel_tol = 1e-12;
#elif defined(MFEM_USE_SINGLE)
@@ -392,5 +400,5 @@ void BPCGSolver::Mult(const Vector &b, Vector &x) const
final_norm = sqrt(delta);
Monitor(final_iter, final_norm, r, x, true);
}
} // namespace mfem::blocksolvers
} // namespace blocksolvers
} // namespace mfem
+27 -19
View File
@@ -49,7 +49,9 @@
#include "darcy_solver.hpp"
#include <memory>
namespace mfem::blocksolvers
namespace mfem
{
namespace blocksolvers
{
/// Parameters for the BramblePasciakSolver method
@@ -68,11 +70,11 @@ protected:
void UpdateVectors();
public:
BPCGSolver(const Operator *ipc, const Operator *ppc): iprec(ipc), pprec(ppc) {}
BPCGSolver(const Operator &ipc, const Operator &ppc) { pprec = &ppc; iprec = &ipc; }
#ifdef MFEM_USE_MPI
BPCGSolver(MPI_Comm comm_, const Operator *ipc, const Operator *ppc)
: IterativeSolver(comm_), iprec(ipc), pprec(ppc) { }
BPCGSolver(MPI_Comm comm_, const Operator &ipc, const Operator &ppc)
: IterativeSolver(comm_) { pprec = &ppc; iprec = &ipc; }
#endif
void SetOperator(const Operator &op) override
@@ -81,9 +83,11 @@ public:
void SetPreconditioner(Solver &pc) override
{ if (Mpi::Root()) { MFEM_WARNING("SetPreconditioner has no effect on BPCGSolver.\n"); } }
virtual void SetIncompletePreconditioner(const Operator *ipc) { iprec = ipc; }
virtual void SetIncompletePreconditioner(const Operator &ipc)
{ iprec = &ipc; }
virtual void SetParticularPreconditioner(const Operator *ppc) { pprec = ppc; }
virtual void SetParticularPreconditioner(const Operator &ppc)
{ pprec = &ppc; }
void Mult(const Vector &b, Vector &x) const override;
};
@@ -112,20 +116,23 @@ public:
1. P. Vassilevski, Multilevel Block Factorization Preconditioners (Appendix
F.3), Springer, 2008.
2. J. Bramble and J. Pasciak. A Preconditioning Technique for Indefinite
2. J. Bramble and J. Pasciak. A Preconditioning Technique for Indefinite
Systems Resulting From Mixed Approximations of Elliptic Problems,
Mathematics of Computation, 50:1-17, 1988. */
class BramblePasciakSolver : public DarcySolver
{
mutable bool use_bpcg;
std::unique_ptr<IterativeSolver> solver_;
std::unique_ptr<BlockOperator> oop_, ipc_;
std::unique_ptr<ProductOperator> mop_, ppc_;
std::unique_ptr<SumOperator> map_;
std::unique_ptr<BlockDiagonalPreconditioner> cpc_;
std::unique_ptr<HypreParMatrix> M_, B_, Q_, S_;
std::unique_ptr<TransposeOperator> Bt_;
OperatorPtr M0_, M1_;
BlockOperator *oop_, *ipc_;
ProductOperator *mop_;
SumOperator *map_;
ProductOperator *ppc_;
BlockDiagonalPreconditioner *cpc_;
std::unique_ptr<HypreParMatrix> M_;
std::unique_ptr<HypreParMatrix> B_;
std::unique_ptr<HypreParMatrix> Q_;
OperatorPtr M0_;
OperatorPtr M1_;
Array<int> ess_zero_dofs_;
void Init(HypreParMatrix &M, HypreParMatrix &B,
@@ -135,8 +142,8 @@ class BramblePasciakSolver : public DarcySolver
public:
/// System and mass preconditioner are constructed from bilinear forms
BramblePasciakSolver(
ParBilinearForm &mVarf,
ParMixedBilinearForm &bVarf,
ParBilinearForm *mVarf,
ParMixedBilinearForm *bVarf,
const BPSParameters &param);
/// System and mass preconditioner are user-provided
@@ -151,8 +158,8 @@ public:
element T:
M_T x_T = lambda_T diag(M_T) x_T.
We set Q_T = alpha * min(lambda_T) * diag(M_T), 0 < alpha < 1. */
static HypreParMatrix *ConstructMassPreconditioner(const ParBilinearForm &mVarf,
const real_t alpha = 0.5);
static HypreParMatrix *ConstructMassPreconditioner(ParBilinearForm &mVarf,
real_t alpha = 0.5);
void Mult(const Vector &x, Vector &y) const override;
void SetOperator(const Operator &op) override { }
@@ -160,6 +167,7 @@ public:
int GetNumIterations() const override { return solver_->GetNumIterations(); }
};
} // namespace mfem::blocksolvers
} // namespace blocksolvers
} // namespace mfem
#endif // MFEM_BP_SOLVER_HPP
+6 -5
View File
@@ -13,9 +13,10 @@
using namespace std;
namespace mfem::blocksolvers
namespace mfem
{
namespace blocksolvers
{
void SetOptions(IterativeSolver& solver, const IterSolveParameters& param)
{
solver.SetPrintLevel(param.print_level);
@@ -48,7 +49,7 @@ BDPMinresSolver::BDPMinresSolver(const HypreParMatrix& M,
prec_.SetDiagonalBlock(0, new HypreDiagScale(M));
prec_.SetDiagonalBlock(1, new HypreBoomerAMG(*S_.As<HypreParMatrix>()));
static_cast<HypreBoomerAMG&>(prec_.GetDiagonalBlock(1)).SetPrintLevel(0);
prec_.owns_blocks = 1;
prec_.owns_blocks = true;
SetOptions(solver_, param);
solver_.SetOperator(op_);
@@ -60,5 +61,5 @@ void BDPMinresSolver::Mult(const Vector & x, Vector & y) const
solver_.Mult(x, y);
for (int dof : ess_zero_dofs_) { y[dof] = 0.0; }
}
} // namespace mfem::blocksolvers
} // namespace blocksolvers
} // namespace mfem
+9 -4
View File
@@ -13,10 +13,13 @@
#define MFEM_DARCY_SOLVER_HPP
#include "mfem.hpp"
#include <memory>
#include <vector>
namespace mfem::blocksolvers
namespace mfem
{
namespace blocksolvers
{
struct IterSolveParameters
{
int print_level = 0;
@@ -29,6 +32,8 @@ struct IterSolveParameters
real_t rel_tol = 1e-5;
#else
#error "Only single and double precision are supported!"
real_t abs_tol = 1e-12;
real_t rel_tol = 1e-9;
#endif
};
@@ -63,7 +68,7 @@ public:
void SetEssZeroDofs(const Array<int>& dofs) { dofs.Copy(ess_zero_dofs_); }
int GetNumIterations() const override { return solver_.GetNumIterations(); }
};
} // namespace mfem::blocksolvers
} // namespace blocksolvers
} // namespace mfem
#endif // MFEM_DARCY_SOLVER_HPP
+107 -106
View File
@@ -13,16 +13,16 @@
using namespace std;
namespace mfem::blocksolvers
namespace mfem
{
static HypreParMatrix* TwoStepsRAP(const HypreParMatrix *Rt,
const HypreParMatrix *A,
const HypreParMatrix *P)
namespace blocksolvers
{
OperatorPtr R(Rt->Transpose());
OperatorPtr RA(ParMult(R.As<HypreParMatrix>(), A));
return ParMult(RA.As<HypreParMatrix>(), P, true);
HypreParMatrix* TwoStepsRAP(const HypreParMatrix& Rt, const HypreParMatrix& A,
const HypreParMatrix& P)
{
OperatorPtr R(Rt.Transpose());
OperatorPtr RA(ParMult(R.As<HypreParMatrix>(), &A));
return ParMult(RA.As<HypreParMatrix>(), &P, true);
}
void GetRowColumnsRef(const SparseMatrix& A, int row, Array<int>& cols)
@@ -59,36 +59,34 @@ DFSSpaces::DFSSpaces(int order, int num_refine, ParMesh *mesh,
if (mesh->Dimension() == 3)
{
hcurl_fec_ = std::make_unique<ND_FECollection>(order+1, mesh->Dimension());
hcurl_fec_.reset(new ND_FECollection(order+1, mesh->Dimension()));
}
else
{
hcurl_fec_ = std::make_unique<H1_FECollection>(order+1, mesh->Dimension());
hcurl_fec_.reset(new H1_FECollection(order+1, mesh->Dimension()));
}
all_bdr_attr_.SetSize(ess_attr.Size(), 1);
hdiv_fes_ = std::make_unique<ParFiniteElementSpace>(mesh, &hdiv_fec_);
l2_fes_ = std::make_unique<ParFiniteElementSpace>(mesh, &l2_fec_);
coarse_hdiv_fes_ = std::make_unique<ParFiniteElementSpace>(*hdiv_fes_);
coarse_l2_fes_ = std::make_unique<ParFiniteElementSpace>(*l2_fes_);
l2_0_fes_ = std::make_unique<ParFiniteElementSpace>(mesh, &l2_0_fec_);
hdiv_fes_.reset(new ParFiniteElementSpace(mesh, &hdiv_fec_));
l2_fes_.reset(new ParFiniteElementSpace(mesh, &l2_fec_));
coarse_hdiv_fes_.reset(new ParFiniteElementSpace(*hdiv_fes_));
coarse_l2_fes_.reset(new ParFiniteElementSpace(*l2_fes_));
l2_0_fes_.reset(new ParFiniteElementSpace(mesh, &l2_0_fec_));
l2_0_fes_->SetUpdateOperatorType(Operator::MFEM_SPARSEMAT);
el_l2dof_.reserve(num_refine+1);
el_l2dof_.push_back(ElemToDof(*coarse_l2_fes_));
data_.agg_hdivdof.resize(num_refine);
data_.agg_l2dof.resize(num_refine);
data_.P_hdiv.resize(num_refine);
data_.P_l2.resize(num_refine);
data_.P_hdiv.resize(num_refine, OperatorPtr(Operator::Hypre_ParCSR));
data_.P_l2.resize(num_refine, OperatorPtr(Operator::Hypre_ParCSR));
data_.Q_l2.resize(num_refine);
hdiv_fes_->GetEssentialTrueDofs(ess_attr, data_.coarsest_ess_hdivdofs);
data_.C.resize(num_refine+1);
data_.Ae.resize(num_refine+1);
hcurl_fes_ = std::make_unique<ParFiniteElementSpace>(mesh, hcurl_fec_.get());
coarse_hcurl_fes_ = std::make_unique<ParFiniteElementSpace>(*hcurl_fes_);
data_.P_hcurl.resize(num_refine);
hcurl_fes_.reset(new ParFiniteElementSpace(mesh, hcurl_fec_.get()));
coarse_hcurl_fes_.reset(new ParFiniteElementSpace(*hcurl_fes_));
data_.P_hcurl.resize(num_refine, OperatorPtr(Operator::Hypre_ParCSR));
}
SparseMatrix* AggToInteriorDof(const Array<int>& bdr_truedofs,
@@ -106,8 +104,8 @@ SparseMatrix* AggToInteriorDof(const Array<int>& bdr_truedofs,
agg_tdof_T.As<HypreParMatrix>()->GetDiag(tdof_agg);
agg_tdof_T.As<HypreParMatrix>()->GetOffd(is_shared, trash);
int *I = new int[tdof_agg.NumRows()+1]();
int *J = new int[tdof_agg.NumNonZeroElems()];
int * I = new int [tdof_agg.NumRows()+1]();
int * J = new int[tdof_agg.NumNonZeroElems()];
Array<int> is_bdr;
FiniteElementSpace::ListToMarker(bdr_truedofs, tdof_agg.NumRows(), is_bdr);
@@ -121,7 +119,7 @@ SparseMatrix* AggToInteriorDof(const Array<int>& bdr_truedofs,
J[counter++] = tdof_agg.GetRowColumns(i)[0];
}
auto *D = new real_t[I[tdof_agg.NumRows()]];
real_t * D = new real_t[I[tdof_agg.NumRows()]];
std::fill_n(D, I[tdof_agg.NumRows()], 1.0);
SparseMatrix intdof_agg(I, J, D, tdof_agg.NumRows(), tdof_agg.NumCols());
@@ -148,21 +146,20 @@ void DFSSpaces::MakeDofRelationTables(int level)
void DFSSpaces::CollectDFSData()
{
auto GetP = [&](std::unique_ptr<OperatorPtr> &P,
std::unique_ptr<ParFiniteElementSpace> &cfes,
ParFiniteElementSpace& fes, const bool remove_zero)
auto GetP = [this](OperatorPtr& P, unique_ptr<ParFiniteElementSpace>& cfes,
ParFiniteElementSpace& fes, bool remove_zero)
{
fes.Update();
auto T = new OperatorHandle(Operator::Hypre_ParCSR);
fes.GetTrueTransferOperator(*cfes, *T);
P.reset(T);
if (remove_zero) { P->As<HypreParMatrix>()->DropSmallEntries(1e-16); }
fes.GetTrueTransferOperator(*cfes, P);
if (remove_zero)
{
P.As<HypreParMatrix>()->DropSmallEntries(1e-16);
}
(level_ < (int)data_.P_l2.size()-1) ? cfes->Update() : cfes.reset();
};
GetP(data_.P_hdiv[level_], coarse_hdiv_fes_, *hdiv_fes_, true);
GetP(data_.P_l2[level_], coarse_l2_fes_, *l2_fes_, false);
MakeDofRelationTables(level_);
GetP(data_.P_hcurl[level_], coarse_hcurl_fes_, *hcurl_fes_, true);
@@ -174,9 +171,7 @@ void DFSSpaces::CollectDFSData()
data_.C[level_+1].Reset(curl.ParallelAssemble());
mfem::Array<int> ess_hcurl_tdof;
hcurl_fes_->GetEssentialTrueDofs(ess_bdr_attr_, ess_hcurl_tdof);
data_.Ae[level_+1].reset(
data_.C[level_+1].As<HypreParMatrix>()
->EliminateCols(ess_hcurl_tdof));
data_.C[level_+1].As<HypreParMatrix>()->EliminateCols(ess_hcurl_tdof);
++level_;
@@ -194,7 +189,7 @@ void DFSSpaces::DataFinalize()
SparseMatrix P_l2;
for (int l = (int)data_.P_l2.size()-1; l >= 0; --l)
{
data_.P_l2[l]->As<HypreParMatrix>()->GetDiag(P_l2);
data_.P_l2[l].As<HypreParMatrix>()->GetDiag(P_l2);
OperatorPtr PT_l2(Transpose(P_l2));
auto PTW = Mult(*PT_l2.As<SparseMatrix>(), *W.As<SparseMatrix>());
auto cW = Mult(*PTW, P_l2);
@@ -250,7 +245,7 @@ SaddleSchwarzSmoother::SaddleSchwarzSmoother(const HypreParMatrix& M,
const SparseMatrix& agg_hdivdof,
const SparseMatrix& agg_l2dof,
const HypreParMatrix& P_l2,
const ProductOperator& Q_l2)
const HypreParMatrix& Q_l2)
: Solver(M.NumRows() + B.NumRows()), agg_hdivdof_(agg_hdivdof),
agg_l2dof_(agg_l2dof), solvers_loc_(agg_l2dof.NumRows())
{
@@ -317,27 +312,23 @@ void SaddleSchwarzSmoother::Mult(const Vector & x, Vector & y) const
blk_y.GetBlock(1) -= coarse_l2_projection;
}
DivFreeSolver::DivFreeSolver(const HypreParMatrix &M,
const HypreParMatrix &B,
DivFreeSolver::DivFreeSolver(const HypreParMatrix &M, const HypreParMatrix& B,
const DFSData& data)
: DarcySolver(M.NumRows(), B.NumRows()), data_(data), param_(data.param),
BT_(B.Transpose()),
BBT_solver_(B, param_.BBT_solve_param),
ops_offsets_(data.P_l2.size()+1),
ops_(ops_offsets_.size()),
blk_Ps_(ops_.size()-1),
smoothers_(ops_.size())
BT_(B.Transpose()), BBT_solver_(B, param_.BBT_solve_param),
ops_offsets_(data.P_l2.size()+1), ops_(ops_offsets_.size()),
blk_Ps_(ops_.Size()-1), smoothers_(ops_.Size())
{
ops_offsets_.back().MakeRef(DarcySolver::offsets_);
ops_.back() = std::make_unique<BlockOperator>(ops_offsets_.back());
ops_.back()->SetBlock(0, 0, const_cast<HypreParMatrix*>(&M));
ops_.back()->SetBlock(1, 0, const_cast<HypreParMatrix*>(&B));
ops_.back()->SetBlock(0, 1, BT_.Ptr());
ops_.Last() = new BlockOperator(ops_offsets_.back());
ops_.Last()->SetBlock(0, 0, const_cast<HypreParMatrix*>(&M));
ops_.Last()->SetBlock(1, 0, const_cast<HypreParMatrix*>(&B));
ops_.Last()->SetBlock(0, 1, BT_.Ptr());
for (int l = data.P_l2.size(); l >= 0; --l)
{
auto &M_f = static_cast<const HypreParMatrix&>(ops_[l]->GetBlock(0, 0));
auto &B_f = static_cast<const HypreParMatrix&>(ops_[l]->GetBlock(1, 0));
auto& M_f = static_cast<const HypreParMatrix&>(ops_[l]->GetBlock(0, 0));
auto& B_f = static_cast<const HypreParMatrix&>(ops_[l]->GetBlock(1, 0));
if (l == 0)
{
@@ -352,112 +343,123 @@ DivFreeSolver::DivFreeSolver(const HypreParMatrix &M,
const IterSolveParameters& param = param_.coarse_solve_param;
auto coarse_solver = new BDPMinresSolver(M_f, B_f, param);
if (ops_.size() > 1)
if (ops_.Size() > 1)
{
coarse_solver->SetEssZeroDofs(data.coarsest_ess_hdivdofs);
}
smoothers_[l].reset(coarse_solver);
smoothers_[l] = coarse_solver;
continue;
}
auto P_hdiv_l = data.P_hdiv[l-1]->As<HypreParMatrix>();
auto P_l2_l = data.P_l2[l-1]->As<HypreParMatrix>();
HypreParMatrix& P_hdiv_l = *data.P_hdiv[l-1].As<HypreParMatrix>();
HypreParMatrix& P_l2_l = *data.P_l2[l-1].As<HypreParMatrix>();
SparseMatrix& agg_hdivdof_l = *data.agg_hdivdof[l-1].As<SparseMatrix>();
SparseMatrix& agg_l2dof_l = *data.agg_l2dof[l-1].As<SparseMatrix>();
ProductOperator& Q_l2_l = *data.Q_l2[l-1].As<ProductOperator>();
auto* C_l = data.C[l].As<HypreParMatrix>();
HypreParMatrix& Q_l2_l = *data.Q_l2[l-1].As<HypreParMatrix>();
HypreParMatrix* C_l = data.C[l].As<HypreParMatrix>();
auto S0 = new SaddleSchwarzSmoother(M_f, B_f, agg_hdivdof_l,
agg_l2dof_l, *P_l2_l, Q_l2_l);
agg_l2dof_l, P_l2_l, Q_l2_l);
if (param_.coupled_solve)
{
auto S1 = new BlockDiagonalPreconditioner(ops_offsets_[l]);
S1->SetDiagonalBlock(0, new AuxSpaceSmoother(M_f, C_l));
S1->owns_blocks = 1;
smoothers_[l] =
std::make_unique<ProductSolver>(ops_[l].get(), S0, S1, false, true, true);
S1->owns_blocks = true;
smoothers_[l] = new ProductSolver(ops_[l], S0, S1, false, true, true);
}
else
{
smoothers_[l].reset(S0);
smoothers_[l] = S0;
}
HypreParMatrix* M_c = TwoStepsRAP(P_hdiv_l, &M_f, P_hdiv_l);
HypreParMatrix* B_c = TwoStepsRAP(P_l2_l, &B_f, P_hdiv_l);
HypreParMatrix* M_c = TwoStepsRAP(P_hdiv_l, M_f, P_hdiv_l);
HypreParMatrix* B_c = TwoStepsRAP(P_l2_l, B_f, P_hdiv_l);
ops_offsets_[l-1].SetSize(3, 0);
ops_offsets_[l-1][1] = M_c->NumRows();
ops_offsets_[l-1][2] = M_c->NumRows() + B_c->NumRows();
blk_Ps_[l-1] =
std::make_unique<BlockOperator>(ops_offsets_[l], ops_offsets_[l-1]);
blk_Ps_[l-1]->SetBlock(0, 0, P_hdiv_l);
blk_Ps_[l-1]->SetBlock(1, 1, P_l2_l);
blk_Ps_[l-1] = new BlockOperator(ops_offsets_[l], ops_offsets_[l-1]);
blk_Ps_[l-1]->SetBlock(0, 0, &P_hdiv_l);
blk_Ps_[l-1]->SetBlock(1, 1, &P_l2_l);
ops_[l-1] =
std::make_unique<BlockOperator>(ops_offsets_[l-1]);
ops_[l-1] = new BlockOperator(ops_offsets_[l-1]);
ops_[l-1]->SetBlock(0, 0, M_c);
ops_[l-1]->SetBlock(1, 0, B_c);
ops_[l-1]->SetBlock(0, 1, B_c->Transpose());
ops_[l-1]->owns_blocks = 1;
ops_[l-1]->owns_blocks = true;
}
Array<bool> own_ops(ops_.Size());
Array<bool> own_smoothers(smoothers_.Size());
Array<bool> own_Ps(blk_Ps_.Size());
own_ops = true;
own_smoothers = true;
own_Ps = true;
if (data_.P_l2.size() == 0) { return; }
Array<bool> own_ops(ops_.size());
Array<bool> own_smoothers(smoothers_.size());
Array<bool> own_blk_Ps(blk_Ps_.size());
own_ops = false, own_smoothers = false, own_blk_Ps = false;
Array<Solver*> smoothers(smoothers_.size());
if (param_.coupled_solve)
{
solver_.Reset(new GMRESSolver(B.GetComm()));
solver_.As<GMRESSolver>()->SetOperator(*(ops_.back()));
Array<BlockOperator*> ops(ops_.size()), blk_Ps(blk_Ps_.size());
for (size_t i = 0; i < ops_.size(); ++i) { ops[i] = ops_[i].get(); }
for (size_t i = 0; i < blk_Ps_.size(); ++i) { blk_Ps[i] = blk_Ps_[i].get(); }
for (size_t i = 0; i < smoothers_.size(); ++i) { smoothers[i] = smoothers_[i].get(); }
prec_.Reset(new Multigrid(ops, smoothers, blk_Ps,
own_ops, own_smoothers, own_blk_Ps));
solver_.As<GMRESSolver>()->SetOperator(*(ops_.Last()));
prec_.Reset(new Multigrid(ops_, smoothers_, blk_Ps_,
own_ops, own_smoothers, own_Ps));
}
else
{
Array<HypreParMatrix*> ops(data_.P_hcurl.size()+1);
Array<Solver*> smoothers(ops.Size());
Array<HypreParMatrix*> Ps(data_.P_hcurl.size());
auto C_finest = data.C.back().As<HypreParMatrix>();
ops.Last() = TwoStepsRAP(C_finest, &M, C_finest);
own_Ps = false;
HypreParMatrix& C_finest = *data.C.back().As<HypreParMatrix>();
ops.Last() = TwoStepsRAP(C_finest, M, C_finest);
ops.Last()->EliminateZeroRows();
ops.Last()->DropSmallEntries(1e-14);
solver_.Reset(new CGSolver(B.GetComm()));
solver_.As<CGSolver>()->SetOperator(*ops.Last());
smoothers.Last() = new HypreSmoother(*ops.Last());
static_cast<HypreSmoother*>(smoothers.Last())->SetOperatorSymmetry(true);
for (int l = Ps.Size()-1; l >= 0; --l)
{
Ps[l] = data_.P_hcurl[l]->As<HypreParMatrix>();
ops[l] = TwoStepsRAP(Ps[l], ops[l+1], Ps[l]);
Ps[l] = data_.P_hcurl[l].As<HypreParMatrix>();
ops[l] = TwoStepsRAP(*Ps[l], *ops[l+1], *Ps[l]);
ops[l]->DropSmallEntries(1e-14);
smoothers[l] = new HypreSmoother(*ops[l]);
static_cast<HypreSmoother*>(smoothers[l])->SetOperatorSymmetry(true);
}
own_ops = true, own_smoothers = true;
prec_.Reset(new Multigrid(ops, smoothers, Ps,
own_ops, own_smoothers, own_blk_Ps));
prec_.Reset(new Multigrid(ops, smoothers, Ps, own_ops, own_smoothers, own_Ps));
}
solver_.As<IterativeSolver>()->SetPreconditioner(*prec_.As<Solver>());
SetOptions(*solver_.As<IterativeSolver>(), param_);
}
DivFreeSolver::~DivFreeSolver()
{
if (param_.coupled_solve) { return; }
for (int i = 0; i < ops_.Size(); ++i)
{
delete ops_[i];
delete smoothers_[i];
if (i == ops_.Size() - 1) { break; }
delete blk_Ps_[i];
}
}
void DivFreeSolver::SolveParticular(const Vector& rhs, Vector& sol) const
{
std::vector<Vector> rhss(smoothers_.size()), sols(smoothers_.size());
std::vector<Vector> rhss(smoothers_.Size());
std::vector<Vector> sols(smoothers_.Size());
rhss.back().SetDataAndSize(const_cast<real_t*>(rhs.HostRead()), rhs.Size());
sols.back().SetDataAndSize(sol.HostWrite(), sol.Size());
for (int l = blk_Ps_.size()-1; l >= 0; --l)
for (int l = blk_Ps_.Size()-1; l >= 0; --l)
{
rhss[l].SetSize(blk_Ps_[l]->NumCols());
sols[l].SetSize(blk_Ps_[l]->NumCols());
@@ -468,12 +470,12 @@ void DivFreeSolver::SolveParticular(const Vector& rhs, Vector& sol) const
blk_Ps_[l]->MultTranspose(rhss[l+1], rhss[l]);
}
for (size_t l = 0; l < smoothers_.size(); ++l)
for (int l = 0; l < smoothers_.Size(); ++l)
{
smoothers_[l]->Mult(rhss[l], sols[l]);
}
for (size_t l = 0; l < blk_Ps_.size(); ++l)
for (int l = 0; l < blk_Ps_.Size(); ++l)
{
Vector P_sol(blk_Ps_[l]->NumRows());
blk_Ps_[l]->Mult(sols[l], P_sol);
@@ -505,12 +507,12 @@ void DivFreeSolver::Mult(const Vector & x, Vector & y) const
MFEM_VERIFY(x.Size() == offsets_[2], "MLDivFreeSolver: x size is invalid");
MFEM_VERIFY(y.Size() == offsets_[2], "MLDivFreeSolver: y size is invalid");
if (ops_.size() == 1) { smoothers_[0]->Mult(x, y); return; }
if (ops_.Size() == 1) { smoothers_[0]->Mult(x, y); return; }
BlockVector blk_y(y, offsets_);
BlockVector resid(offsets_);
ops_.back()->Mult(y, resid);
ops_.Last()->Mult(y, resid);
add(1.0, x, -1.0, resid, resid);
BlockVector correction(offsets_);
@@ -537,7 +539,7 @@ void DivFreeSolver::Mult(const Vector & x, Vector & y) const
ch.Clear();
ch.Start();
ops_.back()->Mult(y, resid);
ops_.Last()->Mult(y, resid);
add(1.0, x, -1.0, resid, resid);
SolveDivFree(resid.GetBlock(0), correction.GetBlock(0));
@@ -551,7 +553,7 @@ void DivFreeSolver::Mult(const Vector & x, Vector & y) const
ch.Clear();
ch.Start();
auto& M = dynamic_cast<const HypreParMatrix&>(ops_.back()->GetBlock(0, 0));
auto& M = dynamic_cast<const HypreParMatrix&>(ops_.Last()->GetBlock(0, 0));
M.Mult(-1.0, correction.GetBlock(0), 1.0, resid.GetBlock(0));
SolvePotential(resid.GetBlock(0), correction.GetBlock(1));
blk_y.GetBlock(1) += correction.GetBlock(1);
@@ -565,12 +567,11 @@ void DivFreeSolver::Mult(const Vector & x, Vector & y) const
int DivFreeSolver::GetNumIterations() const
{
if (ops_.size() == 1)
if (ops_.Size() == 1)
{
return static_cast<BDPMinresSolver*>
(smoothers_.at(0).get())->GetNumIterations();
return static_cast<BDPMinresSolver*>(smoothers_[0])->GetNumIterations();
}
return solver_.As<IterativeSolver>()->GetNumIterations();
}
} // namespace mfem::blocksolvers
} // namespace blocksolvers
} // namespace mfem
+28 -24
View File
@@ -13,11 +13,11 @@
#define MFEM_DIVFREE_SOLVER_HPP
#include "darcy_solver.hpp"
#include <memory>
namespace mfem::blocksolvers
namespace mfem
{
namespace blocksolvers
{
/// Parameters for the divergence free solver
struct DFSParameters : IterSolveParameters
{
@@ -35,18 +35,14 @@ struct DFSParameters : IterSolveParameters
/// Data for the divergence free solver
struct DFSData
{
using UniqueOperatorPtr = std::unique_ptr<OperatorPtr>;
using UniqueHypreParMatrix = std::unique_ptr<HypreParMatrix>;
std::vector<OperatorPtr> agg_hdivdof; // agglomerates to H(div) dofs table
std::vector<OperatorPtr> agg_l2dof; // agglomerates to L2 dofs table
std::vector<UniqueOperatorPtr> P_hdiv; // Interpolation matrix for H(div) space
std::vector<UniqueOperatorPtr> P_l2; // Interpolation matrix for L2 space
std::vector<UniqueOperatorPtr> P_hcurl; // Interpolation for kernel space of div
std::vector<OperatorPtr> Q_l2; // Q_l2[l] = (W_{l+1})^{-1} P_l2[l]^T W_l
Array<int> coarsest_ess_hdivdofs; // coarsest level essential H(div) dofs
std::vector<OperatorPtr> C; // discrete curl: ND -> RT, map to Null(B)
std::vector<UniqueHypreParMatrix> Ae;
std::vector<OperatorPtr> agg_hdivdof; // agglomerates to H(div) dofs table
std::vector<OperatorPtr> agg_l2dof; // agglomerates to L2 dofs table
std::vector<OperatorPtr> P_hdiv; // Interpolation matrix for H(div) space
std::vector<OperatorPtr> P_l2; // Interpolation matrix for L2 space
std::vector<OperatorPtr> P_hcurl; // Interpolation for kernel space of div
std::vector<OperatorPtr> Q_l2; // Q_l2[l] = (W_{l+1})^{-1} P_l2[l]^T W_l
Array<int> coarsest_ess_hdivdofs; // coarsest level essential H(div) dofs
std::vector<OperatorPtr> C; // discrete curl: ND -> RT, map to Null(B)
DFSParameters param;
};
@@ -96,7 +92,8 @@ public:
/// Compute the product B * B^T and solve it with CG preconditioned by BoomerAMG
class BBTSolver : public Solver
{
OperatorPtr BBT_, BBT_prec_;
OperatorPtr BBT_;
OperatorPtr BBT_prec_;
CGSolver BBT_solver_;
public:
BBTSolver(const HypreParMatrix &B, IterSolveParameters param);
@@ -118,11 +115,14 @@ public:
/// [ B 0 ]
class SaddleSchwarzSmoother : public Solver
{
const SparseMatrix &agg_hdivdof_, &agg_l2dof_;
const SparseMatrix& agg_hdivdof_;
const SparseMatrix& agg_l2dof_;
OperatorPtr coarse_l2_projector_;
Array<int> offsets_;
mutable Array<int> offsets_loc_, hdivdofs_loc_, l2dofs_loc_;
mutable Array<int> offsets_loc_;
mutable Array<int> hdivdofs_loc_;
mutable Array<int> l2dofs_loc_;
std::vector<OperatorPtr> solvers_loc_;
public:
/** SaddleSchwarzSmoother solves local saddle point problems defined on a
@@ -140,7 +140,7 @@ public:
const SparseMatrix& agg_hdivdof,
const SparseMatrix& agg_l2dof,
const HypreParMatrix& P_l2,
const ProductOperator& Q_l2);
const HypreParMatrix& Q_l2);
void Mult(const Vector &x, Vector &y) const override;
void MultTranspose(const Vector &x, Vector &y) const override { Mult(x, y); }
void SetOperator(const Operator &op) override { }
@@ -178,10 +178,11 @@ class DivFreeSolver : public DarcySolver
OperatorPtr BT_;
BBTSolver BBT_solver_;
std::vector<Array<int>> ops_offsets_;
std::vector<std::unique_ptr<BlockOperator>> ops_;
std::vector<std::unique_ptr<BlockOperator>> blk_Ps_;
std::vector<std::unique_ptr<Solver>> smoothers_;
OperatorPtr prec_, solver_;
Array<BlockOperator*> ops_;
Array<BlockOperator*> blk_Ps_;
Array<Solver*> smoothers_;
OperatorPtr prec_;
OperatorPtr solver_;
void SolveParticular(const Vector& rhs, Vector& sol) const;
void SolveDivFree(const Vector& rhs, Vector& sol) const;
@@ -189,11 +190,14 @@ class DivFreeSolver : public DarcySolver
public:
DivFreeSolver(const HypreParMatrix& M, const HypreParMatrix &B,
const DFSData& data);
~DivFreeSolver();
void Mult(const Vector &x, Vector &y) const override;
void SetOperator(const Operator &op) override { }
int GetNumIterations() const override;
};
} // namespace mfem::blocksolvers
} // namespace blocksolvers
} // namespace mfem
#endif // MFEM_DIVFREE_SOLVER_HPP

Some files were not shown because too many files have changed in this diff Show More