Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
adaf2bbec6 | ||
|
|
b8aa60060b | ||
|
|
38c243ab05 |
@@ -94,16 +94,6 @@ inputs:
|
||||
description: If true, do not set any CXXFLAGS or LDFLAGS.
|
||||
default: false
|
||||
|
||||
# Unfortunately, "uses:" fields cannot have references to variables like
|
||||
# ${{env.MFEM_ACTIONS_VERSION}}, so the branch/tag name has to be hard coded.
|
||||
# Therefore, in the future, when updating the version of the
|
||||
# mfem/github-actions to use, we'll have to replace:
|
||||
# - all definitions of MFEM_ACTIONS_VERSION and
|
||||
# - all "uses:" fields that refer to mfem/github-actions.
|
||||
MFEM_ACTIONS_VERSION:
|
||||
description: Version (branch or tag) of the mfem/github-actions to use.
|
||||
default: v2.7
|
||||
|
||||
runs:
|
||||
using: 'composite'
|
||||
steps:
|
||||
@@ -128,7 +118,6 @@ runs:
|
||||
echo UBSAN_LDFLAGS=${{inputs.UBSAN_LDFLAGS}} >> $GITHUB_ENV
|
||||
echo MSAN_CXXFLAGS=${{inputs.MSAN_CXXFLAGS}} >> $GITHUB_ENV
|
||||
echo MSAN_LDFLAGS=${{inputs.MSAN_LDFLAGS}} >> $GITHUB_ENV
|
||||
echo MFEM_ACTIONS_VERSION=${{inputs.MFEM_ACTIONS_VERSION}} >> $GITHUB_ENV
|
||||
shell: bash
|
||||
|
||||
- name: Env (dir)
|
||||
|
||||
@@ -53,7 +53,7 @@ runs:
|
||||
run: echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.UBSAN_CXXFLAGS}} >> $GITHUB_ENV
|
||||
shell: bash
|
||||
|
||||
- uses: mfem/github-actions/build-mfem@v2.7
|
||||
- uses: mfem/github-actions/build-mfem@v2.5
|
||||
if: ${{steps.debug.outputs.cache-hit != 'true'}}
|
||||
env:
|
||||
CXXFLAGS: ${{env.CXXFLAGS}}
|
||||
@@ -82,7 +82,7 @@ runs:
|
||||
run: find . -type f -name '*.o' -delete
|
||||
shell: bash
|
||||
|
||||
- uses: actions/upload-artifact@v7
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: build-${{inputs.par}}-${{inputs.sanitizer}}
|
||||
path: mfem/build
|
||||
|
||||
@@ -12,11 +12,6 @@
|
||||
name: 'Install MPI'
|
||||
description: 'Installs MPI and set up its environment variables'
|
||||
|
||||
inputs:
|
||||
NO_FLAGS:
|
||||
description: If true, do not set any CXXFLAGS or LDFLAGS.
|
||||
default: false
|
||||
|
||||
runs:
|
||||
using: 'composite'
|
||||
steps:
|
||||
@@ -32,7 +27,6 @@ runs:
|
||||
shell: bash
|
||||
|
||||
- name: Env (bis)
|
||||
if: ${{ inputs.NO_FLAGS != 'true' }}
|
||||
run: |
|
||||
echo CXXFLAGS=${{env.CXXFLAGS}} ${{env.MPI_INC}} >> $GITHUB_ENV
|
||||
echo LDFLAGS=${{env.LDFLAGS}} ${{env.MPI_LIB}} >> $GITHUB_ENV
|
||||
|
||||
@@ -49,7 +49,7 @@ runs:
|
||||
par: ${{inputs.par}}
|
||||
sanitizer: ${{inputs.sanitizer}}
|
||||
|
||||
- uses: actions/download-artifact@v8
|
||||
- uses: actions/download-artifact@v4
|
||||
with:
|
||||
name: build-${{inputs.par}}-${{inputs.sanitizer}}
|
||||
path: mfem/build
|
||||
|
||||
@@ -37,14 +37,14 @@ runs:
|
||||
with:
|
||||
path: ${{env.HYPRE_DIR}}
|
||||
fail-on-cache-miss: true
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{env.MFEM_ACTIONS_VERSION}}
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
|
||||
|
||||
- uses: actions/cache/restore@v5 # Cache for Metis
|
||||
if: ${{inputs.par == 'true'}}
|
||||
with:
|
||||
path: ${{env.METIS_DIR}}
|
||||
fail-on-cache-miss: true
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
|
||||
|
||||
- name: Hypre/Metis links
|
||||
if: ${{inputs.par == 'true'}}
|
||||
|
||||
@@ -1,42 +0,0 @@
|
||||
# MFEM Pull Request Review Agent Guide
|
||||
|
||||
## Purpose and scope
|
||||
Review MFEM PRs for correctness, maintainability, performance, portability, test coverage, and MFEM consistency. Use the diff and PR context; reference source files, tests, and CI results when available. Follow `CONTRIBUTING.md`, especially Developer Guidelines, PR rules, checklist, and testing.
|
||||
|
||||
## Critical review pillars
|
||||
- Correctness and numerical behavior
|
||||
- API and user-facing impact
|
||||
- Performance implications
|
||||
- Maintainability and portability
|
||||
|
||||
## Review workflow
|
||||
1. Read the PR description, linked issues, and intended behavior.
|
||||
2. Inspect the diff before commenting.
|
||||
3. Identify affected MFEM components, examples, tests, build or docs changes, and downstream APIs.
|
||||
4. Analyze the code against the critical review pillars.
|
||||
5. Compare the change against nearby code and MFEM patterns; flag unmotivated deviations.
|
||||
6. Check whether tests and documentation were updated appropriately.
|
||||
7. Review CI results and suggest actions.
|
||||
8. Produce a structured review with prioritized findings.
|
||||
9. Always limit conclusions to available evidence.
|
||||
|
||||
## MFEM-specific review checklist
|
||||
- Component-aware scope: identify the touched subsystem (FEM, solvers, preconditioners, linear algebra, mesh, examples, miniapps, build, or docs) and assess its impact against the review pillars.
|
||||
- Numerical and algorithmic behavior: assess issues in convergence, stability, tolerances, precision, iteration limits, and failure handling. If clear opportunities exist to improve the algorithmic approach, call them out with expected impact.
|
||||
- API and user-facing impact: assess backward compatibility, user-visible behavior and default changes, migration impact, deprecations, and whether documentation clearly explains user-facing API changes.
|
||||
- Data structure and memory semantics: assess ownership, lifetime, aliasing, container behavior, and device-host synchronization.
|
||||
- Parallel and serial behavior: assess whether the change preserves equivalent semantics in serial and parallel modes where applicable; if logic is currently mode-specific, check whether extension to the other mode is straightforward (clear abstractions, no hard-wired assumptions), document constraints, and call out expected behavior differences explicitly.
|
||||
- Backend and portability impact: assess likely cross-backend risks in CPU, CUDA, HIP, OCCA, RAJA, partial assembly, fallback paths, compiler compatibility, and platform assumptions.
|
||||
- Build, dependency, and configuration impact: assess CMake or make changes, optional dependency behavior, and feature-flag interactions.
|
||||
- Tests and docs alignment: check available regression or unit coverage evidence for changed behavior, and ensure docs are updated for new flags, APIs, options, or behavior changes.
|
||||
- MFEM developer-guideline fit: keep code lean, simple, general, logically separated, and portable; suggest C++17 improvements when they clearly improve safety, clarity, or maintainability.
|
||||
- New source files, examples, or miniapps: if a PR adds source/header files, verify they are properly wired into the relevant `makefile` and `CMakeLists.txt`, referenced in docs where applicable (including `doc/CodeDocumentation.dox`), and added to top-level `.gitignore` only when generated artifacts require it.
|
||||
- Changelog: verify `CHANGELOG` is updated if the PR introduces significant new features or user-facing changes.
|
||||
- MFEM conventions: use `real_t`; use `mfem::out`/`mfem::err` instead of `std::cout`/`std::cerr` in library code; flag large/binary files; if AI assistance is apparent but undisclosed, suggest following `CONTRIBUTING.md`.
|
||||
- Edge cases: if the PR touches complex or error-prone areas, suggest additional tests for edge cases, failure modes, and parallel behavior.
|
||||
|
||||
## Commenting guidelines
|
||||
- Keep comments concise, actionable, and grounded in the diff.
|
||||
- Focus on correctness, behavior changes, and user impact over style nits.
|
||||
- Be professional, concise, collaborative, technically precise, and avoid unsupported assumptions.
|
||||
|
||||
@@ -13,7 +13,7 @@ Note that some of these scripts use the shared MFEM GitHub Actions from the exte
|
||||
|
||||
<https://github.com/mfem/github-actions>
|
||||
|
||||
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch (or tag) in the above from which the action is taken.
|
||||
For a particular action, e.g. `mfem/github-actions/build-mfem@v2.5`, the `v2.5` suffix denotes the branch in the above from which the action is taken.
|
||||
|
||||
The current CI workflows are:
|
||||
|
||||
@@ -29,12 +29,16 @@ Runs a number of static repository-level sanity checks.
|
||||
|
||||
- `branch-history` guards against accidental commits of large files using the `--history` option of the `config/githooks/pre-push` script.
|
||||
|
||||
## `mfem-analysis.yml` (`build-analysis`)
|
||||
|
||||
Checks if the code builds and satisfies minimal requirements.
|
||||
|
||||
- `gitignore` builds hypre, METIS, and MFEM using `mfem/github-actions/build-hypre`, `mfem/github-actions/build-metis`, and `mfem/github-actions/build-mfem` and checks for correct `.gitignore` settings by running the `tests/scripts/gitignore` script.
|
||||
|
||||
## `builds-and-tests.yml`
|
||||
|
||||
Runs a matrix of builds and tests runs with different compilers, OS, mfem/hypre settings, etc. Also processes and upload Codecov reports.
|
||||
|
||||
One matrix job runs `tests/scripts/gitignore` after `make test-noclean` to check generated artifacts against `.gitignore`.
|
||||
|
||||
Uses the following GitHub Actions from <https://github.com/mfem/github-actions>:
|
||||
|
||||
- `mfem/github-actions/build-hypre`
|
||||
|
||||
@@ -40,7 +40,6 @@ env:
|
||||
METIS_ARCHIVE_MAC: metis-4.0.3-mac.tgz
|
||||
METIS_TOP_DIR: metis-4.0.3
|
||||
MFEM_TOP_DIR: mfem
|
||||
MFEM_ACTIONS_VERSION: v2.7
|
||||
|
||||
# Note for future improvements:
|
||||
#
|
||||
@@ -111,7 +110,6 @@ jobs:
|
||||
build-system: make
|
||||
hypre-target: int64
|
||||
precision: fp64
|
||||
gitignore-check: YES
|
||||
- os: ubuntu-latest
|
||||
target: opt
|
||||
codecov: NO
|
||||
@@ -142,10 +140,6 @@ jobs:
|
||||
|
||||
continue-on-error: ${{ matrix.enzyme && true || false }}
|
||||
|
||||
# Enable ccache for all jobs except Windows (would need sccache).
|
||||
env:
|
||||
USE_CCACHE: ${{ matrix.os != 'windows-latest' }}
|
||||
|
||||
steps:
|
||||
# Fix 'No space left on device' errors for Ubuntu builds.
|
||||
- name: Run Actions Cleaner
|
||||
@@ -176,6 +170,20 @@ jobs:
|
||||
env
|
||||
shell: bash
|
||||
|
||||
# For info on Xcode see:
|
||||
# - https://github.com/actions/runner-images/issues/12541
|
||||
# - https://github.com/actions/runner-images/blob/releases/macos-15-arm64/20250811/images/macos/macos-15-arm64-Readme.md#xcode
|
||||
- name: Xcode version setup (MacOS)
|
||||
if: matrix.os == 'macos-latest'
|
||||
run: |
|
||||
XCODE_PATH="/Applications/Xcode_16.4.app"
|
||||
echo "> sudo xcode-select -s ${XCODE_PATH}"
|
||||
sudo xcode-select -s ${XCODE_PATH}
|
||||
echo "> g++ -v"
|
||||
g++ -v
|
||||
echo "> clang++ -v"
|
||||
clang++ -v
|
||||
|
||||
# Only get MPI if defined for the job.
|
||||
# TODO: It would be nice to have only one step, e.g. with a dedicated
|
||||
# action, but I (@adrienbernede) don't see how at the moment.
|
||||
@@ -220,11 +228,11 @@ jobs:
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-${{ matrix.precision }}-v2.5
|
||||
|
||||
- name: get hypre
|
||||
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os != 'windows-latest'
|
||||
uses: mfem/github-actions/build-hypre@v2.7
|
||||
uses: mfem/github-actions/build-hypre@v2.5
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
@@ -234,7 +242,7 @@ jobs:
|
||||
|
||||
- name: get hypre (Windows)
|
||||
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true' && matrix.os == 'windows-latest'
|
||||
uses: mfem/github-actions/build-hypre@v2.7
|
||||
uses: mfem/github-actions/build-hypre@v2.5
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
@@ -250,11 +258,11 @@ jobs:
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
|
||||
|
||||
- name: install metis
|
||||
if: matrix.mpi == 'par' && matrix.os != 'windows-latest' && steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.7
|
||||
uses: mfem/github-actions/build-metis@v2.5
|
||||
with:
|
||||
archive: ${{ matrix.os != 'macos-latest' && env.METIS_ARCHIVE || env.METIS_ARCHIVE_MAC }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
@@ -294,55 +302,9 @@ jobs:
|
||||
echo "OMPI_CC=$LLVM_PREFIX/bin/clang" >> $GITHUB_ENV
|
||||
echo "OMPI_CXX=$LLVM_PREFIX/bin/clang++" >> $GITHUB_ENV
|
||||
|
||||
# Restore the compiler cache (ccache). The key embeds the run id, so new
|
||||
# runs save a fresh snapshot; the restore-keys prefix warm-starts from the
|
||||
# most recent prior run (incl. the base branch for PRs).
|
||||
- name: cache ccache
|
||||
if: ${{ env.USE_CCACHE == 'true' }}
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: .ccache
|
||||
key: ccache-${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}${{ matrix.enzyme && '-enzyme' || '' }}-${{ github.run_id }}
|
||||
restore-keys: |
|
||||
ccache-${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}${{ matrix.enzyme && '-enzyme' || '' }}-
|
||||
|
||||
# Configure ccache and select how it is injected into the MFEM build:
|
||||
# - make: set CXX="ccache g++"; for MPI, OMPI_CXX="ccache g++" so mpicxx
|
||||
# runs ccache around g++ (not ccache around the mpicxx wrapper).
|
||||
# - cmake: set CMAKE_<LANG>_COMPILER_LAUNCHER=ccache.
|
||||
# - enzyme: wrap the brew clang++ via OMPI_CXX.
|
||||
# The chosen options are passed through build-mfem's 'config-options'
|
||||
# input (see the build step below).
|
||||
- name: configure ccache
|
||||
if: ${{ env.USE_CCACHE == 'true' }}
|
||||
run: |
|
||||
command -v ccache >/dev/null 2>&1 || {
|
||||
if [[ "${{ runner.os }}" == "Linux" ]]; then
|
||||
sudo apt-get update && sudo apt-get install -y ccache
|
||||
else
|
||||
brew install ccache
|
||||
fi
|
||||
}
|
||||
echo "CCACHE_DIR=${{ github.workspace }}/.ccache" >> $GITHUB_ENV
|
||||
echo "CCACHE_MAXSIZE=1G" >> $GITHUB_ENV
|
||||
echo "CCACHE_COMPILERCHECK=content" >> $GITHUB_ENV
|
||||
# Ignore header timestamps (restamped by each checkout) so direct mode hits.
|
||||
echo "CCACHE_SLOPPINESS=include_file_mtime,include_file_ctime,time_macros" >> $GITHUB_ENV
|
||||
# Hash absolute paths relative to the workspace.
|
||||
echo "CCACHE_BASEDIR=${{ github.workspace }}" >> $GITHUB_ENV
|
||||
if [[ "${{ matrix.enzyme }}" == "true" ]]; then
|
||||
echo "OMPI_CXX=ccache $LLVM_PREFIX/bin/clang++" >> $GITHUB_ENV
|
||||
elif [[ "${{ matrix.build-system }}" == "cmake" ]]; then
|
||||
echo 'CCACHE_CONFIG_OPTS=-DCMAKE_CXX_COMPILER_LAUNCHER=ccache -DCMAKE_C_COMPILER_LAUNCHER=ccache' >> $GITHUB_ENV
|
||||
else
|
||||
echo "OMPI_CXX=ccache g++" >> $GITHUB_ENV
|
||||
echo 'CCACHE_CONFIG_OPTS=CXX="ccache g++" MPICXX="mpicxx"' >> $GITHUB_ENV
|
||||
fi
|
||||
shell: bash
|
||||
|
||||
# MFEM build and test
|
||||
- name: build
|
||||
uses: mfem/github-actions/build-mfem@v2.7
|
||||
uses: mfem/github-actions/build-mfem@v2.5
|
||||
env:
|
||||
VCPKG_DEFAULT_BINARY_CACHE: ${{ github.workspace }}/vcpkg_cache
|
||||
with:
|
||||
@@ -355,14 +317,9 @@ jobs:
|
||||
metis-dir: ${{ env.METIS_TOP_DIR }}
|
||||
mfem-dir: ${{ env.MFEM_TOP_DIR }}
|
||||
precision: ${{ matrix.precision }}
|
||||
config-options: ${{ matrix.config-opts }} ${{ env.CCACHE_CONFIG_OPTS }}
|
||||
config-options: ${{ matrix.config-opts }}
|
||||
library-only: ${{ matrix.target == 'dbg' && matrix.os != 'ubuntu-latest' }}
|
||||
|
||||
- name: ccache stats
|
||||
if: ${{ env.USE_CCACHE == 'true' }}
|
||||
run: ccache -s
|
||||
shell: bash
|
||||
|
||||
# Run checks (and only checks) on debug targets
|
||||
- name: checks
|
||||
if: matrix.build-system == 'make' && matrix.target == 'dbg'
|
||||
@@ -373,13 +330,7 @@ jobs:
|
||||
- name: tests
|
||||
if: matrix.build-system == 'make' && (matrix.target == 'opt' || matrix.os == 'ubuntu-latest')
|
||||
run: |
|
||||
cd ${{ env.MFEM_TOP_DIR }}
|
||||
if [[ "${{ matrix.gitignore-check }}" == "YES" ]]; then
|
||||
make test-noclean
|
||||
else
|
||||
make test
|
||||
fi
|
||||
shell: bash
|
||||
cd ${{ env.MFEM_TOP_DIR }} && make test
|
||||
|
||||
- name: cmake checks
|
||||
if: matrix.build-system == 'cmake' && matrix.target == 'dbg'
|
||||
@@ -424,16 +375,8 @@ jobs:
|
||||
# Code coverage (process and upload reports)
|
||||
- name: codecov
|
||||
if: matrix.codecov == 'YES'
|
||||
uses: mfem/github-actions/upload-coverage@v2.7
|
||||
uses: mfem/github-actions/upload-coverage@v2.5
|
||||
with:
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}-${{ matrix.precision }}
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}
|
||||
project_dir: ${{ env.MFEM_TOP_DIR }}
|
||||
directories: "fem general linalg mesh"
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
|
||||
- name: gitignore
|
||||
if: matrix.gitignore-check == 'YES'
|
||||
run: |
|
||||
cd ${{ env.MFEM_TOP_DIR }}/tests/scripts
|
||||
./runtest gitignore
|
||||
|
||||
@@ -1,42 +0,0 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
---
|
||||
# A closed PR's caches can never be restored again, so delete them to free
|
||||
# space against the 10 GB per-repo cache limit.
|
||||
name: Cleanup PR caches
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [closed]
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
jobs:
|
||||
cleanup:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Delete caches for the closed PR
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
PR_REF: refs/pull/${{ github.event.pull_request.number }}/merge
|
||||
run: |
|
||||
echo "Deleting caches for $PR_REF"
|
||||
while :; do
|
||||
ids=$(gh cache list --ref "$PR_REF" --limit 100 --json id --jq '.[].id')
|
||||
[ -n "$ids" ] || break
|
||||
echo "$ids" | while read -r id; do
|
||||
[ -n "$id" ] || continue
|
||||
echo "Deleting cache $id"
|
||||
gh cache delete "$id" || echo " (already gone)"
|
||||
done
|
||||
done
|
||||
@@ -14,19 +14,9 @@ name: "Static Analysis"
|
||||
on:
|
||||
push:
|
||||
branches: ["master", "next"]
|
||||
paths-ignore: &docs-only-paths
|
||||
- "**/*.md"
|
||||
- "doc/**"
|
||||
- ".binder/**"
|
||||
- "CITATION.cff"
|
||||
- "LICENSE"
|
||||
- "NOTICE"
|
||||
- "CHANGELOG"
|
||||
- "INSTALL"
|
||||
pull_request:
|
||||
# The branches below must be a subset of the branches above
|
||||
branches: ["master"]
|
||||
paths-ignore: *docs-only-paths
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
name: "Build Analysis"
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
- next
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
HYPRE_ARCHIVE: v2.19.0.tar.gz
|
||||
HYPRE_TOP_DIR: hypre-2.19.0
|
||||
METIS_ARCHIVE: metis-4.0.3.tar.gz
|
||||
METIS_TOP_DIR: metis-4.0.3
|
||||
COVERAGE_ENV: mfem-coverage
|
||||
|
||||
jobs:
|
||||
gitignore:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- name: checkout MFEM
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
path: mfem
|
||||
|
||||
- name: Get MPI (Linux)
|
||||
run: |
|
||||
sudo apt-get install openmpi-bin libopenmpi-dev
|
||||
export OMPI_MCA_rmaps_base_oversubscribe=1
|
||||
|
||||
- name: Cache Hypre Install
|
||||
id: hypre-cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-ompi-build-${{ env.HYPRE_TOP_DIR }}-v2.5
|
||||
|
||||
- name: Get Hypre
|
||||
if: steps.hypre-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-hypre@v2.5
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
target: int32
|
||||
|
||||
- name: Cache Metis Install
|
||||
id: metis-cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.5
|
||||
|
||||
- name: Install Metis
|
||||
if: steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.5
|
||||
with:
|
||||
archive: ${{ env.METIS_ARCHIVE }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
|
||||
# MFEM build and test
|
||||
- name: build-mfem
|
||||
uses: mfem/github-actions/build-mfem@v2.5
|
||||
with:
|
||||
os: ${{ runner.os }}
|
||||
target: opt
|
||||
codecov: NO
|
||||
mpi: par
|
||||
build-system: make
|
||||
hypre-dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
metis-dir: ${{ env.METIS_TOP_DIR }}
|
||||
mfem-dir: mfem
|
||||
|
||||
- name: test (no clean)
|
||||
run: |
|
||||
cd mfem && make test-noclean
|
||||
|
||||
- name: gitignore
|
||||
run: |
|
||||
cd mfem/tests/scripts
|
||||
./runtest gitignore
|
||||
@@ -13,7 +13,6 @@ name: "Checks"
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
pull-requests: read
|
||||
|
||||
on:
|
||||
push:
|
||||
@@ -30,11 +29,6 @@ concurrency:
|
||||
# by checking if the workflow trigger is 'push' ("github.event_name == 'push'")
|
||||
# and if we are in a fork ("github.event.pull_request.head.repo.full_name !=
|
||||
# github.repository").
|
||||
#
|
||||
# The logic for the branch-history check is slightly different, since that check
|
||||
# also inspects the PR's labels to allow for overriding failures. In this case,
|
||||
# we run on all 'pull_request' triggers, but only run for 'push' triggers that
|
||||
# do not correspond to any open PRs.
|
||||
|
||||
jobs:
|
||||
file-headers-check:
|
||||
@@ -134,7 +128,10 @@ jobs:
|
||||
|
||||
branch-history:
|
||||
if: |
|
||||
github.ref != 'refs/heads/next' && github.ref != 'refs/heads/master'
|
||||
github.ref != 'refs/heads/next' &&
|
||||
github.ref != 'refs/heads/master' &&
|
||||
(github.event_name == 'push' ||
|
||||
github.event.pull_request.head.repo.full_name != github.repository)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: checkout mfem
|
||||
@@ -142,27 +139,7 @@ jobs:
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: check for pull request
|
||||
id: check_pr
|
||||
if: github.event_name == 'push'
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
pr_exists=$(gh pr list --repo "$GITHUB_REPOSITORY" \
|
||||
--head "$GITHUB_REF_NAME" \
|
||||
--state open \
|
||||
--json number \
|
||||
--jq 'length > 0')
|
||||
echo "pr_exists=$pr_exists" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: branch-history
|
||||
id: branch_history
|
||||
if: |
|
||||
(github.event_name == 'pull_request' ||
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
steps.check_pr.outputs.pr_exists == 'false')
|
||||
continue-on-error: ${{ contains(github.event.pull_request.labels.*.name,
|
||||
'branch-history-override') }}
|
||||
run: |
|
||||
# We override origin to make sure we point to the main repo.
|
||||
# This is to have consistent test results on PRs from forks.
|
||||
@@ -170,9 +147,3 @@ jobs:
|
||||
git remote add origin https://github.com/mfem/mfem.git
|
||||
git checkout -b gh-actions-branch-history
|
||||
./config/githooks/pre-push --history
|
||||
|
||||
- name: report branch-history override
|
||||
if: steps.branch_history.outcome == 'failure'
|
||||
run: |
|
||||
echo "::warning::branch-history check failed, but the" \
|
||||
"'branch-history-override' label is set."
|
||||
|
||||
@@ -19,22 +19,18 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Cache
|
||||
id: cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{env.HYPRE_DIR}}
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-${{ env.MFEM_ACTIONS_VERSION }}
|
||||
key: ${{runner.os}}-ompi-build-${{env.HYPRE_DIR}}-int32-fp64-v2.5
|
||||
- name: Setup
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
uses: ./.github/actions/sanitize/mpi
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Build
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-hypre@v2.7
|
||||
uses: mfem/github-actions/build-hypre@v2.5
|
||||
with:
|
||||
archive: ${{env.HYPRE_TGZ}}
|
||||
dir: ${{env.HYPRE_DIR}}
|
||||
|
||||
@@ -19,22 +19,18 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: ./.github/actions/sanitize/config
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Cache
|
||||
id: cache
|
||||
uses: actions/cache@v5
|
||||
with:
|
||||
path: ${{env.METIS_DIR}}
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-${{env.MFEM_ACTIONS_VERSION}}
|
||||
key: ${{runner.os}}-build-${{env.METIS_DIR}}-v2.5
|
||||
- name: Setup
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
uses: ./.github/actions/sanitize/mpi
|
||||
with:
|
||||
NO_FLAGS: true
|
||||
- name: Build
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.7
|
||||
uses: mfem/github-actions/build-metis@v2.5
|
||||
with:
|
||||
archive: ${{env.METIS_TGZ}}
|
||||
dir: ${{env.METIS_DIR}}
|
||||
|
||||
@@ -146,7 +146,7 @@ jobs:
|
||||
if: ${{steps.restore.outputs.cache-hit != 'true'}}
|
||||
working-directory: mfem/build/tests/unit
|
||||
run: find . -type f -name '*.o' -delete
|
||||
- uses: actions/upload-artifact@v7
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
|
||||
path: mfem/build/tests/unit/${{env.unit_tests}}
|
||||
@@ -172,7 +172,7 @@ jobs:
|
||||
par: ${{inputs.par}}
|
||||
sanitizer: ${{inputs.sanitizer}}
|
||||
cache-path: mfem/build/tests/unit/${{env.unit_tests}}
|
||||
- uses: actions/download-artifact@v8
|
||||
- uses: actions/download-artifact@v4
|
||||
if: ${{steps.restore.outputs.cache-hit != 'true'}}
|
||||
with:
|
||||
name: tests-${{inputs.par}}-${{inputs.sanitizer}}
|
||||
|
||||
@@ -17,17 +17,7 @@ permissions:
|
||||
on:
|
||||
push:
|
||||
branches: ["master", "next"]
|
||||
paths-ignore: &docs-only-paths
|
||||
- "**/*.md"
|
||||
- "doc/**"
|
||||
- ".binder/**"
|
||||
- "CITATION.cff"
|
||||
- "LICENSE"
|
||||
- "NOTICE"
|
||||
- "CHANGELOG"
|
||||
- "INSTALL"
|
||||
pull_request:
|
||||
paths-ignore: *docs-only-paths
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
|
||||
@@ -260,7 +260,6 @@ miniapps/meshing/polar-nc
|
||||
miniapps/meshing/mesh-quality
|
||||
miniapps/meshing/hpref
|
||||
miniapps/meshing/phpref
|
||||
miniapps/meshing/pref321
|
||||
miniapps/meshing/mobius-strip.mesh
|
||||
miniapps/meshing/klein-bottle.mesh
|
||||
miniapps/meshing/toroid-*.mesh
|
||||
|
||||
@@ -102,14 +102,12 @@ report_baseline:
|
||||
mkdir -p ${MACHINE_NAME}
|
||||
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-${BASELINE_TEST}-${CI_COMMIT_REF_SLUG}"
|
||||
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir ${rundir})
|
||||
status=0
|
||||
cp ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/* ${rundir} || { status=1; }
|
||||
cp ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/* ${rundir}
|
||||
printf "%s\n" "" "Pipeline URL:" "$CI_PIPELINE_URL" \
|
||||
>> ${rundir}/pipeline.txt
|
||||
# We create an autotest-email.html file, because that's how we signal
|
||||
# that there was an error / diff (temporary).
|
||||
if [[ $status -ne 0 ]] || \
|
||||
[[ -f ${rundir}/${BASELINE_TEST}.err ]] || \
|
||||
if [[ -f ${rundir}/${BASELINE_TEST}.err ]] || \
|
||||
[[ -f ${rundir}/${BASELINE_TEST}-${MACHINE_NAME}.diff ]]; then
|
||||
cp ${rundir}/pipeline.txt ${rundir}/autotest-email.html
|
||||
fi
|
||||
|
||||
@@ -11,160 +11,36 @@
|
||||
Version 4.9.1 (development)
|
||||
===========================
|
||||
|
||||
- Added policy for AI-assisted contribution to CONTRIBUTING.md.
|
||||
- Policy for AI-assisted contribution added to CONTRIBUTING.md
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Improved FindPointsGSLIB surface mesh capability with support for simplices
|
||||
and an option to specify axis-aligned bounding box padding for near-surface
|
||||
point queries.
|
||||
- Replaced legacy simplex quadrature rules with symmetric positive-weight
|
||||
rules for triangles (orders 0-25) and tetrahedra (orders 0-20). These
|
||||
rules guarantee all-positive weights and interior quadrature points,
|
||||
improving numerical stability. Higher orders fall back to Grundmann-Moller.
|
||||
Triangle rules: Witherden & Vincent, Comput. Math. Appl. 69(10):1232-1241,
|
||||
2015.
|
||||
Tet rules (d=1-13): Witherden & Vincent (ibid).
|
||||
Tet rules (d=14-20): Chuluunbaatar et al., Comput. Math. Appl. 124:89-97,
|
||||
2022.
|
||||
|
||||
- Added GPU-enabled partial assembly for simplicial Bernstein H1 basis based on
|
||||
ragged tensor algorithms (see DOI: 10.1137/11082539X) for mass and diffusion
|
||||
integrators.
|
||||
|
||||
- Replaced legacy simplex quadrature rules with symmetric positive weight rules
|
||||
for triangles (orders 0-25) and tetrahedra (orders 0-20). These rules
|
||||
guarantee all-positive weights and interior quadrature points, improving
|
||||
numerical stability. Higher orders fall back to Grundmann-Moller.
|
||||
* Triangle rules: Witherden and Vincent, DOI: 10.1016/j.camwa.2015.03.017
|
||||
* Tet rules (d=1-13): Witherden and Vincent (same as above)
|
||||
* Tet rules (d=14-20): Chuluunbaatar et al., DOI: 10.1016/j.camwa.2022.08.016
|
||||
|
||||
- Added support for general 1D Gauss-Jacobi quadrature rules and Stroud conical
|
||||
quadrature rules on triangles and tetrahedra.
|
||||
|
||||
- Improved the GridFunction projection routines. Projections work for Scalar,
|
||||
- Improved the gridfunction projection routines. Projections work for Scalar,
|
||||
Vector and VectorFE, also NURBS versions. Optionally different types of
|
||||
projections can be selected, default behavior has not changed.
|
||||
projections can be selected, default behaviour has not changed.
|
||||
|
||||
- Added GridFunction projection methods for trace spaces, i.e., project
|
||||
coefficients on the mesh skeleton.
|
||||
|
||||
- Added methods to estimate function extremum using piecewise linear bounds plus
|
||||
- Added methods to estimate function extremum using piecewise linear bounds +
|
||||
recursive subdivision.
|
||||
|
||||
- Extend FindPointsGSLIB to support surface meshes.
|
||||
|
||||
- Added support for complex-valued mixed bilinear forms via the new classes
|
||||
MixedSesquilinearForm and ParMixedSesquilinearForm, mirroring the existing
|
||||
SesquilinearForm classes. Rectangular complex operators are now also
|
||||
handled correctly by ComplexSparseMatrix::GetSystemMatrix and
|
||||
ComplexHypreParMatrix::GetSystemMatrix, which previously assumed equal
|
||||
trial and test spaces.
|
||||
|
||||
- Added FiniteElementSpace::GetBoundaryLoopEdgeDofs to extract the edge DOFs on
|
||||
the perimeter loop of a set of boundary elements, with a ParFiniteElementSpace
|
||||
overload that reconciles the selection across processor boundaries so the
|
||||
result is partition invariant. This is useful for imposing boundary conditions
|
||||
on boundary edge DOFs.
|
||||
|
||||
- Added a MaxAbs reduction to GroupCommunicator that selects the signed value of
|
||||
largest magnitude across a group, keeping its sign. Equal-magnitude ties
|
||||
resolve deterministically to the positive value.
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
- Added support for nonuniform anisotropic mesh refinement on parallel quad/hex
|
||||
meshes with arbitrary spacing in each direction. This enables in particular
|
||||
3:1 refinement in parallel, as demonstrated in the new meshing miniapp pref321.
|
||||
|
||||
- Added option to guarantee mesh validity during TMOP-based r-adaptivity, using
|
||||
bounds on the determinant of the mesh transformation Jacobian.
|
||||
|
||||
- Added PA support for TMOP's adaptive limiting functionality. Multiple
|
||||
GridFunctions and Coefficients can be combined to form a composite term.
|
||||
|
||||
- Improved support for 1D NURBS meshes with variable order, including using
|
||||
the patches construct for 1D NURBS meshes.
|
||||
|
||||
- Added the option to include material interfaces (faces separating elements
|
||||
with different element attributes) as additional boundary elements, for
|
||||
parallel visualization, e.g. with GLVis. This is supported by both the Print
|
||||
and PrintAsOne methods of ParMesh. See ParMesh::SetPrintInterfaces().
|
||||
|
||||
Linear and nonlinear solvers
|
||||
----------------------------
|
||||
- Added support for trace spaces in PRefinementTransferOperator. This is used in
|
||||
PRefinement multigrid methods for problems posed on trace spaces (see e.g. the
|
||||
DPG miniapps).
|
||||
|
||||
- Added new class MultiVector: an array of Vectors of different sizes where each
|
||||
Vector can be allocated independently. Also, added associated methods in class
|
||||
Operator: MultMV, MultTransposeMV, and GetGradientMV, that use MultiVector
|
||||
objects for input and/or output parameters. [PR #5249]
|
||||
|
||||
GPU computing
|
||||
-------------
|
||||
- Improved partial assembly for VectorDivergenceIntegrator with shared-memory
|
||||
kernels, kernel registration, and transpose support.
|
||||
|
||||
- Improved partial-assembly diagonal kernels for VectorMassIntegrator (shared-
|
||||
memory specializations) and ElasticityIntegrator (no scratch Q-vector).
|
||||
|
||||
- Added PA gradient and diagonal support for VectorConvectionNLFIntegrator
|
||||
(AssembleGradPA, AddMultGradPA, AssembleGradDiagonalPA).
|
||||
|
||||
- Added device assembly support for 3D H(curl) VectorFEDomainLFIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedScalarWeakGradientIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedDotProductIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedScalarCrossProductIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedScalarWeakCrossProductIntegrator.
|
||||
|
||||
- Added partial assembly support for MixedVectorGradientIntegrator for H1->RT.
|
||||
|
||||
- Added support for device partial assembly CurlInterpolator.
|
||||
This supports 2D and 3D variants:
|
||||
2D H1 (out-of-plane) to RT (in-plane)
|
||||
2D ND (in-plane) to Integral L2 (out-of-plane)
|
||||
3D ND to RT
|
||||
|
||||
- Added NVIDIA cuDSS library interface. Implementation examples have been
|
||||
added to ex1 and ex1p. See https://developer.nvidia.com/cudss for more
|
||||
details. Supported versions >= 0.6.0.
|
||||
|
||||
- Allow specifying GPU kernel launch bounds for native and RAJA GPU backends.
|
||||
|
||||
- Changed VectorFEMassIntegrator to use kernel specialization dispatch for
|
||||
partial assembly.
|
||||
|
||||
- Added support for FiniteElement::MapType::INTEGRAL spaces to
|
||||
QuadratureInterpolator.
|
||||
|
||||
- Added support for FiniteElement::MapType::INTEGRAL spaces to
|
||||
MixedScalarCurlIntegrator.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- The Lorentz miniapp (in miniapps/electromagnetics) has been updated to
|
||||
leverage the ParticleSet capability.
|
||||
|
||||
- Added (Complex)PRefinementMultigrid solver option in the DPG miniapps.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Fixed signed DOF handling in ParGridFunction reading (read constructor) and
|
||||
saving via SaveAsOne(). Simplified the process of applying the DOF signs by
|
||||
using the new method ApplyDofSigns() in class ParFiniteElementSpace: the
|
||||
method will return immediately if no sign flips are needed.
|
||||
|
||||
- Added support for coefficient-weighted LOR transfer in
|
||||
L2ProjectionGridTransfer. The transfer conserves the weighted mass, for
|
||||
example when transferring velocity while conserving density-weighted momentum.
|
||||
This is illustrated in the lor-transfer and plor-transfer miniapps.
|
||||
|
||||
- Added support for saving DataCollection output on the node-local storage,
|
||||
instead of requiring that the filesystem is shared among all the ranks.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- Removed ProjectGrad from 2D RT elements. Users should use ProjectCurl instead.
|
||||
This also fixes a bug where ProjectCurl was returning the negative curl,
|
||||
identical to ProjectGrad.
|
||||
- Electromagnetics/lorentz miniapp has been updated to leverage the ParticleSet
|
||||
capability.
|
||||
|
||||
|
||||
Version 4.9, released on Dec 11, 2025
|
||||
|
||||
+13
-20
@@ -88,9 +88,18 @@ if (MFEM_USE_STRUMPACK OR MFEM_USE_MUMPS)
|
||||
# Just needed to find the MPI_Fortran libraries to link with
|
||||
set(XSDK_ENABLE_Fortran ON)
|
||||
endif()
|
||||
# RAJA requires C++20:
|
||||
if ((MFEM_USE_UMPIRE OR MFEM_USE_RAJA) AND ("${CMAKE_CXX_STANDARD}" LESS "20"))
|
||||
set(CMAKE_CXX_STANDARD 20 CACHE STRING "C++ standard to use." FORCE)
|
||||
# Ginkgo requires C++17:
|
||||
if ((MFEM_USE_GINKGO) AND ("${CMAKE_CXX_STANDARD}" LESS "17"))
|
||||
set(CMAKE_CXX_STANDARD 17 CACHE STRING "C++ standard to use." FORCE)
|
||||
# Google Benchmark, SUNDIALS, STRUMPACK, Tribol, RAJA and Umpire require C++14:
|
||||
elseif ((MFEM_USE_BENCHMARK OR
|
||||
MFEM_USE_SUNDIALS OR
|
||||
MFEM_USE_STRUMPACK OR
|
||||
MFEM_USE_TRIBOL OR
|
||||
MFEM_USE_RAJA OR
|
||||
MFEM_USE_UMPIRE) AND
|
||||
("${CMAKE_CXX_STANDARD}" LESS "14"))
|
||||
set(CMAKE_CXX_STANDARD 14 CACHE STRING "C++ standard to use." FORCE)
|
||||
endif()
|
||||
|
||||
# Include xSDK default CMake file.
|
||||
@@ -230,13 +239,6 @@ else()
|
||||
set(MFEM_DEBUG OFF)
|
||||
endif()
|
||||
|
||||
# Shadow warnings for clang only; GCC's -Wshadow flags more.
|
||||
if (CMAKE_CXX_COMPILER_ID MATCHES "Clang")
|
||||
set(CMAKE_CXX_FLAGS_DEBUG "${CMAKE_CXX_FLAGS_DEBUG} -pedantic -Wall -Wshadow")
|
||||
elseif (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
set(CMAKE_CXX_FLAGS_DEBUG "${CMAKE_CXX_FLAGS_DEBUG} -pedantic -Wall")
|
||||
endif()
|
||||
|
||||
# Shared build on Windows
|
||||
if (WIN32 AND BUILD_SHARED_LIBS)
|
||||
# CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS works only with MSVC?
|
||||
@@ -431,15 +433,6 @@ if (MFEM_USE_STRUMPACK)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# cuDSS can only be enabled in CUDA
|
||||
if (MFEM_USE_CUDSS)
|
||||
if (MFEM_USE_CUDA)
|
||||
find_package(CUDSS REQUIRED)
|
||||
else()
|
||||
message(FATAL_ERROR " *** cuDSS requires that CUDA be enabled.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# GnuTLS
|
||||
if (MFEM_USE_GNUTLS)
|
||||
find_package(_GnuTLS REQUIRED)
|
||||
@@ -638,7 +631,7 @@ find_package(Threads REQUIRED)
|
||||
set(MFEM_TPLS OPENMP HYPRE LAPACK BLAS SuperLUDist STRUMPACK METIS SuiteSparse
|
||||
SUNDIALS PETSC SLEPC MUMPS AXOM FMS CONDUIT Ginkgo GNUTLS GSLIB HDF5
|
||||
NETCDF MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE
|
||||
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CUDSS CALIPER CODIPACK
|
||||
ADIOS2 MKL_CPARDISO MKL_PARDISO AMGX MAGMA CUSPARSE CUBLAS CALIPER CODIPACK
|
||||
BENCHMARK PARELAG TRIBOL MPI_CXX HIP HIPBLAS HIPSPARSE MOONOLITH BLITZ
|
||||
ALGOIM ENZYME CUDA::cudart)
|
||||
|
||||
|
||||
+65
-64
@@ -3,12 +3,12 @@
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-blue.svg"></a>
|
||||
<a href="https://github.com/mfem/mfem/releases/latest"><img alt="GitHub release" src="https://img.shields.io/github/v/release/mfem/mfem"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions/workflows/repo-check.yml?query=branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml?query=branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-brightgreen.svg"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Arepo-check+branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuild-analysis+branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuilds-and-tests+branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
|
||||
<a href="https://docs.mfem.org/html/index.html"><img alt="Documentation" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
<a href="https://docs.mfem.org/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
</p>
|
||||
|
||||
|
||||
@@ -84,7 +84,7 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
follow the [MFEM PR Rules](#mfem-pr-rules).
|
||||
- When your contribution is fully working and ready to be reviewed, add
|
||||
the `ready-for-review` label.
|
||||
- PRs are treated similarly to journal submission, with an "editor" assigning two
|
||||
- PRs are treated similarly to journal submission with an "editor" assigning two
|
||||
reviewers to evaluate the changes.
|
||||
- The reviewers have 3 weeks to evaluate the PR and work with the author to
|
||||
fix issues and implement improvements.
|
||||
@@ -125,7 +125,7 @@ The MFEM source code has the following structure:
|
||||
│ ├── petsc
|
||||
│ ├── pumi
|
||||
│ ├── sundials
|
||||
│ └── superlu
|
||||
| └── superlu
|
||||
├── fem
|
||||
│ ├── ceed
|
||||
│ ├── dfem
|
||||
@@ -137,6 +137,10 @@ The MFEM source code has the following structure:
|
||||
│ ├── moonolith
|
||||
│ ├── qinterp
|
||||
│ └── tmop
|
||||
│ | ├── assemble
|
||||
│ | ├── metrics
|
||||
│ | ├── mult
|
||||
│ | └── tools
|
||||
├── general
|
||||
├── linalg
|
||||
│ ├── batched
|
||||
@@ -149,10 +153,11 @@ The MFEM source code has the following structure:
|
||||
│ ├── common
|
||||
│ ├── contact
|
||||
│ ├── dfem
|
||||
│ ├── diag-smoothers
|
||||
│ ├── dpg
|
||||
│ ├── electromagnetics
|
||||
│ ├── fluids
|
||||
│ │ ├── navier
|
||||
│ │ └── schrodinger-flow
|
||||
│ ├── gslib
|
||||
│ ├── hdiv-linear-solver
|
||||
│ ├── hooke
|
||||
@@ -162,7 +167,6 @@ The MFEM source code has the following structure:
|
||||
│ ├── nurbs
|
||||
│ ├── parelag
|
||||
│ ├── performance
|
||||
│ ├── plasma
|
||||
│ ├── shifted
|
||||
│ ├── solvers
|
||||
│ ├── spde
|
||||
@@ -193,15 +197,15 @@ respectively.
|
||||
|
||||
- The main finite element classes are:
|
||||
+ [`FiniteElement`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElementCollection.html)
|
||||
+ [`FiniteElementCollection`](https://docs.mfem.org/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1FiniteElementSpace.html)
|
||||
+ [`GridFunction`](https://docs.mfem.org/html/classmfem_1_1GridFunction.html)
|
||||
+ [`BilinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1BilinearFormIntegrator.html) and [`LinearFormIntegrator`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html)
|
||||
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
|
||||
+ [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearFormIntegrator.html), [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://docs.mfem.org/html/classmfem_1_1MixedBilinearForm.html)
|
||||
|
||||
- The main linear algebra classes and sources are
|
||||
+ [`Operator`](https://docs.mfem.org/html/classmfem_1_1Operator.html) and [`BilinearForm`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html)
|
||||
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1Vector.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
|
||||
+ [`Vector`](https://docs.mfem.org/html/classmfem_1_1BilinearForm.html) and [`LinearForm`](https://docs.mfem.org/html/classmfem_1_1LinearForm.html)
|
||||
+ [`DenseMatrix`](https://docs.mfem.org/html/classmfem_1_1DenseMatrix.html) and [`SparseMatrix`](https://docs.mfem.org/html/classmfem_1_1SparseMatrix.html)
|
||||
+ Sparse [smoothers](https://docs.mfem.org/html/sparsesmoothers_8hpp.html) and linear [solvers](https://docs.mfem.org/html/solvers_8hpp.html)
|
||||
|
||||
@@ -213,8 +217,8 @@ shared geometric entities between different tasks. The parallel source files
|
||||
have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
|
||||
|
||||
- The main parallel classes are
|
||||
+ [`ParMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
|
||||
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParNCMesh.html)
|
||||
+ [`ParMesh`](https://docs.mfem.org/html/solvers_8hpp.html)
|
||||
+ [`ParNCMesh`](https://docs.mfem.org/html/classmfem_1_1ParMesh.html)
|
||||
+ [`ParFiniteElementSpace`](https://docs.mfem.org/html/classmfem_1_1ParFiniteElementSpace.html)
|
||||
+ [`ParGridFunction`](https://docs.mfem.org/html/classmfem_1_1ParGridFunction.html)
|
||||
+ [`ParBilinearForm`](https://docs.mfem.org/html/classmfem_1_1ParBilinearForm.html) and [`ParLinearForm`](https://docs.mfem.org/html/classmfem_1_1ParLinearForm.html)
|
||||
@@ -224,14 +228,14 @@ have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
|
||||
#### GPU and general device support
|
||||
|
||||
GPU and multi-core CPU support is based on device kernels supporting different
|
||||
backends (CUDA, HIP, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
|
||||
backends (CUDA, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
|
||||
device/host memory manager.
|
||||
|
||||
- The main device-relevant classes and sources are:
|
||||
+ [`Device`](https://docs.mfem.org/html/device_8hpp.html)
|
||||
+ [`MemoryManager`](https://docs.mfem.org/html/mem_manager_8hpp.html)
|
||||
+ the [`mfem::forall`](https://docs.mfem.org/html/forall_8hpp.html) function
|
||||
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html), [`hip.hpp`](https://docs.mfem.org/html/hip_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
|
||||
+ the [`cuda.hpp`](https://docs.mfem.org/html/cuda_8hpp.html) and [`occa.hpp`](https://docs.mfem.org/html/occa_8hpp.html) files
|
||||
|
||||
#### Utilities, building and documentation
|
||||
- The `general/` directory contains C++ classes that serve as utilities for
|
||||
@@ -245,8 +249,8 @@ device/host memory manager.
|
||||
- `examples` and `miniapps` respectively gather simple and more fully-featured
|
||||
demonstrations of the usage on MFEM. They both rely on `data/` for the
|
||||
collection of meshes.
|
||||
- The `tests/` directory contains a unit test suite, additional tests, and
|
||||
benchmarks.
|
||||
- The `tests/` directory contains a unit test suite and will later contain more
|
||||
tests that run example codes.
|
||||
|
||||
See also the [code overview](https://mfem.org/code-overview/) section on the MFEM
|
||||
website.
|
||||
@@ -280,8 +284,8 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
the top of https://github.com/mfem.
|
||||
- Consider making your membership public by going to https://github.com/orgs/mfem/people
|
||||
and clicking on the organization visibility drop box next to your name.
|
||||
- Project discussions and announcements will be posted at https://github.com/orgs/mfem/discussions,
|
||||
tagging the `@mfem/everyone` team when appropriate.
|
||||
- Project discussions and announcements will be posted at
|
||||
https://github.com/orgs/mfem/teams/everyone.
|
||||
|
||||
#### Structure
|
||||
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
|
||||
@@ -341,12 +345,11 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
- Well-designed simple code is frequently more general and powerful.
|
||||
- Lean code base is easier to understand by new collaborators.
|
||||
- New features should be added only if they are necessary or generally useful.
|
||||
- Introduction of language constructs not currently used in MFEM should be
|
||||
- Introduction of language constructions not currently used in MFEM should be
|
||||
justified and generally avoided (to maintain portability to various systems
|
||||
and compilers, including early access hardware).
|
||||
- We prefer basic C++. Use C++17 features judiciously, prioritizing readability,
|
||||
consistency with existing MFEM code, and portability to different systems,
|
||||
compilers and device backends.
|
||||
- We prefer basic C++ and the C++03 standard, to keep the code readable by
|
||||
a large audience and to make sure it compiles anywhere.
|
||||
|
||||
- *Keep the code general and reasonably efficient*
|
||||
- The main goal is fast prototyping for research and application development.
|
||||
@@ -389,7 +392,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
- When your branch is ready for other developers to review / comment on
|
||||
the code, create a pull request towards `mfem:master`.
|
||||
|
||||
- Pull requests typically have titles like:
|
||||
- Pull request typically have titles like:
|
||||
|
||||
`Description [new-feature-dev]`
|
||||
|
||||
@@ -410,12 +413,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
|
||||
team will add reviewers as appropriate.
|
||||
|
||||
- List outstanding TODO items in the description.
|
||||
- List outstanding TODO items in the description, see PR #222 for an example.
|
||||
|
||||
- When your contribution is fully working and ready to be reviewed, add
|
||||
or request the `ready-for-review` label.
|
||||
the `ready-for-review` label.
|
||||
|
||||
- PRs are treated similarly to journal submission, with an "editor" assigning
|
||||
- PRs are treated similarly to journal submission with an "editor" assigning
|
||||
two reviewers to evaluate the changes. The reviewers have 3 weeks to evaluate
|
||||
the PR and work with the author to implement improvements and fix issues.
|
||||
|
||||
@@ -441,7 +444,7 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
checks in GitHub Actions enforce MFEM-specific rules which are explained in
|
||||
the error messages and the `tests/scripts` directory.
|
||||
|
||||
- Also note that the tests `branch-history` and `repo-check` found in GitHub
|
||||
- Also note that the tests `branch-history` and `repos-checks` found in GitHub
|
||||
Actions can be triggered automatically before each push using git hooks. See
|
||||
the [git hooks README](config/githooks/README.md) for a detailed explanation.
|
||||
|
||||
@@ -498,15 +501,15 @@ Everyone on the MFEM team can be asked to serve as a reviewer on a PR in their a
|
||||
|
||||
3. To ensure the quality of the PR by making sure that the code adheres to the [Developer Guidelines](#developer-guidelines), e.g. all methods, data members, and functions have documentation, including data ownership and lifetime, new examples/miniapps have a corresponding PR in mfem/web, major features have `CHANGELOG` entries, etc.
|
||||
|
||||
4. To seek help from the editors in case of difficulties.
|
||||
3. To seek help from the editors in case of difficulties.
|
||||
|
||||
5. To complete the review in a timely manner: 3 weeks from assignment.
|
||||
4. To complete the review in a timely manner: 3 weeks from assignment.
|
||||
|
||||
6. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
|
||||
5. To test the PR thoroughly before merging in *next*. The PR author is also encouraged to perform testing and inform the reviewers about the results.
|
||||
|
||||
7. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
|
||||
6. To monitor the PR impact on the testing in the *next* branch and alert the editors that the PR is ready for merging in *master*.
|
||||
|
||||
8. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
|
||||
7. The review of bugfixes should be expedited proportional to their importance. The review window can be much less than three weeks in such cases.
|
||||
|
||||
#### Responsibilities of Authors
|
||||
|
||||
@@ -532,30 +535,30 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Code builds.
|
||||
- [ ] Code passes `make style`.
|
||||
- [ ] Update `CHANGELOG`:
|
||||
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
|
||||
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
|
||||
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
|
||||
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
|
||||
- [ ] Update `INSTALL`:
|
||||
- [ ] Has a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
|
||||
- [ ] Have the version ranges for any required or optional libraries changed?
|
||||
- [ ] Does `make` or `cmake` have a new target?
|
||||
- [ ] Did the requirements or the installation process change? *(rare)*
|
||||
- [ ] Had a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
|
||||
- [ ] Have the version ranges for any required or optional libraries changed?
|
||||
- [ ] Does `make` or `cmake` have a new target?
|
||||
- [ ] Did the requirements or the installation process change? *(rare)*
|
||||
- [ ] Update continuous integration server configurations if necessary (e.g. with new version requirements for each of MFEM's dependencies)
|
||||
- [ ] `.github`
|
||||
- [ ] `.appveyor.yml`
|
||||
- [ ] `.github`
|
||||
- [ ] `.appveyor.yml`
|
||||
- [ ] Update `.gitignore`:
|
||||
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
|
||||
- [ ] Add new patterns (just for the new files above) and re-run the above test.
|
||||
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
|
||||
- [ ] Add new patterns (just for the new files above) and re-run the above test.
|
||||
- [ ] New examples:
|
||||
- [ ] All sample runs at the top of the example source file work.
|
||||
- [ ] Update `examples/makefile`:
|
||||
- [ ] All sample runs at the top of the example source file work.
|
||||
- [ ] Update `examples/makefile`:
|
||||
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
|
||||
- [ ] Add any files generated by it to the `clean` target.
|
||||
- [ ] Add the example binary and any files generated by it to the top-level `.gitignore` file.
|
||||
- [ ] Update `examples/CMakeLists.txt`:
|
||||
- [ ] Update `examples/CMakeLists.txt`:
|
||||
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
|
||||
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
|
||||
- [ ] List the new example in `doc/CodeDocumentation.dox`.
|
||||
- [ ] If new examples directory (e.g. `examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] If new examples directory (e.g.`examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
@@ -572,13 +575,13 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
|
||||
- [ ] Consider adding a new test for the new miniapp.
|
||||
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
|
||||
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
|
||||
- [ ] If new miniapps directory (e.g. `miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
|
||||
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
|
||||
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] The miniapps go at the end of the page, and are usually listed only under a specific "Application (PDE)" category.
|
||||
- [ ] Add a short description of the miniapp in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] New capability:
|
||||
- [ ] All new public, protected, and private classes, methods, data members, and functions have full Doxygen-style documentation in source comments. Documentation should include descriptions of member data, function arguments and return values, template parameters, and prerequisites for calling new functions.
|
||||
- [ ] Pointer arguments and return values must specify whether ownership is being transferred or lent with the call.
|
||||
@@ -680,7 +683,7 @@ MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
- [ ] Update URL shortlinks:
|
||||
- [ ] Create a shortlink at [http://bit.ly/](http://bit.ly/) for the release tarball, e.g. https://mfem.github.io/releases/mfem-3.1.tgz.
|
||||
- [ ] (LLNL only) Add and commit the new shortlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
|
||||
- [ ] Add the new shortlinks to the MFEM package in `spack`.
|
||||
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
|
||||
- [ ] Update website in `mfem/web` repo:
|
||||
- Update version and shortlinks in `src/index.md` and `src/download.md`.
|
||||
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
|
||||
@@ -732,24 +735,22 @@ commit or push, see the [README](config/githooks/README.md) in the `config/githo
|
||||
directory.
|
||||
|
||||
|
||||
### GitHub Actions smoke tests
|
||||
|
||||
### Linux and Mac smoke tests
|
||||
We use GitHub Actions to drive the default tests on the `master` and `next`
|
||||
branches. See the `.github/workflows` files and the logs at
|
||||
[https://github.com/mfem/mfem/actions](https://github.com/mfem/mfem/actions).
|
||||
|
||||
GitHub Actions testing should be kept lightweight, as there is a time
|
||||
constraint on jobs. The current workflows cover Linux, macOS, and Windows
|
||||
configurations.
|
||||
Testing using GitHub Actions should be kept lightweight, as there is a time
|
||||
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
|
||||
|
||||
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
|
||||
- Tests on the `next` branch are currently scheduled to run each night.
|
||||
|
||||
### Additional Windows smoke test
|
||||
|
||||
We also use Appveyor to test building with the MS Visual C++ compiler in a Windows
|
||||
environment, as well as to test the CMake build. See the `.appveyor.yml` file
|
||||
and the build logs at
|
||||
### Windows smoke test
|
||||
We use Appveyor to test building with the MS Visual C++ compiler in a Windows
|
||||
environment, as well as to test the CMake build. See the `.appveyor` file and the
|
||||
build logs at
|
||||
[https://ci.appveyor.com/project/mfem/mfem](https://ci.appveyor.com/project/mfem/mfem).
|
||||
|
||||
CMake is used to generate the MSVC Project files and drive the build. A release
|
||||
|
||||
@@ -38,13 +38,14 @@ the option MFEM_USE_METIS.
|
||||
MFEM also includes support for devices such as GPUs, and programming models such
|
||||
as CUDA, HIP, OCCA, OpenMP and RAJA.
|
||||
|
||||
- Starting with version 4.9, MFEM requires a C++17 compiler.
|
||||
- Starting with version 4.0, MFEM requires a C++11 compiler. We recommend using
|
||||
a newer compiler, e.g. GCC version 4.9 or higher.
|
||||
|
||||
- CUDA support requires an NVIDIA GPU and an installation of the CUDA Toolkit
|
||||
https://developer.nvidia.com/cuda-toolkit
|
||||
|
||||
- HIP support requires an AMD GPU and an installation of the ROCm software stack
|
||||
https://rocm.docs.amd.com
|
||||
https://rocmdocs.amd.com
|
||||
|
||||
- OCCA support requires the OCCA library
|
||||
https://libocca.org
|
||||
@@ -82,9 +83,9 @@ Serial build:
|
||||
Parallel build:
|
||||
(download hypre and METIS 4 from above URLs)
|
||||
(build METIS 4 in ../metis-4.0 relative to mfem/)
|
||||
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
(build hypre in ../hypre relative to mfem/)
|
||||
make parallel -j 4
|
||||
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
|
||||
CUDA build:
|
||||
make cuda -j 4
|
||||
@@ -114,14 +115,14 @@ Serial build:
|
||||
Parallel build:
|
||||
(download hypre and METIS 4 from above URLs)
|
||||
(build METIS 4 in ../metis-4.0 relative to mfem/)
|
||||
(for METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
(build hypre in ../hypre relative to mfem/)
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES
|
||||
make -j 4
|
||||
(For METIS 5, see https://mfem.org/building/#parallel-build-using-metis-5)
|
||||
|
||||
Parallel build with fetching of hypre and METIS:
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DMFEM_FETCH_TPLS=YES
|
||||
make -j 4
|
||||
|
||||
@@ -133,8 +134,7 @@ CUDA build:
|
||||
|
||||
HIP build:
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 \
|
||||
-DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
|
||||
cmake <mfem-source-dir> -DMFEM_USE_HIP=YES -DHIP_ARCH=gfx942 -DCMAKE_CXX_COMPILER=amdclang++ -DCMAKE_HIP_COMPILER=amdclang++
|
||||
make -j 4
|
||||
|
||||
Example codes (serial/parallel, depending on the build):
|
||||
@@ -269,7 +269,6 @@ Compilers:
|
||||
CXX - C++ compiler, serial build
|
||||
MPICXX - MPI C++ compiler, parallel build
|
||||
CUDA_CXX - The CUDA compiler, 'nvcc' or 'clang++'
|
||||
HIP_CXX - The HIP compiler, e.g. 'hipcc'
|
||||
|
||||
Compiler options:
|
||||
OPTIM_FLAGS - Options for optimized build
|
||||
@@ -396,11 +395,6 @@ MFEM_USE_STRUMPACK = YES/NO
|
||||
classes. When enabled, this option uses the STRUMPACK_* library options, see
|
||||
below.
|
||||
|
||||
MFEM_USE_CUDSS = YES/NO
|
||||
Enable MFEM functionality based on the cuDSS library. When using cuDSS, CUDA
|
||||
support must be also enabled in MFEM, i.e. MFEM_USE_CUDA=YES must be set.
|
||||
When enabled, this option uses the CUDSS_* library options, see below.
|
||||
|
||||
MFEM_USE_GINKGO = YES/NO
|
||||
Enable MFEM functionality based on the Ginkgo library, which provides
|
||||
iterative linear solvers and preconditioners with OpenMP, CUDA backends, see
|
||||
@@ -560,13 +554,13 @@ MFEM_USE_RAJA = YES/NO
|
||||
MFEM_USE_OCCA = YES/NO
|
||||
Enables support for the OCCA library in MFEM. OCCA is an open-source library
|
||||
which aims to make it easy to program different types of devices (e.g. CPU,
|
||||
GPU, FPGA) by providing a unified API for interacting with JIT-compiled
|
||||
GPU, FPGA) by providing an unified API for interacting with JIT-compiled
|
||||
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
|
||||
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
|
||||
|
||||
MFEM_USE_GSLIB = YES/NO
|
||||
Enables MFEM functionality based on the GSLIB library, and specifically its
|
||||
FindPoints component, which provides robust algorithms to evaluate finite
|
||||
FindPoints component, which provides a robust algorithms to evaluate finite
|
||||
element functions in a collection of points in physical space. When enabled,
|
||||
the user can use the GSLIB-FindPoints methods as shown in miniapps/gslib.
|
||||
|
||||
@@ -725,18 +719,9 @@ The specific libraries and their options are:
|
||||
Options: STRUMPACK_OPT, STRUMPACK_LIB.
|
||||
Versions: STRUMPACK >= 3.0.0.
|
||||
|
||||
- CUDSS (optional), used when MFEM_USE_CUDSS = YES. Note that CUDSS requires
|
||||
CUDA 12.x toolkit and the cuDSS libraries. The supported communication backend
|
||||
is OpenMPI 4.x (default), and OpenMPI 4.x or a later version must be pre-built.
|
||||
The source files in the cuDSS tarball provide guidance for developing custom
|
||||
MPI implementations.
|
||||
URL: https://developer.nvidia.com/cudss
|
||||
https://docs.nvidia.com/cuda/cudss/advanced_features.html#communication-layer-library-in-cudss
|
||||
Options: CUDSS_OPT, CUDSS_LIB.
|
||||
Versions: cuDSS >= 0.6.0.
|
||||
|
||||
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Ginkgo may have additional
|
||||
requirements and module-specific dependencies; see the webpage below.
|
||||
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Note that Ginkgo needs a
|
||||
C++ compiler that supports the C++-17 standard. For additional requirements
|
||||
and dependencies of specific modules, see the Ginkgo webpage below.
|
||||
URL: https://ginkgo-project.github.io
|
||||
Options: GINKGO_OPT, GINKGO_LIB, GINKGO_DIR, GINKGO_BUILD_TYPE (Release or
|
||||
Debug).
|
||||
@@ -808,7 +793,7 @@ The specific libraries and their options are:
|
||||
Options: CONDUIT_OPT, CONDUIT_LIB.
|
||||
Versions: Conduit >= 0.3.1.
|
||||
|
||||
- ADIOS2 (optional), used when MFEM_USE_ADIOS2 = YES.
|
||||
- ADIOS2 (optional) used when MFEM_USE_ADIOS2 = YES.
|
||||
URL: https://adios2.readthedocs.io/
|
||||
Versions: ADIOS >= 2.5.0.
|
||||
|
||||
@@ -884,7 +869,7 @@ The specific libraries and their options are:
|
||||
Options: RAJA_DIR, RAJA_OPT, RAJA_LIB.
|
||||
Versions: RAJA >= 2022.10.3.
|
||||
|
||||
- Moonolith (optional), used when MFEM_USE_MOONOLITH = YES.
|
||||
- Moonolith (optional), use when MFEM_USE_MOONOLITH = YES.
|
||||
URL: https://bitbucket.org/zulianp/par_moonolith
|
||||
Options: MOONOLITH_DIR
|
||||
Versions: MOONOLITH >= 1.1.0.
|
||||
@@ -972,7 +957,7 @@ CMAKE_BUILD_TYPE which can be set to standard values like "Debug", and "Release"
|
||||
To use a specific generator use the "-G <generator>" option of cmake:
|
||||
|
||||
cmake <mfem-source-dir> -G "Xcode"
|
||||
cmake <mfem-source-dir> -G "Visual Studio 17 2022"
|
||||
cmake <mfem-source-dir> -G "Visual Studio 12 2013"
|
||||
cmake <mfem-source-dir> -G "MinGW Makefiles"
|
||||
|
||||
With CMake it is possible to build MFEM as a shared library using the standard
|
||||
@@ -1217,7 +1202,7 @@ larger problems, there are two options:
|
||||
Specific options for HIP
|
||||
========================
|
||||
MFEM expects the `ROCM_PATH` environment variable to be set to the path of the
|
||||
ROCm install, as well as having `$ROCM_PATH/bin` in `PATH`.
|
||||
ROCM install, as well as having `$ROCM_PATH/bin` in `PATH`.
|
||||
|
||||
Specific options for RAJA+HIP+MPI
|
||||
=================================
|
||||
|
||||
@@ -28,7 +28,6 @@ license files. These software products and their licenses are as follows:
|
||||
* AmgXWrapper (linalg/amgxsolver.{hpp,cpp}) -- MIT license
|
||||
* Catch++ (tests/unit/catch.hpp) -- Boost 1.0 license
|
||||
* Gecko (general/gecko.{cpp,hpp}) -- BSD 3-clause license
|
||||
* gslib (fem/gslib.{cpp,hpp}, mesh/bb_grid_map.{cpp,hpp}) -- BSD 3-clause license
|
||||
* Picojson (fem/picojson.h) -- Custom 2-clause license
|
||||
* TinyXML2 (general/tinyxml2.{cpp,h}) -- zlib license
|
||||
* Zstr (general/zstr.hpp) -- MIT license
|
||||
|
||||
@@ -35,7 +35,6 @@ set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
|
||||
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
|
||||
set(MFEM_USE_MUMPS @MFEM_USE_MUMPS@)
|
||||
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
|
||||
set(MFEM_USE_CUDSS @MFEM_USE_CUDSS@)
|
||||
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
|
||||
set(MFEM_USE_AMGX @MFEM_USE_AMGX@)
|
||||
set(MFEM_USE_MAGMA @MFEM_USE_MAGMA@)
|
||||
@@ -110,10 +109,6 @@ if (MFEM_USE_RAJA)
|
||||
find_dependency(RAJA)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_CUDSS)
|
||||
find_dependency(cudss)
|
||||
endif (MFEM_USE_CUDSS)
|
||||
|
||||
if (MFEM_USE_UMPIRE)
|
||||
find_dependency(umpire)
|
||||
endif()
|
||||
|
||||
@@ -108,15 +108,6 @@
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
#cmakedefine MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable MFEM functionality based on the cuDSS library.
|
||||
#cmakedefine MFEM_USE_CUDSS
|
||||
|
||||
// CUDSS communication layer library path
|
||||
#cmakedefine MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
|
||||
|
||||
// CUDSS threading layer library path
|
||||
#cmakedefine MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
|
||||
|
||||
// Enable functionality based on the Ginkgo library.
|
||||
#cmakedefine MFEM_USE_GINKGO
|
||||
|
||||
|
||||
@@ -1,68 +0,0 @@
|
||||
if (NOT cudss_DIR AND CUDSS_DIR)
|
||||
set(cudss_DIR ${CUDSS_DIR}/lib/cmake/cudss)
|
||||
endif()
|
||||
message(STATUS "Looking for CUDSS ...")
|
||||
message(STATUS " in CUDSS_DIR = ${CUDSS_DIR}")
|
||||
message(STATUS " cudss_DIR = ${cudss_DIR}")
|
||||
find_package(cudss)
|
||||
set(CUDSS_FOUND ${cudss_FOUND})
|
||||
set(CUDSS_LIBRARIES "cudss")
|
||||
if (CUDSS_FOUND)
|
||||
message(STATUS
|
||||
"Found CUDSS target: ${CUDSS_LIBRARIES} (version: ${cudss_VERSION})")
|
||||
else()
|
||||
set(msg STATUS)
|
||||
if (CUDSS_FIND_REQUIRED)
|
||||
set(msg FATAL_ERROR)
|
||||
endif()
|
||||
message(${msg}
|
||||
"CUDSS not found. Please set CUDSS_DIR to the install prefix.")
|
||||
endif()
|
||||
|
||||
if(CUDSS_FOUND AND TARGET cudss)
|
||||
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION)
|
||||
if(NOT CUDSS_LIBRARY_LOCATION)
|
||||
get_target_property(CUDSS_LIBRARY_LOCATION cudss IMPORTED_LOCATION_RELEASE)
|
||||
endif()
|
||||
if(CUDSS_LIBRARY_LOCATION)
|
||||
get_filename_component(CUDSS_LIBRARY_DIR "${CUDSS_LIBRARY_LOCATION}" DIRECTORY)
|
||||
else()
|
||||
message(WARNING "Could not determine the location of the cuDSS library.")
|
||||
endif()
|
||||
else()
|
||||
message(WARNING "cuDSS target not available; cannot determine library directory.")
|
||||
endif()
|
||||
|
||||
# Set the full name of the cuDSS threading library if OpenMP is enabled.
|
||||
# The threading layer library (libcudss_mtlayer_gomp.so) is located under the
|
||||
# cuDSS library directory by default.
|
||||
if (MFEM_USE_OPENMP)
|
||||
find_file(
|
||||
CUDSS_THREADING_LIB
|
||||
NAMES libcudss_mtlayer_gomp.so
|
||||
PATHS ${CUDSS_LIBRARY_DIR}
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
if (NOT DEFINED MFEM_CUDSS_THREADING_LIB AND CUDSS_THREADING_LIB)
|
||||
set(MFEM_CUDSS_THREADING_LIB "${CUDSS_THREADING_LIB}")
|
||||
endif()
|
||||
message(STATUS "CUDSS threading layer library: ${MFEM_CUDSS_THREADING_LIB}")
|
||||
endif()
|
||||
|
||||
# Set the full name of the cuDSS communication library if MFEM use OpenMPI.
|
||||
# The communication layer library (libcudss_commlayer_mpi.so) is located under the
|
||||
# cuDSS library directory by default.
|
||||
# The communication layer library is used pre-built communication layers for OpenMPI
|
||||
# by default.
|
||||
if (MFEM_USE_MPI)
|
||||
find_file(
|
||||
CUDSS_COMM_LIB
|
||||
NAMES libcudss_commlayer_openmpi.so
|
||||
PATHS ${CUDSS_LIBRARY_DIR}
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
if (NOT DEFINED MFEM_CUDSS_COMM_LIB AND CUDSS_COMM_LIB)
|
||||
set(MFEM_CUDSS_COMM_LIB "${CUDSS_COMM_LIB}")
|
||||
endif()
|
||||
message(STATUS "CUDSS communication layer library: ${MFEM_CUDSS_COMM_LIB}")
|
||||
endif()
|
||||
@@ -18,17 +18,19 @@
|
||||
|
||||
if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
|
||||
enable_language(C)
|
||||
set(GSLIB_FETCH_VERSION 1.0.9)
|
||||
add_library(GSLIB STATIC IMPORTED)
|
||||
# set options (technically flags because GSLIB does not use cmake)
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
|
||||
set(GSLIB_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
|
||||
if (BUILD_SHARED_LIBS)
|
||||
set(GSLIB_FLAGS "${GSLIB_FLAGS} -fPIC")
|
||||
set(GSLIB_FETCH_VERSION 1.0.9)
|
||||
set(GSLIB_C_FLAGS ${CMAKE_C_FLAGS_${BUILD_TYPE}})
|
||||
if (CMAKE_C_FLAGS)
|
||||
set(GSLIB_C_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
|
||||
endif()
|
||||
if (BUILD_SHARED_LIBS)
|
||||
set(GSLIB_C_FLAGS "${GSLIB_C_FLAGS} -fPIC")
|
||||
endif()
|
||||
add_library(GSLIB STATIC IMPORTED)
|
||||
# define external project and create future include directory so it is present
|
||||
# to pass CMake checks at end of MFEM configuration step
|
||||
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_FLAGS}")
|
||||
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_C_FLAGS}")
|
||||
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/gslib)
|
||||
include(ExternalProject)
|
||||
ExternalProject_Add(gslib
|
||||
@@ -38,7 +40,7 @@ if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
|
||||
UPDATE_DISCONNECTED TRUE
|
||||
PREFIX ${PREFIX}
|
||||
CONFIGURE_COMMAND ""
|
||||
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS=${GSLIB_FLAGS}"
|
||||
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS= ${GSLIB_C_FLAGS}"
|
||||
INSTALL_COMMAND "")
|
||||
file(MAKE_DIRECTORY ${PREFIX}/include)
|
||||
# set imported library target properties
|
||||
|
||||
@@ -44,9 +44,6 @@ if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
|
||||
# set options and associated dependencies
|
||||
set(HYPRE_CMAKE_OPTIONS "")
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
|
||||
if (BUILD_SHARED_LIBS)
|
||||
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_POSITION_INDEPENDENT_CODE:BOOL=ON)
|
||||
endif()
|
||||
# collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
|
||||
get_cmake_property(all_vars VARIABLES)
|
||||
foreach(var ${all_vars})
|
||||
@@ -98,6 +95,7 @@ if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
|
||||
UPDATE_DISCONNECTED TRUE
|
||||
SOURCE_SUBDIR src
|
||||
PREFIX ${HYPRE_INSTALL}
|
||||
BUILD_COMMAND ${CMAKE_COMMAND} --build . -- -j${CMAKE_BUILD_PARALLEL_LEVEL}
|
||||
CMAKE_CACHE_ARGS -DCMAKE_INSTALL_PREFIX:PATH=${HYPRE_INSTALL} -DCMAKE_INSTALL_LIBDIR:PATH=lib ${HYPRE_CMAKE_OPTIONS})
|
||||
file(MAKE_DIRECTORY ${HYPRE_INSTALL}/include)
|
||||
# set imported library target properties
|
||||
|
||||
@@ -19,18 +19,10 @@
|
||||
# - METIS_VERSION_5 (cache variable)
|
||||
|
||||
if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
|
||||
enable_language(C)
|
||||
set(METIS_FETCH_VERSION 4.0.3)
|
||||
add_library(METIS STATIC IMPORTED)
|
||||
# set options (technically flags because METIS does not use cmake)
|
||||
set(METIS_FLAGS "-Wno-implicit-int -Wno-incompatible-pointer-types")
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
|
||||
set(METIS_FLAGS "${METIS_FLAGS} ${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
|
||||
if (BUILD_SHARED_LIBS)
|
||||
set(METIS_FLAGS "${METIS_FLAGS} -fPIC")
|
||||
endif()
|
||||
# define external project
|
||||
message(STATUS "Will fetch METIS ${METIS_FETCH_VERSION} to be built with ${METIS_FLAGS}")
|
||||
message(STATUS "Will fetch METIS ${METIS_FETCH_VERSION} to be built with default options")
|
||||
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/metis)
|
||||
include(ExternalProject)
|
||||
ExternalProject_Add(metis
|
||||
@@ -40,7 +32,7 @@ if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
|
||||
UPDATE_DISCONNECTED TRUE
|
||||
PREFIX ${PREFIX}
|
||||
CONFIGURE_COMMAND tar -xzf ../metis/metis-${METIS_FETCH_VERSION}-mac.tgz --strip=1
|
||||
BUILD_COMMAND $(MAKE) clean && $(MAKE) "OPTFLAGS=${METIS_FLAGS}"
|
||||
BUILD_COMMAND $(MAKE) COPTIONS=-Wno-incompatible-pointer-types
|
||||
INSTALL_COMMAND mkdir -p ${PREFIX}/lib && cp libmetis.a ${PREFIX}/lib/)
|
||||
# set imported library target properties
|
||||
add_dependencies(METIS metis)
|
||||
|
||||
@@ -22,15 +22,15 @@ include(MfemCmakeUtilities)
|
||||
mfem_find_package(SuiteSparse SuiteSparse SuiteSparse_DIR "" "" "" ""
|
||||
"Paths to headers required by SuiteSparse."
|
||||
"Libraries required by SuiteSparse."
|
||||
ADD_COMPONENT "UMFPACK" "include;include/suitesparse;suitesparse" umfpack.h "lib" umfpack
|
||||
ADD_COMPONENT "KLU" "include;include/suitesparse;suitesparse" klu.h "lib" klu
|
||||
ADD_COMPONENT "AMD" "include;include/suitesparse;suitesparse" amd.h "lib" amd
|
||||
ADD_COMPONENT "BTF" "include;include/suitesparse;suitesparse" btf.h "lib" btf
|
||||
ADD_COMPONENT "CHOLMOD" "include;include/suitesparse;suitesparse" cholmod.h "lib" cholmod
|
||||
ADD_COMPONENT "COLAMD" "include;include/suitesparse;suitesparse" colamd.h "lib" colamd
|
||||
ADD_COMPONENT "CAMD" "include;include/suitesparse;suitesparse" camd.h "lib" camd
|
||||
ADD_COMPONENT "CCOLAMD" "include;include/suitesparse;suitesparse" ccolamd.h "lib" ccolamd
|
||||
ADD_COMPONENT "config" "include;include/suitesparse;suitesparse" SuiteSparse_config.h "lib"
|
||||
ADD_COMPONENT "UMFPACK" "include;suitesparse" umfpack.h "lib" umfpack
|
||||
ADD_COMPONENT "KLU" "include;suitesparse" klu.h "lib" klu
|
||||
ADD_COMPONENT "AMD" "include;suitesparse" amd.h "lib" amd
|
||||
ADD_COMPONENT "BTF" "include;suitesparse" btf.h "lib" btf
|
||||
ADD_COMPONENT "CHOLMOD" "include;suitesparse" cholmod.h "lib" cholmod
|
||||
ADD_COMPONENT "COLAMD" "include;suitesparse" colamd.h "lib" colamd
|
||||
ADD_COMPONENT "CAMD" "include;suitesparse" camd.h "lib" camd
|
||||
ADD_COMPONENT "CCOLAMD" "include;suitesparse" ccolamd.h "lib" ccolamd
|
||||
ADD_COMPONENT "config" "include;suitesparse" SuiteSparse_config.h "lib"
|
||||
suitesparseconfig)
|
||||
|
||||
if (SuiteSparse_FOUND AND METIS_VERSION_5)
|
||||
|
||||
@@ -157,10 +157,4 @@ constexpr real_t operator""_r(unsigned long long v)
|
||||
#endif
|
||||
#endif // MFEM_USE_MPI not defined
|
||||
|
||||
#ifndef MFEM_USE_CUDA
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
#error Building with cuDSS (MFEM_USE_CUDSS=YES) requires CUDA (MFEM_USE_CUDA=YES)
|
||||
#endif
|
||||
#endif // MFEM_USE_CUDSS not defined
|
||||
|
||||
#endif // MFEM_CONFIG_HPP
|
||||
|
||||
@@ -108,15 +108,6 @@
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
// #define MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable MFEM functionality based on the cuDSS library.
|
||||
// #define MFEM_USE_CUDSS
|
||||
|
||||
// CUDSS communication layer library path
|
||||
// #define MFEM_CUDSS_COMM_LIB "@MFEM_CUDSS_COMM_LIB@"
|
||||
|
||||
// CUDSS threading layer library path
|
||||
// #define MFEM_CUDSS_THREADING_LIB "@MFEM_CUDSS_THREADING_LIB@"
|
||||
|
||||
// Enable MFEM features based on the Ginkgo library.
|
||||
// #define MFEM_USE_GINKGO
|
||||
|
||||
|
||||
@@ -36,9 +36,6 @@ MFEM_USE_SUPERLU = @MFEM_USE_SUPERLU@
|
||||
MFEM_USE_SUPERLU5 = @MFEM_USE_SUPERLU5@
|
||||
MFEM_USE_MUMPS = @MFEM_USE_MUMPS@
|
||||
MFEM_USE_STRUMPACK = @MFEM_USE_STRUMPACK@
|
||||
MFEM_USE_CUDSS = @MFEM_USE_CUDSS@
|
||||
MFEM_CUDSS_COMM_LIB = @MFEM_CUDSS_COMM_LIB@
|
||||
MFEM_CUDSS_THREADING_LIB = @MFEM_CUDSS_THREADING_LIB@
|
||||
MFEM_USE_GINKGO = @MFEM_USE_GINKGO@
|
||||
MFEM_USE_AMGX = @MFEM_USE_AMGX@
|
||||
MFEM_USE_MAGMA = @MFEM_USE_MAGMA@
|
||||
|
||||
@@ -38,7 +38,6 @@ option(MFEM_USE_SUPERLU "Enable SuperLU_DIST usage" OFF)
|
||||
option(MFEM_USE_SUPERLU5 "Use the old SuperLU_DIST 5.1 version" OFF)
|
||||
option(MFEM_USE_MUMPS "Enable MUMPS usage" OFF)
|
||||
option(MFEM_USE_STRUMPACK "Enable STRUMPACK usage" OFF)
|
||||
option(MFEM_USE_CUDSS "Enable cuDSS usage" OFF)
|
||||
option(MFEM_USE_GINKGO "Enable Ginkgo usage" OFF)
|
||||
option(MFEM_USE_AMGX "Enable AmgX usage" OFF)
|
||||
option(MFEM_USE_MAGMA "Enable MAGMA usage" OFF)
|
||||
|
||||
+4
-49
@@ -27,10 +27,7 @@ MPICXX = mpicxx
|
||||
|
||||
BASE_FLAGS = -std=c++17
|
||||
OPTIM_FLAGS = -O3 $(BASE_FLAGS)
|
||||
|
||||
# The variable WARNING_FLAGS depends on which compiler is used, and is defined
|
||||
# later in this file.
|
||||
DEBUG_FLAGS = $(strip -g $(addprefix $(XCOMPILER),$(WARNING_FLAGS)) $(BASE_FLAGS))
|
||||
DEBUG_FLAGS = -g $(XCOMPILER)-Wall $(BASE_FLAGS)
|
||||
|
||||
# Prefixes for passing flags to the compiler and linker when using CXX or MPICXX
|
||||
CXX_XCOMPILER =
|
||||
@@ -49,10 +46,6 @@ SHARED = NO
|
||||
#
|
||||
# If you set MFEM_USE_ENZYME=YES, must use CUDA_CXX=clang++
|
||||
CUDA_CXX = nvcc
|
||||
# CUDA compute capability used during compilation, e.g. sm_60. Multiple
|
||||
# architectures can be requested as a comma-separated list, e.g. sm_70,sm_80.
|
||||
# A single value may also be one of the nvcc special values "all",
|
||||
# "all-major", or "native".
|
||||
CUDA_ARCH = sm_60
|
||||
# Base CUDA install directory, only needed if building with clang+cuda:
|
||||
# The default setting is:
|
||||
@@ -61,23 +54,11 @@ CUDA_ARCH = sm_60
|
||||
# 3. Use /usr/local/cuda
|
||||
CUDA_DIR = $(or $(CUDA_HOME),$(patsubst %/,%,$(dir \
|
||||
$(patsubst %/,%,$(dir $(shell command -v nvcc))))),/usr/local/cuda)
|
||||
# Derive nvcc/clang architecture flags from CUDA_ARCH. A comma-separated list
|
||||
# expands into one -gencode / --cuda-gpu-arch flag per architecture; otherwise
|
||||
# use the -arch / --cuda-gpu-arch shorthand.
|
||||
MFEM_COMMA := ,
|
||||
CUDA_ARCH_NUMS = $(patsubst sm_%,%,$(subst $(MFEM_COMMA), ,$(CUDA_ARCH)))
|
||||
NVCC_ARCH_FLAGS = $(strip $(if $(findstring $(MFEM_COMMA),$(CUDA_ARCH)),\
|
||||
$(foreach arch,$(CUDA_ARCH_NUMS),\
|
||||
-gencode arch=compute_$(arch)$(MFEM_COMMA)code=sm_$(arch)),\
|
||||
-arch=$(CUDA_ARCH)))
|
||||
CLANG_ARCH_FLAGS = $(strip $(if $(findstring $(MFEM_COMMA),$(CUDA_ARCH)),\
|
||||
$(foreach arch,$(CUDA_ARCH_NUMS),--cuda-gpu-arch=sm_$(arch)),\
|
||||
--cuda-gpu-arch=$(CUDA_ARCH)))
|
||||
# flags for clang+cuda
|
||||
CLANG_CUDA_FLAGS = -xcuda --cuda-path=$(CUDA_DIR) $(CLANG_ARCH_FLAGS)
|
||||
CLANG_CUDA_FLAGS = -xcuda --cuda-path=$(CUDA_DIR) --cuda-gpu-arch=$(CUDA_ARCH)
|
||||
# flags for nvcc
|
||||
NVCC_FLAGS = -x=cu --expt-extended-lambda --expt-relaxed-constexpr \
|
||||
$(NVCC_ARCH_FLAGS) -isystem "$(CUDA_DIR)/include"
|
||||
-arch=$(CUDA_ARCH) -isystem "$(CUDA_DIR)/include"
|
||||
# Prefixes for passing flags to the host compiler and linker when using
|
||||
# CUDA_CXX=nvcc
|
||||
CUDA_XCOMPILER = -Xcompiler=
|
||||
@@ -172,7 +153,6 @@ MFEM_USE_SUPERLU = NO
|
||||
MFEM_USE_SUPERLU5 = NO
|
||||
MFEM_USE_MUMPS = NO
|
||||
MFEM_USE_STRUMPACK = NO
|
||||
MFEM_USE_CUDSS = NO
|
||||
MFEM_USE_GINKGO = NO
|
||||
MFEM_USE_AMGX = NO
|
||||
MFEM_USE_MAGMA = NO
|
||||
@@ -388,19 +368,6 @@ STRUMPACK_OPT = -I$(STRUMPACK_DIR)/include $(SCOTCH_OPT)
|
||||
STRUMPACK_LIB = -L$(STRUMPACK_DIR)/lib -lstrumpack $(MPI_FORTRAN_LIB)\
|
||||
$(SCOTCH_LIB) $(SCALAPACK_LIB)
|
||||
|
||||
# CUDSS library configuration
|
||||
CUDSS_DIR = @MFEM_DIR@/../cudss
|
||||
CUDSS_INCLUDE_DIR = $(CUDSS_DIR)/include
|
||||
CUDSS_LIBRARY_DIR = $(CUDSS_DIR)/lib
|
||||
CUDSS_OPT = -I$(CUDSS_INCLUDE_DIR)
|
||||
CUDSS_LIB = \
|
||||
$(XLINKER)-rpath,$(CUDSS_LIBRARY_DIR) -L$(CUDSS_LIBRARY_DIR) -lcudss
|
||||
# The cuDSS communication and threading libraries.
|
||||
MFEM_CUDSS_COMM_LIB = $(abspath $(wildcard $(or $(CUDSS_COMM_LIB),\
|
||||
$(subst @MFEM_DIR@,$(MFEM_DIR), $(CUDSS_LIBRARY_DIR)/libcudss_commlayer_openmpi.so))))
|
||||
MFEM_CUDSS_THREADING_LIB = $(abspath $(wildcard $(or $(CUDSS_THREADING_LIB),\
|
||||
$(subst @MFEM_DIR@,$(MFEM_DIR),$(CUDSS_LIBRARY_DIR)/libcudss_mtlayer_gomp.so))))
|
||||
|
||||
# Ginkgo library configuration
|
||||
GINKGO_DIR = @MFEM_DIR@/../ginkgo/install
|
||||
GINKGO_SEARCH_DIR = $(subst @MFEM_DIR@,$(MFEM_DIR),$(GINKGO_DIR))
|
||||
@@ -654,7 +621,7 @@ PARELAG_LIB = -L$(PARELAG_DIR)/build/src -lParELAG
|
||||
AXOM_DIR = @MFEM_DIR@/../axom
|
||||
TRIBOL_DIR = @MFEM_DIR@/../tribol
|
||||
TRIBOL_OPT = -I$(TRIBOL_DIR)/include -I$(AXOM_DIR)/include
|
||||
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -ltribol_shared -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
|
||||
TRIBOL_LIB = -L$(TRIBOL_DIR)/lib -ltribol -lredecomp -L$(AXOM_DIR)/lib -laxom_mint\
|
||||
-laxom_slam -laxom_slic -laxom_core
|
||||
|
||||
# Enzyme configuration
|
||||
@@ -678,15 +645,3 @@ VERBOSE = NO
|
||||
|
||||
# Optional build tag
|
||||
MFEM_BUILD_TAG = $(shell uname -snm)
|
||||
|
||||
# Enable -pedantic flag only for gcc or clang. nvcc complains with -pedantic
|
||||
# because of line directives.
|
||||
PEDANTIC_FLAG = $(if \
|
||||
$(findstring NVIDIA,$(shell $(MFEM_CXX) --version 2>&1)),, \
|
||||
$(if $(or \
|
||||
$(findstring gcc version,$(shell $(MFEM_CXX) -v 2>&1)), \
|
||||
$(findstring clang version,$(shell $(MFEM_CXX) -v 2>&1))),-pedantic,))
|
||||
# Enable shadow warnings for clang only; GCC's -Wshadow flags more.
|
||||
SHADOW_WARNING_FLAG = $(if $(findstring clang,\
|
||||
$(shell $(MFEM_HOST_CXX) --version 2>/dev/null)),-Wshadow,)
|
||||
WARNING_FLAGS = $(PEDANTIC_FLAG) -Wall $(SHADOW_WARNING_FLAG)
|
||||
|
||||
@@ -39,8 +39,3 @@ when a picture was added for documentation.
|
||||
If that is the case, make sure the failure is indeed justified, and rerun the
|
||||
push command with the `--no-verify` option. This will skip the hooks, allowing
|
||||
you to push those changes.
|
||||
|
||||
The `branch-history` check is run automatically through GitHub Actions. If a
|
||||
branch is known to have a large number of changes that are legitimate, the
|
||||
check can be overridden by setting the label 'branch-history-override' on the
|
||||
pull request.
|
||||
|
||||
@@ -1,131 +0,0 @@
|
||||
// Define the cube sizes
|
||||
L_outer = 1.0;
|
||||
L_inner = 0.5;
|
||||
|
||||
// Set mesh size and algorithm
|
||||
mesh_size = 0.4;
|
||||
Mesh.Algorithm3D = 1; // Delaunay algorithm for 3D mesh
|
||||
Mesh.CharacteristicLengthFactor = 1.0;
|
||||
Mesh.MshFileVersion = 2.2;
|
||||
|
||||
// Define center point for concentric cubes
|
||||
cx = 0.5;
|
||||
cy = 0.5;
|
||||
cz = 0.5;
|
||||
|
||||
// Define the points (vertices of the outer cube)
|
||||
Point(1) = {cx-L_outer/2, cy-L_outer/2, cz-L_outer/2, mesh_size};
|
||||
Point(2) = {cx+L_outer/2, cy-L_outer/2, cz-L_outer/2, mesh_size};
|
||||
Point(3) = {cx+L_outer/2, cy+L_outer/2, cz-L_outer/2, mesh_size};
|
||||
Point(4) = {cx-L_outer/2, cy+L_outer/2, cz-L_outer/2, mesh_size};
|
||||
Point(5) = {cx-L_outer/2, cy-L_outer/2, cz+L_outer/2, mesh_size};
|
||||
Point(6) = {cx+L_outer/2, cy-L_outer/2, cz+L_outer/2, mesh_size};
|
||||
Point(7) = {cx+L_outer/2, cy+L_outer/2, cz+L_outer/2, mesh_size};
|
||||
Point(8) = {cx-L_outer/2, cy+L_outer/2, cz+L_outer/2, mesh_size};
|
||||
|
||||
// Define the points (vertices of the inner cube)
|
||||
Point(9) = {cx-L_inner/2, cy-L_inner/2, cz-L_inner/2, mesh_size};
|
||||
Point(10) = {cx+L_inner/2, cy-L_inner/2, cz-L_inner/2, mesh_size};
|
||||
Point(11) = {cx+L_inner/2, cy+L_inner/2, cz-L_inner/2, mesh_size};
|
||||
Point(12) = {cx-L_inner/2, cy+L_inner/2, cz-L_inner/2, mesh_size};
|
||||
Point(13) = {cx-L_inner/2, cy-L_inner/2, cz+L_inner/2, mesh_size};
|
||||
Point(14) = {cx+L_inner/2, cy-L_inner/2, cz+L_inner/2, mesh_size};
|
||||
Point(15) = {cx+L_inner/2, cy+L_inner/2, cz+L_inner/2, mesh_size};
|
||||
Point(16) = {cx-L_inner/2, cy+L_inner/2, cz+L_inner/2, mesh_size};
|
||||
|
||||
// Define the lines (edges of the outer cube)
|
||||
Line(1) = {1, 2};
|
||||
Line(2) = {2, 3};
|
||||
Line(3) = {3, 4};
|
||||
Line(4) = {4, 1};
|
||||
Line(5) = {5, 6};
|
||||
Line(6) = {6, 7};
|
||||
Line(7) = {7, 8};
|
||||
Line(8) = {8, 5};
|
||||
Line(9) = {1, 5};
|
||||
Line(10) = {2, 6};
|
||||
Line(11) = {3, 7};
|
||||
Line(12) = {4, 8};
|
||||
|
||||
// Define the lines (edges of the inner cube)
|
||||
Line(13) = {9, 10};
|
||||
Line(14) = {10, 11};
|
||||
Line(15) = {11, 12};
|
||||
Line(16) = {12, 9};
|
||||
Line(17) = {13, 14};
|
||||
Line(18) = {14, 15};
|
||||
Line(19) = {15, 16};
|
||||
Line(20) = {16, 13};
|
||||
Line(21) = {9, 13};
|
||||
Line(22) = {10, 14};
|
||||
Line(23) = {11, 15};
|
||||
Line(24) = {12, 16};
|
||||
|
||||
// Define the surfaces (faces of the outer cube)
|
||||
Line Loop(1) = {1, 2, 3, 4};
|
||||
Plane Surface(1) = {1};
|
||||
|
||||
Line Loop(2) = {5, 6, 7, 8};
|
||||
Plane Surface(2) = {2};
|
||||
|
||||
Line Loop(3) = {9, 5, -10, -1};
|
||||
Plane Surface(3) = {3};
|
||||
|
||||
Line Loop(4) = {10, 6, -11, -2};
|
||||
Plane Surface(4) = {4};
|
||||
|
||||
Line Loop(5) = {11, 7, -12, -3};
|
||||
Plane Surface(5) = {5};
|
||||
|
||||
Line Loop(6) = {12, 8, -9, -4};
|
||||
Plane Surface(6) = {6};
|
||||
|
||||
// Define the surfaces (faces of the inner cube)
|
||||
Line Loop(7) = {13, 14, 15, 16};
|
||||
Plane Surface(7) = {7};
|
||||
|
||||
Line Loop(8) = {17, 18, 19, 20};
|
||||
Plane Surface(8) = {8};
|
||||
|
||||
Line Loop(9) = {21, 17, -22, -13};
|
||||
Plane Surface(9) = {9};
|
||||
|
||||
Line Loop(10) = {22, 18, -23, -14};
|
||||
Plane Surface(10) = {10};
|
||||
|
||||
Line Loop(11) = {23, 19, -24, -15};
|
||||
Plane Surface(11) = {11};
|
||||
|
||||
Line Loop(12) = {24, 20, -21, -16};
|
||||
Plane Surface(12) = {12};
|
||||
|
||||
// Define the volumes
|
||||
Surface Loop(1) = {1, 2, 3, 4, 5, 6};
|
||||
Surface Loop(2) = {7, 8, 9, 10, 11, 12};
|
||||
Volume(1) = {1, 2}; // Outer volume with inner hole
|
||||
Volume(2) = {2}; // Inner volume
|
||||
|
||||
// Assign physical groups
|
||||
Physical Volume(1) = {1}; // Outer volume
|
||||
Physical Volume(2) = {2}; // Inner volume
|
||||
|
||||
// Outer cube surfaces
|
||||
Physical Surface(1) = {1}; // Outer bottom
|
||||
Physical Surface(2) = {2}; // Outer top
|
||||
Physical Surface(3) = {3}; // Outer front
|
||||
Physical Surface(4) = {4}; // Outer right
|
||||
Physical Surface(5) = {5}; // Outer back
|
||||
Physical Surface(6) = {6}; // Outer left
|
||||
|
||||
// Inner cube surfaces
|
||||
Physical Surface(7) = {7}; // Inner bottom (-xy)
|
||||
Physical Surface(8) = {8}; // Inner top (+xy)
|
||||
Physical Surface(9) = {9}; // Inner front (-xz)
|
||||
Physical Surface(10) = {10}; // Inner right (+yz)
|
||||
Physical Surface(11) = {11}; // Inner back (+xz)
|
||||
Physical Surface(12) = {12}; // Inner left (-yz)
|
||||
|
||||
// Mesh control
|
||||
Mesh.OptimizeNetgen = 1;
|
||||
Mesh.Optimize = 1;
|
||||
Mesh.ElementOrder = 1;
|
||||
@@ -1,907 +0,0 @@
|
||||
$MeshFormat
|
||||
2.2 0 8
|
||||
$EndMeshFormat
|
||||
$Nodes
|
||||
138
|
||||
1 0 0 0
|
||||
2 1 0 0
|
||||
3 1 1 0
|
||||
4 0 1 0
|
||||
5 0 0 1
|
||||
6 1 0 1
|
||||
7 1 1 1
|
||||
8 0 1 1
|
||||
9 0.25 0.25 0.25
|
||||
10 0.75 0.25 0.25
|
||||
11 0.75 0.75 0.25
|
||||
12 0.25 0.75 0.25
|
||||
13 0.25 0.25 0.75
|
||||
14 0.75 0.25 0.75
|
||||
15 0.75 0.75 0.75
|
||||
16 0.25 0.75 0.75
|
||||
17 0.3333333333325025 0 0
|
||||
18 0.6666666666657889 0 0
|
||||
19 1 0.3333333333325025 0
|
||||
20 1 0.6666666666657889 0
|
||||
21 0.6666666666675911 1 0
|
||||
22 0.3333333333347203 1 0
|
||||
23 0 0.6666666666675911 0
|
||||
24 0 0.3333333333347203 0
|
||||
25 0.3333333333325025 0 1
|
||||
26 0.6666666666657889 0 1
|
||||
27 1 0.3333333333325025 1
|
||||
28 1 0.6666666666657889 1
|
||||
29 0.6666666666675911 1 1
|
||||
30 0.3333333333347203 1 1
|
||||
31 0 0.6666666666675911 1
|
||||
32 0 0.3333333333347203 1
|
||||
33 0 0 0.3333333333325025
|
||||
34 0 0 0.6666666666657889
|
||||
35 1 0 0.3333333333325025
|
||||
36 1 0 0.6666666666657889
|
||||
37 1 1 0.3333333333325025
|
||||
38 1 1 0.6666666666657889
|
||||
39 0 1 0.3333333333325025
|
||||
40 0 1 0.6666666666657889
|
||||
41 0.5000000000003468 0.25 0.25
|
||||
42 0.75 0.5000000000003468 0.25
|
||||
43 0.5000000000013763 0.75 0.25
|
||||
44 0.25 0.5000000000013763 0.25
|
||||
45 0.5000000000003468 0.25 0.75
|
||||
46 0.75 0.5000000000003468 0.75
|
||||
47 0.5000000000013763 0.75 0.75
|
||||
48 0.25 0.5000000000013763 0.75
|
||||
49 0.25 0.25 0.5000000000003468
|
||||
50 0.75 0.25 0.5000000000003468
|
||||
51 0.75 0.75 0.5000000000003468
|
||||
52 0.25 0.75 0.5000000000003468
|
||||
53 0.7113248654055673 0.4999999999991457 0
|
||||
54 0.2886751345942123 0.5000000000011557 0
|
||||
55 0.5000000000006117 0.7525600817161773 0
|
||||
56 0.4999999999993867 0.2474399182839603 0
|
||||
57 0.2423197548524782 0.7576802451481532 0
|
||||
58 0.757680245147464 0.2423197548520695 0
|
||||
59 0.2423197548507857 0.2423197548513912 0
|
||||
60 0.7576802451491019 0.7576802451486099 0
|
||||
61 0.7113248654055673 0.4999999999991457 1
|
||||
62 0.2886751345942123 0.5000000000011557 1
|
||||
63 0.5000000000006117 0.7525600817161773 1
|
||||
64 0.4999999999993867 0.2474399182839603 1
|
||||
65 0.2423197548524782 0.7576802451481532 1
|
||||
66 0.757680245147464 0.2423197548520695 1
|
||||
67 0.2423197548507857 0.2423197548513912 1
|
||||
68 0.7576802451491019 0.7576802451486099 1
|
||||
69 0.4999999999993203 0 0.301447615129799
|
||||
70 0.4999999999992795 0 0.7028666213189801
|
||||
71 0.7525600817158393 0 0.5007190394076877
|
||||
72 0.2474399182836191 0 0.5007190394076877
|
||||
73 0.7576802451479793 0 0.7576802451479793
|
||||
74 0.2423197548517962 0 0.7576802451477375
|
||||
75 0.7576802451484569 0 0.2423197548510767
|
||||
76 0.2423197548513188 0 0.2423197548513187
|
||||
77 1 0.4999999999993203 0.301447615129799
|
||||
78 1 0.4999999999992795 0.7028666213189801
|
||||
79 1 0.7525600817158394 0.5007190394076877
|
||||
80 1 0.2474399182836191 0.5007190394076877
|
||||
81 1 0.7576802451479794 0.7576802451479794
|
||||
82 1 0.2423197548517962 0.7576802451477376
|
||||
83 1 0.7576802451484569 0.2423197548510768
|
||||
84 1 0.2423197548513188 0.2423197548513188
|
||||
85 0.5000000000008327 1 0.3014476151298047
|
||||
86 0.500000000000961 1 0.7028666213191928
|
||||
87 0.2474399182842484 1 0.5007190394077241
|
||||
88 0.7525600817164384 1 0.5007190394078933
|
||||
89 0.2423197548520873 1 0.7576802451480517
|
||||
90 0.7576802451481496 1 0.2423197548518761
|
||||
91 0.2423197548516099 1 0.2423197548510044
|
||||
92 0.7576802451486874 1 0.7576802451481952
|
||||
93 0 0.5000000000008327 0.3014476151298047
|
||||
94 0 0.500000000000961 0.7028666213191928
|
||||
95 0 0.2474399182842484 0.5007190394077241
|
||||
96 0 0.7525600817164384 0.5007190394078933
|
||||
97 0 0.2423197548520873 0.7576802451480517
|
||||
98 0 0.7576802451481496 0.2423197548518761
|
||||
99 0 0.2423197548516099 0.2423197548510044
|
||||
100 0 0.7576802451486874 0.7576802451481952
|
||||
101 0.3968750000003409 0.603125000000244 0.25
|
||||
102 0.4374999999998713 0.4375000000001287 0.25
|
||||
103 0.5739583333335919 0.5718750000001767 0.25
|
||||
104 0.6093749999999631 0.3906250000003402 0.25
|
||||
105 0.3968750000003409 0.603125000000244 0.75
|
||||
106 0.4374999999998713 0.4375000000001287 0.75
|
||||
107 0.5739583333335919 0.5718750000001767 0.75
|
||||
108 0.6093749999999631 0.3906250000003402 0.75
|
||||
109 0.3806942419826734 0.25 0.3806942419826734
|
||||
110 0.5625000000001735 0.25 0.4375000000000001
|
||||
111 0.4254282069971791 0.25 0.5712615403304835
|
||||
112 0.6093749999998808 0.25 0.6093749999998808
|
||||
113 0.75 0.3806942419826734 0.3806942419826734
|
||||
114 0.75 0.5625000000001735 0.4375000000000001
|
||||
115 0.75 0.4254282069971791 0.5712615403304835
|
||||
116 0.75 0.6093749999998808 0.6093749999998808
|
||||
117 0.3968750000004991 0.75 0.3968750000002804
|
||||
118 0.4375000000000001 0.75 0.5625000000004308
|
||||
119 0.5739583333336153 0.75 0.4281250000000707
|
||||
120 0.6093749999998166 0.75 0.6093749999993661
|
||||
121 0.25 0.3968750000004991 0.3968750000002804
|
||||
122 0.25 0.4375000000000001 0.5625000000004308
|
||||
123 0.25 0.5739583333336153 0.4281250000000707
|
||||
124 0.25 0.6093749999998166 0.6093749999993661
|
||||
125 0.4962939304035875 0.5214350017087855 0.4925553109323813
|
||||
126 0.3432581985549767 0.6471275530923523 0.3554206606168006
|
||||
127 0.6442168181713744 0.5929232373214773 0.3593785632300704
|
||||
128 0.625174517421737 0.3491579444372839 0.3604636129462503
|
||||
129 0.6130544111091688 0.6576364353993367 0.5046720648819553
|
||||
130 0.4281518698369243 0.3632662294430484 0.3548726263205102
|
||||
131 0.3639383531355198 0.3520949250221278 0.4978389662613994
|
||||
132 0.629585530087249 0.3489878162230438 0.5124654846339122
|
||||
133 0.3710853378652663 0.6517586292698121 0.6382643241302075
|
||||
134 0.5917018263727056 0.6522525456211955 0.6390764961119443
|
||||
135 0.3530810314228338 0.4582477062424107 0.6430239639637545
|
||||
136 0.6571010904289998 0.5300774811423468 0.6603970500567977
|
||||
137 0.6484596018596915 0.361399676127967 0.6360267588157177
|
||||
138 0.4782020887035478 0.3534611476388013 0.6141275027013793
|
||||
$EndNodes
|
||||
$Elements
|
||||
760
|
||||
1 2 2 1 1 1 17 59
|
||||
2 2 2 1 1 24 1 59
|
||||
3 2 2 1 1 18 2 58
|
||||
4 2 2 1 1 2 19 58
|
||||
5 2 2 1 1 20 3 60
|
||||
6 2 2 1 1 3 21 60
|
||||
7 2 2 1 1 22 4 57
|
||||
8 2 2 1 1 4 23 57
|
||||
9 2 2 1 1 17 18 56
|
||||
10 2 2 1 1 17 56 59
|
||||
11 2 2 1 1 56 18 58
|
||||
12 2 2 1 1 19 20 53
|
||||
13 2 2 1 1 19 53 58
|
||||
14 2 2 1 1 53 20 60
|
||||
15 2 2 1 1 21 22 55
|
||||
16 2 2 1 1 21 55 60
|
||||
17 2 2 1 1 55 22 57
|
||||
18 2 2 1 1 23 24 54
|
||||
19 2 2 1 1 23 54 57
|
||||
20 2 2 1 1 54 24 59
|
||||
21 2 2 1 1 54 53 55
|
||||
22 2 2 1 1 53 54 56
|
||||
23 2 2 1 1 55 53 60
|
||||
24 2 2 1 1 53 56 58
|
||||
25 2 2 1 1 54 55 57
|
||||
26 2 2 1 1 56 54 59
|
||||
27 2 2 2 2 5 25 67
|
||||
28 2 2 2 2 32 5 67
|
||||
29 2 2 2 2 26 6 66
|
||||
30 2 2 2 2 6 27 66
|
||||
31 2 2 2 2 28 7 68
|
||||
32 2 2 2 2 7 29 68
|
||||
33 2 2 2 2 30 8 65
|
||||
34 2 2 2 2 8 31 65
|
||||
35 2 2 2 2 25 26 64
|
||||
36 2 2 2 2 25 64 67
|
||||
37 2 2 2 2 64 26 66
|
||||
38 2 2 2 2 27 28 61
|
||||
39 2 2 2 2 27 61 66
|
||||
40 2 2 2 2 61 28 68
|
||||
41 2 2 2 2 29 30 63
|
||||
42 2 2 2 2 29 63 68
|
||||
43 2 2 2 2 63 30 65
|
||||
44 2 2 2 2 31 32 62
|
||||
45 2 2 2 2 31 62 65
|
||||
46 2 2 2 2 62 32 67
|
||||
47 2 2 2 2 62 61 63
|
||||
48 2 2 2 2 61 62 64
|
||||
49 2 2 2 2 63 61 68
|
||||
50 2 2 2 2 61 64 66
|
||||
51 2 2 2 2 62 63 65
|
||||
52 2 2 2 2 64 62 67
|
||||
53 2 2 3 3 17 1 76
|
||||
54 2 2 3 3 1 33 76
|
||||
55 2 2 3 3 2 18 75
|
||||
56 2 2 3 3 35 2 75
|
||||
57 2 2 3 3 5 25 74
|
||||
58 2 2 3 3 34 5 74
|
||||
59 2 2 3 3 26 6 73
|
||||
60 2 2 3 3 6 36 73
|
||||
61 2 2 3 3 18 17 69
|
||||
62 2 2 3 3 69 17 76
|
||||
63 2 2 3 3 18 69 75
|
||||
64 2 2 3 3 25 26 70
|
||||
65 2 2 3 3 25 70 74
|
||||
66 2 2 3 3 70 26 73
|
||||
67 2 2 3 3 33 34 72
|
||||
68 2 2 3 3 33 72 76
|
||||
69 2 2 3 3 72 34 74
|
||||
70 2 2 3 3 36 35 71
|
||||
71 2 2 3 3 71 35 75
|
||||
72 2 2 3 3 36 71 73
|
||||
73 2 2 3 3 69 70 71
|
||||
74 2 2 3 3 70 69 72
|
||||
75 2 2 3 3 69 71 75
|
||||
76 2 2 3 3 72 69 76
|
||||
77 2 2 3 3 71 70 73
|
||||
78 2 2 3 3 70 72 74
|
||||
79 2 2 4 4 19 2 84
|
||||
80 2 2 4 4 2 35 84
|
||||
81 2 2 4 4 3 20 83
|
||||
82 2 2 4 4 37 3 83
|
||||
83 2 2 4 4 6 27 82
|
||||
84 2 2 4 4 36 6 82
|
||||
85 2 2 4 4 28 7 81
|
||||
86 2 2 4 4 7 38 81
|
||||
87 2 2 4 4 20 19 77
|
||||
88 2 2 4 4 77 19 84
|
||||
89 2 2 4 4 20 77 83
|
||||
90 2 2 4 4 27 28 78
|
||||
91 2 2 4 4 27 78 82
|
||||
92 2 2 4 4 78 28 81
|
||||
93 2 2 4 4 35 36 80
|
||||
94 2 2 4 4 35 80 84
|
||||
95 2 2 4 4 80 36 82
|
||||
96 2 2 4 4 38 37 79
|
||||
97 2 2 4 4 79 37 83
|
||||
98 2 2 4 4 38 79 81
|
||||
99 2 2 4 4 77 78 79
|
||||
100 2 2 4 4 78 77 80
|
||||
101 2 2 4 4 77 79 83
|
||||
102 2 2 4 4 80 77 84
|
||||
103 2 2 4 4 79 78 81
|
||||
104 2 2 4 4 78 80 82
|
||||
105 2 2 5 5 21 3 90
|
||||
106 2 2 5 5 3 37 90
|
||||
107 2 2 5 5 4 22 91
|
||||
108 2 2 5 5 39 4 91
|
||||
109 2 2 5 5 7 29 92
|
||||
110 2 2 5 5 38 7 92
|
||||
111 2 2 5 5 30 8 89
|
||||
112 2 2 5 5 8 40 89
|
||||
113 2 2 5 5 22 21 85
|
||||
114 2 2 5 5 85 21 90
|
||||
115 2 2 5 5 22 85 91
|
||||
116 2 2 5 5 29 30 86
|
||||
117 2 2 5 5 29 86 92
|
||||
118 2 2 5 5 86 30 89
|
||||
119 2 2 5 5 37 38 88
|
||||
120 2 2 5 5 37 88 90
|
||||
121 2 2 5 5 88 38 92
|
||||
122 2 2 5 5 40 39 87
|
||||
123 2 2 5 5 87 39 91
|
||||
124 2 2 5 5 40 87 89
|
||||
125 2 2 5 5 85 86 87
|
||||
126 2 2 5 5 86 85 88
|
||||
127 2 2 5 5 85 87 91
|
||||
128 2 2 5 5 88 85 90
|
||||
129 2 2 5 5 87 86 89
|
||||
130 2 2 5 5 86 88 92
|
||||
131 2 2 6 6 1 24 99
|
||||
132 2 2 6 6 33 1 99
|
||||
133 2 2 6 6 23 4 98
|
||||
134 2 2 6 6 4 39 98
|
||||
135 2 2 6 6 32 5 97
|
||||
136 2 2 6 6 5 34 97
|
||||
137 2 2 6 6 8 31 100
|
||||
138 2 2 6 6 40 8 100
|
||||
139 2 2 6 6 24 23 93
|
||||
140 2 2 6 6 93 23 98
|
||||
141 2 2 6 6 24 93 99
|
||||
142 2 2 6 6 31 32 94
|
||||
143 2 2 6 6 31 94 100
|
||||
144 2 2 6 6 94 32 97
|
||||
145 2 2 6 6 34 33 95
|
||||
146 2 2 6 6 95 33 99
|
||||
147 2 2 6 6 34 95 97
|
||||
148 2 2 6 6 39 40 96
|
||||
149 2 2 6 6 39 96 98
|
||||
150 2 2 6 6 96 40 100
|
||||
151 2 2 6 6 93 94 95
|
||||
152 2 2 6 6 94 93 96
|
||||
153 2 2 6 6 93 95 99
|
||||
154 2 2 6 6 96 93 98
|
||||
155 2 2 6 6 95 94 97
|
||||
156 2 2 6 6 94 96 100
|
||||
157 2 2 7 7 9 41 102
|
||||
158 2 2 7 7 44 9 102
|
||||
159 2 2 7 7 41 10 104
|
||||
160 2 2 7 7 10 42 104
|
||||
161 2 2 7 7 42 11 103
|
||||
162 2 2 7 7 11 43 103
|
||||
163 2 2 7 7 43 12 101
|
||||
164 2 2 7 7 12 44 101
|
||||
165 2 2 7 7 102 41 104
|
||||
166 2 2 7 7 42 103 104
|
||||
167 2 2 7 7 43 101 103
|
||||
168 2 2 7 7 101 44 102
|
||||
169 2 2 7 7 101 102 103
|
||||
170 2 2 7 7 103 102 104
|
||||
171 2 2 8 8 13 45 106
|
||||
172 2 2 8 8 48 13 106
|
||||
173 2 2 8 8 45 14 108
|
||||
174 2 2 8 8 14 46 108
|
||||
175 2 2 8 8 46 15 107
|
||||
176 2 2 8 8 15 47 107
|
||||
177 2 2 8 8 47 16 105
|
||||
178 2 2 8 8 16 48 105
|
||||
179 2 2 8 8 106 45 108
|
||||
180 2 2 8 8 46 107 108
|
||||
181 2 2 8 8 47 105 107
|
||||
182 2 2 8 8 105 48 106
|
||||
183 2 2 8 8 105 106 107
|
||||
184 2 2 8 8 107 106 108
|
||||
185 2 2 9 9 41 9 109
|
||||
186 2 2 9 9 9 49 109
|
||||
187 2 2 9 9 10 41 110
|
||||
188 2 2 9 9 50 10 110
|
||||
189 2 2 9 9 13 45 111
|
||||
190 2 2 9 9 49 13 111
|
||||
191 2 2 9 9 45 14 112
|
||||
192 2 2 9 9 14 50 112
|
||||
193 2 2 9 9 41 109 110
|
||||
194 2 2 9 9 111 45 112
|
||||
195 2 2 9 9 109 49 111
|
||||
196 2 2 9 9 50 110 112
|
||||
197 2 2 9 9 110 109 111
|
||||
198 2 2 9 9 110 111 112
|
||||
199 2 2 10 10 42 10 113
|
||||
200 2 2 10 10 10 50 113
|
||||
201 2 2 10 10 11 42 114
|
||||
202 2 2 10 10 51 11 114
|
||||
203 2 2 10 10 14 46 115
|
||||
204 2 2 10 10 50 14 115
|
||||
205 2 2 10 10 46 15 116
|
||||
206 2 2 10 10 15 51 116
|
||||
207 2 2 10 10 42 113 114
|
||||
208 2 2 10 10 115 46 116
|
||||
209 2 2 10 10 113 50 115
|
||||
210 2 2 10 10 51 114 116
|
||||
211 2 2 10 10 114 113 115
|
||||
212 2 2 10 10 114 115 116
|
||||
213 2 2 11 11 43 11 119
|
||||
214 2 2 11 11 11 51 119
|
||||
215 2 2 11 11 12 43 117
|
||||
216 2 2 11 11 52 12 117
|
||||
217 2 2 11 11 15 47 120
|
||||
218 2 2 11 11 51 15 120
|
||||
219 2 2 11 11 47 16 118
|
||||
220 2 2 11 11 16 52 118
|
||||
221 2 2 11 11 117 43 119
|
||||
222 2 2 11 11 47 118 120
|
||||
223 2 2 11 11 119 51 120
|
||||
224 2 2 11 11 52 117 118
|
||||
225 2 2 11 11 118 117 119
|
||||
226 2 2 11 11 118 119 120
|
||||
227 2 2 12 12 9 44 121
|
||||
228 2 2 12 12 49 9 121
|
||||
229 2 2 12 12 44 12 123
|
||||
230 2 2 12 12 12 52 123
|
||||
231 2 2 12 12 48 13 122
|
||||
232 2 2 12 12 13 49 122
|
||||
233 2 2 12 12 16 48 124
|
||||
234 2 2 12 12 52 16 124
|
||||
235 2 2 12 12 121 44 123
|
||||
236 2 2 12 12 48 122 124
|
||||
237 2 2 12 12 49 121 122
|
||||
238 2 2 12 12 123 52 124
|
||||
239 2 2 12 12 122 121 123
|
||||
240 2 2 12 12 122 123 124
|
||||
241 4 2 1 1 105 62 106 107
|
||||
242 4 2 1 1 102 54 101 103
|
||||
243 4 2 1 1 118 86 120 119
|
||||
244 4 2 1 1 124 94 123 122
|
||||
245 4 2 1 1 52 39 12 96
|
||||
246 4 2 1 1 52 12 39 87
|
||||
247 4 2 1 1 88 38 15 51
|
||||
248 4 2 1 1 79 15 38 51
|
||||
249 4 2 1 1 80 14 50 36
|
||||
250 4 2 1 1 71 50 14 36
|
||||
251 4 2 1 1 120 88 15 51
|
||||
252 4 2 1 1 116 15 79 51
|
||||
253 4 2 1 1 50 14 112 71
|
||||
254 4 2 1 1 52 117 12 87
|
||||
255 4 2 1 1 111 69 109 110
|
||||
256 4 2 1 1 114 77 115 113
|
||||
257 4 2 1 1 124 123 94 96
|
||||
258 4 2 1 1 120 86 88 119
|
||||
259 4 2 1 1 14 45 64 108
|
||||
260 4 2 1 1 103 54 101 55
|
||||
261 4 2 1 1 62 105 63 107
|
||||
262 4 2 1 1 63 29 15 47
|
||||
263 4 2 1 1 14 45 26 64
|
||||
264 4 2 1 1 51 11 88 37
|
||||
265 4 2 1 1 80 50 10 35
|
||||
266 4 2 1 1 49 33 9 72
|
||||
267 4 2 1 1 71 10 50 35
|
||||
268 4 2 1 1 79 11 51 37
|
||||
269 4 2 1 1 49 9 33 95
|
||||
270 4 2 1 1 12 123 52 96
|
||||
271 4 2 1 1 96 16 52 40
|
||||
272 4 2 1 1 49 13 34 72
|
||||
273 4 2 1 1 87 52 16 40
|
||||
274 4 2 1 1 13 49 34 95
|
||||
275 4 2 1 1 43 101 12 55
|
||||
276 4 2 1 1 43 12 22 55
|
||||
277 4 2 1 1 106 61 108 107
|
||||
278 4 2 1 1 102 103 104 53
|
||||
279 4 2 1 1 50 115 14 80
|
||||
280 4 2 1 1 80 10 50 113
|
||||
281 4 2 1 1 9 109 49 72
|
||||
282 4 2 1 1 47 15 63 107
|
||||
283 4 2 1 1 124 16 52 96
|
||||
284 4 2 1 1 49 121 9 95
|
||||
285 4 2 1 1 15 28 68 46
|
||||
286 4 2 1 1 81 28 15 46
|
||||
287 4 2 1 1 92 15 29 47
|
||||
288 4 2 1 1 27 82 14 46
|
||||
289 4 2 1 1 14 66 27 46
|
||||
290 4 2 1 1 26 45 14 73
|
||||
291 4 2 1 1 78 115 116 114
|
||||
292 4 2 1 1 112 70 111 110
|
||||
293 4 2 1 1 79 11 114 51
|
||||
294 4 2 1 1 71 50 10 110
|
||||
295 4 2 1 1 10 41 104 56
|
||||
296 4 2 1 1 77 80 115 113
|
||||
297 4 2 1 1 69 109 72 111
|
||||
298 4 2 1 1 63 16 30 47
|
||||
299 4 2 1 1 25 45 13 64
|
||||
300 4 2 1 1 52 16 118 87
|
||||
301 4 2 1 1 95 13 49 122
|
||||
302 4 2 1 1 57 12 23 44
|
||||
303 4 2 1 1 119 11 88 51
|
||||
304 4 2 1 1 41 17 9 56
|
||||
305 4 2 1 1 18 41 10 56
|
||||
306 4 2 1 1 21 11 43 55
|
||||
307 4 2 1 1 63 105 16 47
|
||||
308 4 2 1 1 13 45 106 64
|
||||
309 4 2 1 1 9 102 41 56
|
||||
310 4 2 1 1 72 49 13 111
|
||||
311 4 2 1 1 11 103 43 55
|
||||
312 4 2 1 1 70 112 71 110
|
||||
313 4 2 1 1 116 79 78 114
|
||||
314 4 2 1 1 32 13 48 67
|
||||
315 4 2 1 1 16 89 30 47
|
||||
316 4 2 1 1 32 48 13 97
|
||||
317 4 2 1 1 48 16 100 31
|
||||
318 4 2 1 1 121 123 93 122
|
||||
319 4 2 1 1 119 117 118 85
|
||||
320 4 2 1 1 10 19 42 84
|
||||
321 4 2 1 1 17 9 76 41
|
||||
322 4 2 1 1 24 9 59 44
|
||||
323 4 2 1 1 42 19 10 58
|
||||
324 4 2 1 1 10 41 18 75
|
||||
325 4 2 1 1 20 83 11 42
|
||||
326 4 2 1 1 90 11 43 21
|
||||
327 4 2 1 1 53 104 102 56
|
||||
328 4 2 1 1 106 61 64 108
|
||||
329 4 2 1 1 87 118 117 85
|
||||
330 4 2 1 1 95 121 93 122
|
||||
331 4 2 1 1 80 50 115 113
|
||||
332 4 2 1 1 109 49 72 111
|
||||
333 4 2 1 1 78 27 28 46
|
||||
334 4 2 1 1 31 94 48 32
|
||||
335 4 2 1 1 19 77 20 42
|
||||
336 4 2 1 1 23 24 44 93
|
||||
337 4 2 1 1 123 124 52 96
|
||||
338 4 2 1 1 88 120 119 51
|
||||
339 4 2 1 1 91 43 85 117
|
||||
340 4 2 1 1 39 98 12 96
|
||||
341 4 2 1 1 12 91 39 87
|
||||
342 4 2 1 1 15 92 38 88
|
||||
343 4 2 1 1 38 81 15 79
|
||||
344 4 2 1 1 36 80 14 82
|
||||
345 4 2 1 1 14 71 36 73
|
||||
346 4 2 1 1 103 102 54 53
|
||||
347 4 2 1 1 61 106 62 107
|
||||
348 4 2 1 1 86 29 30 47
|
||||
349 4 2 1 1 26 45 70 25
|
||||
350 4 2 1 1 21 85 22 43
|
||||
351 4 2 1 1 94 93 123 122
|
||||
352 4 2 1 1 119 118 86 85
|
||||
353 4 2 1 1 12 96 93 123
|
||||
354 4 2 1 1 115 78 77 114
|
||||
355 4 2 1 1 69 111 70 110
|
||||
356 4 2 1 1 28 27 61 46
|
||||
357 4 2 1 1 48 62 31 32
|
||||
358 4 2 1 1 88 15 92 120
|
||||
359 4 2 1 1 79 81 15 116
|
||||
360 4 2 1 1 71 14 112 73
|
||||
361 4 2 1 1 114 116 79 51
|
||||
362 4 2 1 1 112 50 71 110
|
||||
363 4 2 1 1 91 12 117 87
|
||||
364 4 2 1 1 80 115 14 82
|
||||
365 4 2 1 1 63 15 29 68
|
||||
366 4 2 1 1 64 26 14 66
|
||||
367 4 2 1 1 47 63 105 107
|
||||
368 4 2 1 1 43 103 101 55
|
||||
369 4 2 1 1 35 10 80 84
|
||||
370 4 2 1 1 33 76 9 72
|
||||
371 4 2 1 1 37 11 88 90
|
||||
372 4 2 1 1 35 71 10 75
|
||||
373 4 2 1 1 37 79 11 83
|
||||
374 4 2 1 1 9 99 33 95
|
||||
375 4 2 1 1 68 63 15 107
|
||||
376 4 2 1 1 99 44 93 121
|
||||
377 4 2 1 1 18 69 41 17
|
||||
378 4 2 1 1 10 35 2 84
|
||||
379 4 2 1 1 15 7 28 81
|
||||
380 4 2 1 1 3 37 90 11
|
||||
381 4 2 1 1 108 64 14 66
|
||||
382 4 2 1 1 51 38 79 37
|
||||
383 4 2 1 1 12 4 23 98
|
||||
384 4 2 1 1 12 39 91 4
|
||||
385 4 2 1 1 49 33 72 34
|
||||
386 4 2 1 1 10 71 110 75
|
||||
387 4 2 1 1 114 11 79 83
|
||||
388 4 2 1 1 40 96 16 100
|
||||
389 4 2 1 1 34 13 74 72
|
||||
390 4 2 1 1 16 87 40 89
|
||||
391 4 2 1 1 34 97 13 95
|
||||
392 4 2 1 1 20 19 42 53
|
||||
393 4 2 1 1 54 23 24 44
|
||||
394 4 2 1 1 87 16 118 89
|
||||
395 4 2 1 1 95 97 13 122
|
||||
396 4 2 1 1 57 22 12 55
|
||||
397 4 2 1 1 41 102 104 56
|
||||
398 4 2 1 1 64 45 106 108
|
||||
399 4 2 1 1 101 57 12 55
|
||||
400 4 2 1 1 74 13 5 25
|
||||
401 4 2 1 1 13 34 74 5
|
||||
402 4 2 1 1 8 40 16 100
|
||||
403 4 2 1 1 16 31 65 8
|
||||
404 4 2 1 1 82 36 6 14
|
||||
405 4 2 1 1 102 9 59 56
|
||||
406 4 2 1 1 106 67 13 64
|
||||
407 4 2 1 1 119 88 11 90
|
||||
408 4 2 1 1 95 49 121 122
|
||||
409 4 2 1 1 52 118 117 87
|
||||
410 4 2 1 1 124 16 96 100
|
||||
411 4 2 1 1 80 10 113 84
|
||||
412 4 2 1 1 109 9 76 72
|
||||
413 4 2 1 1 72 13 70 111
|
||||
414 4 2 1 1 99 9 121 95
|
||||
415 4 2 1 1 30 65 16 63
|
||||
416 4 2 1 1 13 67 25 64
|
||||
417 4 2 1 1 11 55 53 103
|
||||
418 4 2 1 1 123 12 44 93
|
||||
419 4 2 1 1 56 104 10 58
|
||||
420 4 2 1 1 10 18 56 58
|
||||
421 4 2 1 1 21 60 11 55
|
||||
422 4 2 1 1 17 59 9 56
|
||||
423 4 2 1 1 16 63 62 105
|
||||
424 4 2 1 1 48 105 16 62
|
||||
425 4 2 1 1 111 45 13 70
|
||||
426 4 2 1 1 119 86 88 85
|
||||
427 4 2 1 1 123 93 94 96
|
||||
428 4 2 1 1 115 78 80 77
|
||||
429 4 2 1 1 111 72 69 70
|
||||
430 4 2 1 1 55 103 54 53
|
||||
431 4 2 1 1 63 61 62 107
|
||||
432 4 2 1 1 14 27 66 6
|
||||
433 4 2 1 1 99 9 1 24
|
||||
434 4 2 1 1 49 33 34 95
|
||||
435 4 2 1 1 96 52 39 40
|
||||
436 4 2 1 1 22 12 91 4
|
||||
437 4 2 1 1 20 11 83 3
|
||||
438 4 2 1 1 29 15 92 7
|
||||
439 4 2 1 1 18 10 75 2
|
||||
440 4 2 1 1 36 50 80 35
|
||||
441 4 2 1 1 18 17 41 56
|
||||
442 4 2 1 1 51 88 38 37
|
||||
443 4 2 1 1 62 31 16 48
|
||||
444 4 2 1 1 62 16 31 65
|
||||
445 4 2 1 1 26 45 25 64
|
||||
446 4 2 1 1 104 58 42 10
|
||||
447 4 2 1 1 29 63 30 47
|
||||
448 4 2 1 1 77 78 79 114
|
||||
449 4 2 1 1 71 69 70 110
|
||||
450 4 2 1 1 42 103 11 53
|
||||
451 4 2 1 1 93 94 95 122
|
||||
452 4 2 1 1 118 87 86 85
|
||||
453 4 2 1 1 61 62 106 64
|
||||
454 4 2 1 1 53 102 54 56
|
||||
455 4 2 1 1 78 82 27 46
|
||||
456 4 2 1 1 94 48 32 97
|
||||
457 4 2 1 1 86 30 89 47
|
||||
458 4 2 1 1 78 28 81 46
|
||||
459 4 2 1 1 26 70 45 73
|
||||
460 4 2 1 1 100 94 48 31
|
||||
461 4 2 1 1 86 92 29 47
|
||||
462 4 2 1 1 76 69 17 41
|
||||
463 4 2 1 1 42 19 77 84
|
||||
464 4 2 1 1 77 83 20 42
|
||||
465 4 2 1 1 18 41 69 75
|
||||
466 4 2 1 1 43 85 90 21
|
||||
467 4 2 1 1 91 43 117 12
|
||||
468 4 2 1 1 48 62 32 67
|
||||
469 4 2 1 1 27 66 61 46
|
||||
470 4 2 1 1 68 28 61 46
|
||||
471 4 2 1 1 54 24 59 44
|
||||
472 4 2 1 1 19 42 53 58
|
||||
473 4 2 1 1 23 54 57 44
|
||||
474 4 2 1 1 12 93 96 98
|
||||
475 4 2 1 1 99 44 121 9
|
||||
476 4 2 1 1 42 58 104 53
|
||||
477 4 2 1 1 21 43 22 55
|
||||
478 4 2 1 1 53 20 11 42
|
||||
479 4 2 1 1 53 11 20 60
|
||||
480 4 2 1 1 106 67 48 13
|
||||
481 4 2 1 1 36 71 50 35
|
||||
482 4 2 1 1 52 87 39 40
|
||||
483 4 2 1 1 48 67 106 62
|
||||
484 4 2 1 1 72 70 13 74
|
||||
485 4 2 1 1 9 33 99 1
|
||||
486 4 2 1 1 3 37 11 83
|
||||
487 4 2 1 1 13 34 5 97
|
||||
488 4 2 1 1 12 39 4 98
|
||||
489 4 2 1 1 14 36 6 73
|
||||
490 4 2 1 1 15 38 92 7
|
||||
491 4 2 1 1 8 40 89 16
|
||||
492 4 2 1 1 75 35 2 10
|
||||
493 4 2 1 1 30 16 65 8
|
||||
494 4 2 1 1 14 27 6 82
|
||||
495 4 2 1 1 26 14 66 6
|
||||
496 4 2 1 1 16 31 8 100
|
||||
497 4 2 1 1 32 13 67 5
|
||||
498 4 2 1 1 13 67 5 25
|
||||
499 4 2 1 1 68 7 28 15
|
||||
500 4 2 1 1 29 15 7 68
|
||||
501 4 2 1 1 57 4 23 12
|
||||
502 4 2 1 1 22 12 4 57
|
||||
503 4 2 1 1 10 19 84 2
|
||||
504 4 2 1 1 9 59 1 24
|
||||
505 4 2 1 1 17 9 59 1
|
||||
506 4 2 1 1 18 10 2 58
|
||||
507 4 2 1 1 11 21 90 3
|
||||
508 4 2 1 1 20 11 3 60
|
||||
509 4 2 1 1 16 62 63 65
|
||||
510 4 2 1 1 11 53 55 60
|
||||
511 4 2 1 1 78 81 79 116
|
||||
512 4 2 1 1 71 112 70 73
|
||||
513 4 2 1 1 86 88 92 120
|
||||
514 4 2 1 1 96 94 124 100
|
||||
515 4 2 1 1 94 97 95 122
|
||||
516 4 2 1 1 89 118 87 86
|
||||
517 4 2 1 1 15 38 7 81
|
||||
518 4 2 1 1 102 44 59 9
|
||||
519 4 2 1 1 59 44 102 54
|
||||
520 4 2 1 1 110 71 69 75
|
||||
521 4 2 1 1 79 77 114 83
|
||||
522 4 2 1 1 76 69 109 72
|
||||
523 4 2 1 1 113 77 80 84
|
||||
524 4 2 1 1 93 99 121 95
|
||||
525 4 2 1 1 87 117 91 85
|
||||
526 4 2 1 1 90 119 43 11
|
||||
527 4 2 1 1 80 78 115 82
|
||||
528 4 2 1 1 88 119 85 90
|
||||
529 4 2 1 1 90 43 119 85
|
||||
530 4 2 1 1 46 108 66 61
|
||||
531 4 2 1 1 46 66 108 14
|
||||
532 4 2 1 1 26 14 6 73
|
||||
533 4 2 1 1 44 101 57 12
|
||||
534 4 2 1 1 44 57 101 54
|
||||
535 4 2 1 1 46 107 68 15
|
||||
536 4 2 1 1 46 68 107 61
|
||||
537 4 2 1 1 9 33 1 76
|
||||
538 4 2 1 1 100 124 48 94
|
||||
539 4 2 1 1 100 48 124 16
|
||||
540 4 2 1 1 25 70 13 45
|
||||
541 4 2 1 1 56 53 104 58
|
||||
542 4 2 1 1 61 64 108 66
|
||||
543 4 2 1 1 106 62 67 64
|
||||
544 4 2 1 1 102 59 54 56
|
||||
545 4 2 1 1 32 13 5 97
|
||||
546 4 2 1 1 30 16 8 89
|
||||
547 4 2 1 1 13 70 25 74
|
||||
548 4 2 1 1 97 122 48 13
|
||||
549 4 2 1 1 75 110 41 69
|
||||
550 4 2 1 1 54 57 101 55
|
||||
551 4 2 1 1 97 48 122 94
|
||||
552 4 2 1 1 47 118 89 86
|
||||
553 4 2 1 1 47 89 118 16
|
||||
554 4 2 1 1 17 9 1 76
|
||||
555 4 2 1 1 11 21 3 60
|
||||
556 4 2 1 1 10 19 2 58
|
||||
557 4 2 1 1 75 41 110 10
|
||||
558 4 2 1 1 61 63 68 107
|
||||
559 4 2 1 1 70 111 45 112
|
||||
560 4 2 1 1 78 115 46 116
|
||||
561 4 2 1 1 77 114 42 113
|
||||
562 4 2 1 1 69 41 109 110
|
||||
563 4 2 1 1 107 61 108 46
|
||||
564 4 2 1 1 103 42 104 53
|
||||
565 4 2 1 1 118 86 47 120
|
||||
566 4 2 1 1 94 124 48 122
|
||||
567 4 2 1 1 117 119 43 85
|
||||
568 4 2 1 1 123 44 121 93
|
||||
569 4 2 1 1 101 54 102 44
|
||||
570 4 2 1 1 48 105 62 106
|
||||
571 4 2 1 1 91 43 12 22
|
||||
572 4 2 1 1 91 43 22 85
|
||||
573 4 2 1 1 93 12 23 98
|
||||
574 4 2 1 1 93 23 12 44
|
||||
575 4 2 1 1 46 81 116 15
|
||||
576 4 2 1 1 46 116 81 78
|
||||
577 4 2 1 1 92 47 120 15
|
||||
578 4 2 1 1 120 47 92 86
|
||||
579 4 2 1 1 46 82 115 78
|
||||
580 4 2 1 1 46 115 82 14
|
||||
581 4 2 1 1 73 45 112 70
|
||||
582 4 2 1 1 73 112 45 14
|
||||
583 4 2 1 1 99 44 9 24
|
||||
584 4 2 1 1 99 44 24 93
|
||||
585 4 2 1 1 84 42 113 77
|
||||
586 4 2 1 1 84 113 42 10
|
||||
587 4 2 1 1 76 41 109 69
|
||||
588 4 2 1 1 109 41 76 9
|
||||
589 4 2 1 1 42 83 114 77
|
||||
590 4 2 1 1 42 114 83 11
|
||||
591 4 2 2 2 135 13 122 131
|
||||
592 4 2 2 2 132 138 125 110
|
||||
593 4 2 2 2 138 108 125 106
|
||||
594 4 2 2 2 136 108 125 137
|
||||
595 4 2 2 2 110 138 125 131
|
||||
596 4 2 2 2 136 107 134 125
|
||||
597 4 2 2 2 122 135 131 125
|
||||
598 4 2 2 2 13 49 122 131
|
||||
599 4 2 2 2 136 108 137 46
|
||||
600 4 2 2 2 110 138 131 111
|
||||
601 4 2 2 2 132 138 110 112
|
||||
602 4 2 2 2 106 107 108 125
|
||||
603 4 2 2 2 138 112 137 45
|
||||
604 4 2 2 2 104 127 125 128
|
||||
605 4 2 2 2 113 115 114 125
|
||||
606 4 2 2 2 131 122 125 121
|
||||
607 4 2 2 2 112 45 14 137
|
||||
608 4 2 2 2 110 131 125 130
|
||||
609 4 2 2 2 115 116 114 125
|
||||
610 4 2 2 2 132 138 112 137
|
||||
611 4 2 2 2 50 137 115 132
|
||||
612 4 2 2 2 104 127 128 42
|
||||
613 4 2 2 2 105 134 125 133
|
||||
614 4 2 2 2 129 120 51 119
|
||||
615 4 2 2 2 104 130 128 125
|
||||
616 4 2 2 2 105 134 133 47
|
||||
617 4 2 2 2 135 106 138 125
|
||||
618 4 2 2 2 106 105 107 125
|
||||
619 4 2 2 2 138 108 106 45
|
||||
620 4 2 2 2 104 130 125 102
|
||||
621 4 2 2 2 104 127 42 103
|
||||
622 4 2 2 2 109 131 49 111
|
||||
623 4 2 2 2 106 135 48 105
|
||||
624 4 2 2 2 126 101 43 125
|
||||
625 4 2 2 2 138 137 125 108
|
||||
626 4 2 2 2 104 127 103 125
|
||||
627 4 2 2 2 116 114 129 51
|
||||
628 4 2 2 2 136 108 46 107
|
||||
629 4 2 2 2 136 108 107 125
|
||||
630 4 2 2 2 52 124 123 125
|
||||
631 4 2 2 2 132 112 110 50
|
||||
632 4 2 2 2 44 121 125 123
|
||||
633 4 2 2 2 131 122 121 49
|
||||
634 4 2 2 2 105 134 47 107
|
||||
635 4 2 2 2 106 13 135 138
|
||||
636 4 2 2 2 52 133 125 118
|
||||
637 4 2 2 2 103 43 101 125
|
||||
638 4 2 2 2 117 43 119 125
|
||||
639 4 2 2 2 105 134 107 125
|
||||
640 4 2 2 2 137 115 132 125
|
||||
641 4 2 2 2 50 137 132 112
|
||||
642 4 2 2 2 106 135 105 125
|
||||
643 4 2 2 2 102 101 44 125
|
||||
644 4 2 2 2 104 130 102 41
|
||||
645 4 2 2 2 138 45 13 111
|
||||
646 4 2 2 2 131 138 135 13
|
||||
647 4 2 2 2 138 112 45 111
|
||||
648 4 2 2 2 131 138 13 111
|
||||
649 4 2 2 2 107 134 15 136
|
||||
650 4 2 2 2 118 52 117 125
|
||||
651 4 2 2 2 101 126 12 44
|
||||
652 4 2 2 2 104 130 41 128
|
||||
653 4 2 2 2 138 137 108 45
|
||||
654 4 2 2 2 133 118 16 47
|
||||
655 4 2 2 2 44 101 126 125
|
||||
656 4 2 2 2 133 134 118 47
|
||||
657 4 2 2 2 133 134 125 118
|
||||
658 4 2 2 2 124 122 123 125
|
||||
659 4 2 2 2 129 118 119 125
|
||||
660 4 2 2 2 102 104 103 125
|
||||
661 4 2 2 2 135 48 13 106
|
||||
662 4 2 2 2 123 122 121 125
|
||||
663 4 2 2 2 103 127 43 125
|
||||
664 4 2 2 2 106 45 13 138
|
||||
665 4 2 2 2 103 11 127 42
|
||||
666 4 2 2 2 127 51 129 114
|
||||
667 4 2 2 2 13 49 131 111
|
||||
668 4 2 2 2 137 108 14 46
|
||||
669 4 2 2 2 15 47 134 107
|
||||
670 4 2 2 2 10 41 128 104
|
||||
671 4 2 2 2 16 118 133 52
|
||||
672 4 2 2 2 46 107 15 136
|
||||
673 4 2 2 2 41 102 9 130
|
||||
674 4 2 2 2 128 50 132 110
|
||||
675 4 2 2 2 10 41 110 128
|
||||
676 4 2 2 2 50 137 112 14
|
||||
677 4 2 2 2 130 102 9 44
|
||||
678 4 2 2 2 105 133 16 47
|
||||
679 4 2 2 2 127 11 103 43
|
||||
680 4 2 2 2 128 130 41 110
|
||||
681 4 2 2 2 116 134 15 51
|
||||
682 4 2 2 2 137 45 14 108
|
||||
683 4 2 2 2 12 126 101 43
|
||||
684 4 2 2 2 133 48 135 105
|
||||
685 4 2 2 2 128 42 10 104
|
||||
686 4 2 2 2 131 9 109 49
|
||||
687 4 2 2 2 118 117 119 125
|
||||
688 4 2 2 2 102 103 101 125
|
||||
689 4 2 2 2 129 118 125 134
|
||||
690 4 2 2 2 125 129 116 114
|
||||
691 4 2 2 2 117 126 43 125
|
||||
692 4 2 2 2 126 117 12 52
|
||||
693 4 2 2 2 52 126 117 125
|
||||
694 4 2 2 2 12 117 126 43
|
||||
695 4 2 2 2 127 51 114 11
|
||||
696 4 2 2 2 114 129 127 125
|
||||
697 4 2 2 2 127 119 43 125
|
||||
698 4 2 2 2 119 11 127 43
|
||||
699 4 2 2 2 127 113 114 125
|
||||
700 4 2 2 2 127 113 42 114
|
||||
701 4 2 2 2 128 130 110 125
|
||||
702 4 2 2 2 15 47 120 134
|
||||
703 4 2 2 2 127 11 114 42
|
||||
704 4 2 2 2 120 47 118 134
|
||||
705 4 2 2 2 44 130 102 125
|
||||
706 4 2 2 2 44 126 12 123
|
||||
707 4 2 2 2 123 44 126 125
|
||||
708 4 2 2 2 129 118 134 120
|
||||
709 4 2 2 2 110 132 128 125
|
||||
710 4 2 2 2 128 127 125 113
|
||||
711 4 2 2 2 128 127 113 42
|
||||
712 4 2 2 2 128 50 110 10
|
||||
713 4 2 2 2 113 42 10 128
|
||||
714 4 2 2 2 129 134 125 116
|
||||
715 4 2 2 2 127 51 11 119
|
||||
716 4 2 2 2 129 134 116 51
|
||||
717 4 2 2 2 127 51 119 129
|
||||
718 4 2 2 2 129 119 127 125
|
||||
719 4 2 2 2 110 131 130 109
|
||||
720 4 2 2 2 109 130 9 131
|
||||
721 4 2 2 2 121 130 9 44
|
||||
722 4 2 2 2 122 48 13 135
|
||||
723 4 2 2 2 44 121 130 125
|
||||
724 4 2 2 2 110 138 111 112
|
||||
725 4 2 2 2 131 138 125 135
|
||||
726 4 2 2 2 115 137 14 46
|
||||
727 4 2 2 2 126 52 12 123
|
||||
728 4 2 2 2 123 126 52 125
|
||||
729 4 2 2 2 50 137 14 115
|
||||
730 4 2 2 2 136 137 115 46
|
||||
731 4 2 2 2 121 9 131 49
|
||||
732 4 2 2 2 121 131 130 125
|
||||
733 4 2 2 2 131 130 9 121
|
||||
734 4 2 2 2 113 132 115 125
|
||||
735 4 2 2 2 113 50 115 132
|
||||
736 4 2 2 2 128 50 10 113
|
||||
737 4 2 2 2 132 113 128 125
|
||||
738 4 2 2 2 128 50 113 132
|
||||
739 4 2 2 2 133 105 135 125
|
||||
740 4 2 2 2 52 124 125 133
|
||||
741 4 2 2 2 133 48 105 16
|
||||
742 4 2 2 2 16 133 124 52
|
||||
743 4 2 2 2 136 137 125 115
|
||||
744 4 2 2 2 132 138 137 125
|
||||
745 4 2 2 2 41 130 9 109
|
||||
746 4 2 2 2 134 120 15 51
|
||||
747 4 2 2 2 129 118 120 119
|
||||
748 4 2 2 2 129 120 134 51
|
||||
749 4 2 2 2 135 122 124 125
|
||||
750 4 2 2 2 135 48 124 122
|
||||
751 4 2 2 2 133 48 16 124
|
||||
752 4 2 2 2 133 135 124 125
|
||||
753 4 2 2 2 133 48 124 135
|
||||
754 4 2 2 2 116 136 134 125
|
||||
755 4 2 2 2 115 136 116 125
|
||||
756 4 2 2 2 46 115 136 116
|
||||
757 4 2 2 2 136 134 15 116
|
||||
758 4 2 2 2 46 136 15 116
|
||||
759 4 2 2 2 109 41 130 110
|
||||
760 4 2 2 2 110 131 109 111
|
||||
$EndElements
|
||||
@@ -1,77 +0,0 @@
|
||||
// Square-in-square 2D geometry for MFEM
|
||||
// Creates concentric squares with different material attributes
|
||||
|
||||
// Define the square sizes
|
||||
L_outer = 2.0;
|
||||
L_inner = 0.5;
|
||||
|
||||
// Set mesh size and algorithm
|
||||
mesh_size = 1.0;
|
||||
Mesh.Algorithm = 6; // Frontal-Delaunay for 2D triangular mesh
|
||||
Mesh.CharacteristicLengthFactor = 1.0;
|
||||
Mesh.MshFileVersion = 2.2;
|
||||
|
||||
// Define center point for concentric squares
|
||||
cx = 0.0;
|
||||
cy = 0.0;
|
||||
|
||||
// Define the points (vertices of the outer square)
|
||||
Point(1) = {cx-L_outer/2, cy-L_outer/2, 0, mesh_size}; // bottom-left outer
|
||||
Point(2) = {cx+L_outer/2, cy-L_outer/2, 0, mesh_size}; // bottom-right outer
|
||||
Point(3) = {cx+L_outer/2, cy+L_outer/2, 0, mesh_size}; // top-right outer
|
||||
Point(4) = {cx-L_outer/2, cy+L_outer/2, 0, mesh_size}; // top-left outer
|
||||
|
||||
// Define the points (vertices of the inner square)
|
||||
Point(5) = {cx-L_inner/2, cy-L_inner/2, 0, mesh_size}; // bottom-left inner
|
||||
Point(6) = {cx+L_inner/2, cy-L_inner/2, 0, mesh_size}; // bottom-right inner
|
||||
Point(7) = {cx+L_inner/2, cy+L_inner/2, 0, mesh_size}; // top-right inner
|
||||
Point(8) = {cx-L_inner/2, cy+L_inner/2, 0, mesh_size}; // top-left inner
|
||||
|
||||
// Define the lines (edges of the outer square)
|
||||
Line(1) = {1, 2}; // bottom edge
|
||||
Line(2) = {2, 3}; // right edge
|
||||
Line(3) = {3, 4}; // top edge
|
||||
Line(4) = {4, 1}; // left edge
|
||||
|
||||
// Define the lines (edges of the inner square)
|
||||
Line(5) = {5, 6}; // bottom edge
|
||||
Line(6) = {6, 7}; // right edge
|
||||
Line(7) = {7, 8}; // top edge
|
||||
Line(8) = {8, 5}; // left edge
|
||||
|
||||
// Define the surfaces
|
||||
// Outer square boundary
|
||||
Line Loop(1) = {1, 2, 3, 4};
|
||||
|
||||
// Inner square boundary (hole in the outer region)
|
||||
Line Loop(2) = {5, 6, 7, 8};
|
||||
|
||||
// Define the surface areas
|
||||
// Outer region (annular region between squares)
|
||||
Plane Surface(1) = {1, 2}; // Outer loop minus inner loop (creates hole)
|
||||
|
||||
// Inner region (solid inner square)
|
||||
Plane Surface(2) = {2}; // Inner loop only
|
||||
|
||||
// Assign physical groups for materials
|
||||
Physical Surface(1) = {1}; // Outer material (annular region)
|
||||
Physical Surface(2) = {2}; // Inner material (solid square)
|
||||
|
||||
// Physical lines for boundary conditions
|
||||
// Outer square boundary edges
|
||||
Physical Line(1) = {1}; // outer bottom
|
||||
Physical Line(2) = {2}; // outer right
|
||||
Physical Line(3) = {3}; // outer top
|
||||
Physical Line(4) = {4}; // outer left
|
||||
|
||||
// Inner square boundary edges
|
||||
Physical Line(5) = {5}; // inner bottom
|
||||
Physical Line(6) = {6}; // inner right
|
||||
Physical Line(7) = {7}; // inner top
|
||||
Physical Line(8) = {8}; // inner left
|
||||
|
||||
// Mesh control for quality
|
||||
Mesh.OptimizeNetgen = 1;
|
||||
Mesh.Optimize = 1;
|
||||
Mesh.ElementOrder = 1;
|
||||
Mesh.RecombineAll = 0; // Keep triangular elements (don't recombine to quads)
|
||||
@@ -1,50 +0,0 @@
|
||||
$MeshFormat
|
||||
2.2 0 8
|
||||
$EndMeshFormat
|
||||
$Nodes
|
||||
13
|
||||
1 -1 -1 0
|
||||
2 1 -1 0
|
||||
3 1 1 0
|
||||
4 -1 1 0
|
||||
5 -0.25 -0.25 0
|
||||
6 0.25 -0.25 0
|
||||
7 0.25 0.25 0
|
||||
8 -0.25 0.25 0
|
||||
9 -2.752797989558076e-12 -1 0
|
||||
10 1 -2.752797989558076e-12 0
|
||||
11 2.752797989558076e-12 1 0
|
||||
12 -1 2.752797989558076e-12 0
|
||||
13 0 0 0
|
||||
$EndNodes
|
||||
$Elements
|
||||
28
|
||||
1 1 2 1 1 1 9
|
||||
2 1 2 1 1 9 2
|
||||
3 1 2 2 2 2 10
|
||||
4 1 2 2 2 10 3
|
||||
5 1 2 3 3 3 11
|
||||
6 1 2 3 3 11 4
|
||||
7 1 2 4 4 4 12
|
||||
8 1 2 4 4 12 1
|
||||
9 1 2 5 5 5 6
|
||||
10 1 2 6 6 6 7
|
||||
11 1 2 7 7 7 8
|
||||
12 1 2 8 8 8 5
|
||||
13 2 2 1 1 6 5 9
|
||||
14 2 2 1 1 5 8 12
|
||||
15 2 2 1 1 7 6 10
|
||||
16 2 2 1 1 8 7 11
|
||||
17 2 2 1 1 9 5 1
|
||||
18 2 2 1 1 5 12 1
|
||||
19 2 2 1 1 6 9 2
|
||||
20 2 2 1 1 10 6 2
|
||||
21 2 2 1 1 7 10 3
|
||||
22 2 2 1 1 11 7 3
|
||||
23 2 2 1 1 8 11 4
|
||||
24 2 2 1 1 8 4 12
|
||||
25 2 2 2 2 5 6 13
|
||||
26 2 2 2 2 8 5 13
|
||||
27 2 2 2 2 6 7 13
|
||||
28 2 2 2 2 7 8 13
|
||||
$EndElements
|
||||
@@ -1,38 +0,0 @@
|
||||
MFEM mesh v1.0
|
||||
|
||||
#
|
||||
# MFEM Geometry Types (see fem/geom.hpp):
|
||||
#
|
||||
# POINT = 0
|
||||
# SEGMENT = 1
|
||||
# TRIANGLE = 2
|
||||
# SQUARE = 3
|
||||
# TETRAHEDRON = 4
|
||||
# CUBE = 5
|
||||
# PRISM = 6
|
||||
# PYRAMID = 7
|
||||
|
||||
dimension
|
||||
2
|
||||
|
||||
elements
|
||||
2
|
||||
1 3 0 1 4 3
|
||||
1 2 1 2 4
|
||||
|
||||
boundary
|
||||
5
|
||||
1 1 0 1
|
||||
1 1 1 2
|
||||
1 1 2 4
|
||||
1 1 4 3
|
||||
1 1 3 0
|
||||
|
||||
vertices
|
||||
5
|
||||
2
|
||||
0 0
|
||||
1 0
|
||||
2 0
|
||||
0 1
|
||||
1 1
|
||||
@@ -1083,8 +1083,7 @@ EXCLUDE_PATTERNS =
|
||||
# ANamespace::AClass, ANamespace::*Test
|
||||
|
||||
EXCLUDE_SYMBOLS = mfem::internal \
|
||||
mfem::kernels::internal \
|
||||
mfem::future::detail
|
||||
mfem::kernels::internal
|
||||
|
||||
# The EXAMPLE_PATH tag can be used to specify one or more files or directories
|
||||
# that contain example code fragments that are included (see the \include
|
||||
|
||||
@@ -201,7 +201,6 @@ namespace mfem {
|
||||
* - <a class="el" href="nurbs__naca__cmesh_8cpp_source.html">NURBS NACA Mesher</a>: generate NURBS based mesh around a NACA foil
|
||||
* - <a class="el" href="nurbs__printfunc_8cpp_source.html">NURBS Printer</a>: print the NURBS-basis
|
||||
* - <a class="el" href="nurbs__mesh_info_8cpp_source.html">NURBS Mesh info</a>: print the info of a NURBS mesh
|
||||
* - <a class="el" href="nurbs__surface_8cpp_source.html">NURBS Surface</a>: interpolate a 3D Surface in a NURBS Patch
|
||||
*
|
||||
* <H3>Miniapps</H3>
|
||||
* - <a class="el" href="volta_8cpp_source.html">Volta</a>: simple electrostatics simulation code
|
||||
@@ -246,9 +245,6 @@ namespace mfem {
|
||||
* - <a class="el" href="pdiffusion_8cpp_source.html">DPG Diffusion example</a>: DPG formulation for the diffusion problem
|
||||
* - <a class="el" href="pmaxwell_8cpp_source.html">DPG Maxwell example</a>: DPG formulation for the indefinite Maxwell problem
|
||||
* - <a class="el" href="lor__elast_8cpp_source.html">LOR Elasticity</a>: solve linear elasticity with LOR preconditioning on GPUs
|
||||
* - <a class="el" href="reflector_8cpp_source.html">Reflector Miniapp</a>: reflect a mesh about a plane
|
||||
* - <a class="el" href="ref321_8cpp_source.html">3:1 Refinement Miniapp</a>: perform 3:1 anisotropic mesh refinements
|
||||
* - <a class="el" href="pref321_8cpp_source.html">3:1 Refinement Miniapp</a>: parallel 3:1 anisotropic mesh refinements
|
||||
*
|
||||
* See also the <a class="el" href="https://mfem.org/examples/">examples documentation</a> online.
|
||||
*/
|
||||
|
||||
+21
-34
@@ -50,10 +50,6 @@
|
||||
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cpu
|
||||
// ex1 -m ../data/beam-tet.mesh -pa -d ceed-cuda:/gpu/cuda/ref
|
||||
//
|
||||
// Device simplices sample runs:
|
||||
// ex1 -pa -d gpu -m ../data/inline-tet.mesh
|
||||
// ex1 -pa -d gpu -m ../data/inline-tri.mesh
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
// -Delta u = 1 with homogeneous Dirichlet boundary conditions.
|
||||
@@ -142,25 +138,25 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 5. Define a finite element space on the mesh. Here we use continuous
|
||||
// Lagrange finite elements of the specified order.
|
||||
// - If order < 1, we instead use an isoparametric/isogeometric space.
|
||||
// - If the mesh is simplicial and partial assembly is requested,
|
||||
// we use the positive basis, which supports device execution.
|
||||
// Lagrange finite elements of the specified order. If order < 1, we
|
||||
// instead use an isoparametric/isogeometric space.
|
||||
FiniteElementCollection *fec;
|
||||
auto basis_type = (pa && mesh.IsSimplexMesh()) ?
|
||||
BasisType::Positive : BasisType::GaussLobatto;
|
||||
bool delete_fec;
|
||||
if (order > 0)
|
||||
{
|
||||
fec = new H1_FECollection(order, dim, basis_type);
|
||||
fec = new H1_FECollection(order, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
else if (mesh.GetNodes())
|
||||
{
|
||||
fec = mesh.GetNodes()->OwnFEC();
|
||||
delete_fec = false;
|
||||
cout << "Using isoparametric FEs: " << fec->Name() << endl;
|
||||
}
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order = 1, dim, basis_type);
|
||||
fec = new H1_FECollection(order = 1, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
FiniteElementSpace fespace(&mesh, fec);
|
||||
cout << "Number of finite element unknowns: "
|
||||
@@ -228,29 +224,17 @@ int main(int argc, char *argv[])
|
||||
// 11. Solve the linear system A X = B.
|
||||
if (!pa)
|
||||
{
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
if (Device::Allows(Backend::CUDA_MASK))
|
||||
{
|
||||
// Use cuDSS to solve the system.
|
||||
CuDSSSolver cudss_solver;
|
||||
cudss_solver.SetOperator(*A);
|
||||
cudss_solver.Mult(B, X);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
#ifndef MFEM_USE_SUITESPARSE
|
||||
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
|
||||
GSSmoother M((SparseMatrix&)(*A));
|
||||
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
|
||||
// Use a simple symmetric Gauss-Seidel preconditioner with PCG.
|
||||
GSSmoother M((SparseMatrix&)(*A));
|
||||
PCG(*A, M, B, X, 1, 200, 1e-12, 0.0);
|
||||
#else
|
||||
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
|
||||
UMFPackSolver umf_solver;
|
||||
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
umf_solver.SetOperator(*A);
|
||||
umf_solver.Mult(B, X);
|
||||
// If MFEM was compiled with SuiteSparse, use UMFPACK to solve the system.
|
||||
UMFPackSolver umf_solver;
|
||||
umf_solver.Control[UMFPACK_ORDERING] = UMFPACK_ORDERING_METIS;
|
||||
umf_solver.SetOperator(*A);
|
||||
umf_solver.Mult(B, X);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -289,14 +273,17 @@ int main(int argc, char *argv[])
|
||||
if (visualization)
|
||||
{
|
||||
char vishost[] = "localhost";
|
||||
int visport = 19916;
|
||||
int visport = 19916;
|
||||
socketstream sol_sock(vishost, visport);
|
||||
sol_sock.precision(8);
|
||||
sol_sock << "solution\n" << mesh << x << flush;
|
||||
}
|
||||
|
||||
// 15. Free the used memory.
|
||||
if (order > 0) { delete fec; }
|
||||
if (delete_fec)
|
||||
{
|
||||
delete fec;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
+34
-60
@@ -42,11 +42,7 @@
|
||||
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared
|
||||
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/square-mixed.mesh
|
||||
// mpirun -np 4 ex1p -pa -d ceed-cuda:/gpu/cuda/shared -m ../data/fichera-mixed.mesh
|
||||
// mpirun -np 4 ex1p -pa -d ceed-cpu -m ../data/beam-tet.mesh
|
||||
//
|
||||
// Device simplices sample runs:
|
||||
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tet.mesh
|
||||
// mpirun -np 4 ex1p -pa -d gpu -m ../data/inline-tri.mesh
|
||||
// mpirun -np 4 ex1p -m ../data/beam-tet.mesh -pa -d ceed-cpu
|
||||
//
|
||||
// Description: This example code demonstrates the use of MFEM to define a
|
||||
// simple finite element discretization of the Poisson problem
|
||||
@@ -87,9 +83,6 @@ int main(int argc, char *argv[])
|
||||
const char *device_config = "cpu";
|
||||
bool visualization = true;
|
||||
bool algebraic_ceed = false;
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
bool cudss_solver = false;
|
||||
#endif
|
||||
|
||||
OptionsParser args(argc, argv);
|
||||
args.AddOption(&mesh_file, "-m", "--mesh",
|
||||
@@ -109,10 +102,6 @@ int main(int argc, char *argv[])
|
||||
args.AddOption(&algebraic_ceed, "-a", "--algebraic",
|
||||
"-no-a", "--no-algebraic",
|
||||
"Use algebraic Ceed solver");
|
||||
#endif
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
args.AddOption(&cudss_solver, "-cudss", "--cudss-solver", "-no-cudss",
|
||||
"--no-cudss-solver", "Use the cuDSS Solver.");
|
||||
#endif
|
||||
args.AddOption(&visualization, "-vis", "--visualization", "-no-vis",
|
||||
"--no-visualization",
|
||||
@@ -169,20 +158,19 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 7. Define a parallel finite element space on the parallel mesh. Here we
|
||||
// use continuous Lagrange finite elements of the specified order.
|
||||
// - If order < 1, we instead use an isoparametric/isogeometric space.
|
||||
// - If the mesh is simplicial and partial assembly is requested,
|
||||
// we use the positive basis, which supports device execution.
|
||||
// use continuous Lagrange finite elements of the specified order. If
|
||||
// order < 1, we instead use an isoparametric/isogeometric space.
|
||||
FiniteElementCollection *fec;
|
||||
auto basis_type = (pa && pmesh.IsSimplexMesh()) ?
|
||||
BasisType::Positive : BasisType::GaussLobatto;
|
||||
bool delete_fec;
|
||||
if (order > 0)
|
||||
{
|
||||
fec = new H1_FECollection(order, dim, basis_type);
|
||||
fec = new H1_FECollection(order, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
else if (pmesh.GetNodes())
|
||||
{
|
||||
fec = pmesh.GetNodes()->OwnFEC();
|
||||
delete_fec = false;
|
||||
if (myid == 0)
|
||||
{
|
||||
cout << "Using isoparametric FEs: " << fec->Name() << endl;
|
||||
@@ -190,7 +178,8 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
else
|
||||
{
|
||||
fec = new H1_FECollection(order = 1, dim, basis_type);
|
||||
fec = new H1_FECollection(order = 1, dim);
|
||||
delete_fec = true;
|
||||
}
|
||||
ParFiniteElementSpace fespace(&pmesh, fec);
|
||||
HYPRE_BigInt size = fespace.GlobalTrueVSize();
|
||||
@@ -259,51 +248,33 @@ int main(int argc, char *argv[])
|
||||
// 13. Solve the linear system A X = B.
|
||||
// * With full assembly, use the BoomerAMG preconditioner from hypre.
|
||||
// * With partial assembly, use Jacobi smoothing, for now.
|
||||
#ifdef MFEM_USE_CUDSS
|
||||
if (!pa && (Device::Allows(Backend::CUDA_MASK) && cudss_solver))
|
||||
Solver *prec = NULL;
|
||||
if (pa)
|
||||
{
|
||||
// Solve using a direct solver with cuDSS
|
||||
CuDSSSolver cudss_solver(MPI_COMM_WORLD);
|
||||
cudss_solver.SetMatrixSymType(
|
||||
CuDSSSolver::SYMMETRIC_POSITIVE_DEFINITE);
|
||||
cudss_solver.SetMatrixViewType(CuDSSSolver::UPPER);
|
||||
cudss_solver.SetOperator(*A);
|
||||
cudss_solver.Mult(B, X);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
Solver *prec = NULL;
|
||||
if (pa)
|
||||
if (UsesTensorBasis(fespace))
|
||||
{
|
||||
if (UsesTensorBasis(fespace))
|
||||
if (algebraic_ceed)
|
||||
{
|
||||
if (algebraic_ceed)
|
||||
{
|
||||
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
|
||||
}
|
||||
else
|
||||
{
|
||||
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
|
||||
}
|
||||
prec = new ceed::AlgebraicSolver(a, ess_tdof_list);
|
||||
}
|
||||
else
|
||||
{
|
||||
prec = new OperatorJacobiSmoother(a, ess_tdof_list);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
prec = new HypreBoomerAMG;
|
||||
}
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetPrintLevel(1);
|
||||
if (prec)
|
||||
{
|
||||
cg.SetPreconditioner(*prec);
|
||||
}
|
||||
cg.SetOperator(*A);
|
||||
cg.Mult(B, X);
|
||||
delete prec;
|
||||
}
|
||||
else
|
||||
{
|
||||
prec = new HypreBoomerAMG;
|
||||
}
|
||||
CGSolver cg(MPI_COMM_WORLD);
|
||||
cg.SetRelTol(1e-12);
|
||||
cg.SetMaxIter(2000);
|
||||
cg.SetPrintLevel(1);
|
||||
if (prec) { cg.SetPreconditioner(*prec); }
|
||||
cg.SetOperator(*A);
|
||||
cg.Mult(B, X);
|
||||
delete prec;
|
||||
|
||||
// 14. Recover the parallel grid function corresponding to X. This is the
|
||||
// local finite element solution on each processor.
|
||||
@@ -337,7 +308,10 @@ int main(int argc, char *argv[])
|
||||
}
|
||||
|
||||
// 17. Free the used memory.
|
||||
if (order > 0) { delete fec; }
|
||||
if (delete_fec)
|
||||
{
|
||||
delete fec;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -95,15 +95,6 @@ int main(int argc, char *argv[])
|
||||
args.PrintOptions(cout);
|
||||
}
|
||||
|
||||
if (amg_elast && !static_cond && reorder_space)
|
||||
{
|
||||
if (myid == 0)
|
||||
cerr << "\nThe AMG elasticity solver requires ordering byVDIM! "
|
||||
<< "Ignoring the specified option -nodes/--by-nodes.\n"
|
||||
<< endl;
|
||||
reorder_space = false;
|
||||
}
|
||||
|
||||
// 3. Enable hardware devices such as GPUs, and programming models such as
|
||||
// CUDA, OCCA, RAJA and OpenMP based on command line options.
|
||||
Device device(device_config);
|
||||
|
||||
+1
-14
@@ -57,8 +57,6 @@ set(SRCS
|
||||
integ/lininteg_domain_grad.cpp
|
||||
integ/lininteg_domain_vectorfe.cpp
|
||||
integ/nonlininteg_vecconvection_pa.cpp
|
||||
integ/nonlininteg_vecconvection_pa_diag.cpp
|
||||
integ/nonlininteg_vecconvection_pa_grad.cpp
|
||||
integ/nonlininteg_vecconvection_mf.cpp
|
||||
coefficient.cpp
|
||||
complex_fem.cpp
|
||||
@@ -135,7 +133,7 @@ set(SRCS
|
||||
tmop/assemble/diag2.cpp
|
||||
tmop/assemble/grad2_limit.cpp
|
||||
tmop/assemble/grad2.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3_limit.cpp
|
||||
tmop/assemble/diag3.cpp
|
||||
tmop/assemble/grad3_limit.cpp
|
||||
tmop/assemble/grad3.cpp
|
||||
@@ -173,12 +171,8 @@ set(SRCS
|
||||
tmop_tools.cpp
|
||||
tmop_amr.cpp
|
||||
gslib.cpp
|
||||
gslib/findptsedge_local_2.cpp
|
||||
gslib/findptsedge_local_3.cpp
|
||||
gslib/findptssurf_local_3.cpp
|
||||
gslib/findpts_local_2.cpp
|
||||
gslib/findpts_local_3.cpp
|
||||
gslib/interpolate_local_1.cpp
|
||||
gslib/interpolate_local_2.cpp
|
||||
gslib/interpolate_local_3.cpp
|
||||
transfer.cpp
|
||||
@@ -197,20 +191,14 @@ set(HDRS
|
||||
integ/bilininteg_dgtrace_kernels.hpp
|
||||
integ/bilininteg_vecdiffusion_kernels.hpp
|
||||
integ/bilininteg_convection_kernels.hpp
|
||||
integ/bilininteg_diffusion_pa_simplices.hpp
|
||||
integ/bilininteg_diffusion_kernels.hpp
|
||||
integ/bilininteg_elasticity_kernels.hpp
|
||||
integ/bilininteg_hcurl_kernels.hpp
|
||||
integ/bilininteg_hdiv_kernels.hpp
|
||||
integ/bilininteg_hcurlhdiv_kernels.hpp
|
||||
integ/bilininteg_mass_kernels.hpp
|
||||
integ/bilininteg_mass_pa_simplices.hpp
|
||||
integ/bilininteg_vecdiffusion_pa.hpp
|
||||
integ/bilininteg_vecdiv_pa.hpp
|
||||
integ/bilininteg_vecmass_pa.hpp
|
||||
integ/nonlininteg_vecconvection_pa.hpp
|
||||
integ/nonlininteg_vecconvection_pa_diag.hpp
|
||||
integ/nonlininteg_vecconvection_pa_grad.hpp
|
||||
coefficient.hpp
|
||||
complex_fem.hpp
|
||||
convergence.hpp
|
||||
@@ -317,7 +305,6 @@ set(HDRS
|
||||
tmop_tools.hpp
|
||||
tmop_amr.hpp
|
||||
gslib.hpp
|
||||
gslib/gslib_kernel_helpers.hpp
|
||||
transfer.hpp
|
||||
hyperbolic.hpp
|
||||
integrator.hpp
|
||||
|
||||
@@ -1255,31 +1255,6 @@ void BilinearForm::Mult(const Vector &x, Vector &y) const
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::AddMult(const Vector &x, Vector &y, const real_t a) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
ext->AddMult(x, y, a);
|
||||
}
|
||||
else
|
||||
{
|
||||
mat->AddMult(x, y, a);
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::AddMultTranspose(const Vector &x, Vector &y,
|
||||
const real_t a) const
|
||||
{
|
||||
if (ext)
|
||||
{
|
||||
ext->AddMultTranspose(x, y, a);
|
||||
}
|
||||
else
|
||||
{
|
||||
mat->AddMultTranspose(x, y, a);
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::MultTranspose(const Vector & x, Vector & y) const
|
||||
{
|
||||
if (ext)
|
||||
|
||||
@@ -307,8 +307,8 @@ public:
|
||||
{ mat->Mult(x, y); mat_e->AddMult(x, y); }
|
||||
|
||||
/// Add the matrix vector multiple to a vector: $ y += a M x $
|
||||
void AddMult(const Vector &x, Vector &y,
|
||||
const real_t a = 1.0) const override;
|
||||
void AddMult(const Vector &x, Vector &y, const real_t a = 1.0) const override
|
||||
{ mat -> AddMult (x, y, a); }
|
||||
|
||||
/** @brief Add the original uneliminated matrix vector multiple to a vector.
|
||||
The original matrix is $ M + Me $ so we have:
|
||||
@@ -318,7 +318,8 @@ public:
|
||||
|
||||
/// Add the matrix transpose vector multiplication: $ y += a M^T x $
|
||||
void AddMultTranspose(const Vector & x, Vector & y,
|
||||
const real_t a = 1.0) const override;
|
||||
const real_t a = 1.0) const override
|
||||
{ mat->AddMultTranspose(x, y, a); }
|
||||
|
||||
/** @brief Add the original uneliminated matrix transpose vector
|
||||
multiple to a vector. The original matrix is $ M + M_e $
|
||||
|
||||
@@ -1997,11 +1997,7 @@ void PADiscreteLinearOperatorExtension::Assemble()
|
||||
}
|
||||
else
|
||||
{
|
||||
const L2ElementRestriction* l2_elem_restrict =
|
||||
dynamic_cast<const L2ElementRestriction*>(elem_restrict_test);
|
||||
MFEM_VERIFY(l2_elem_restrict,
|
||||
"A real ElementRestriction is required in this setting!");
|
||||
test_multiplicity = 1.0;
|
||||
mfem_error("A real ElementRestriction is required in this setting!");
|
||||
}
|
||||
|
||||
auto tm = test_multiplicity.ReadWrite();
|
||||
@@ -2040,13 +2036,7 @@ void PADiscreteLinearOperatorExtension::AddMult(
|
||||
}
|
||||
else
|
||||
{
|
||||
const L2ElementRestriction* l2_elem_restrict =
|
||||
dynamic_cast<const L2ElementRestriction*>(elem_restrict_test);
|
||||
MFEM_VERIFY(l2_elem_restrict,
|
||||
"In this setting you need a real ElementRestriction!");
|
||||
tempY.SetSize(y.Size());
|
||||
l2_elem_restrict->MultTranspose(localTest, tempY);
|
||||
y += tempY;
|
||||
mfem_error("In this setting you need a real ElementRestriction!");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+4
-22
@@ -1345,8 +1345,7 @@ real_t DiffusionIntegrator::ComputeFluxEnergy
|
||||
}
|
||||
|
||||
const IntegrationRule &DiffusionIntegrator::GetRule(
|
||||
const FiniteElement &trial_fe, const FiniteElement &test_fe,
|
||||
const bool stroud)
|
||||
const FiniteElement &trial_fe, const FiniteElement &test_fe)
|
||||
{
|
||||
int order;
|
||||
if (trial_fe.Space() == FunctionSpace::Pk)
|
||||
@@ -1363,15 +1362,7 @@ const IntegrationRule &DiffusionIntegrator::GetRule(
|
||||
{
|
||||
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
|
||||
if (stroud)
|
||||
{
|
||||
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
else
|
||||
{
|
||||
return IntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
return IntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
|
||||
MassIntegrator::MassIntegrator(const IntegrationRule *ir)
|
||||
@@ -1458,8 +1449,7 @@ void MassIntegrator::AssembleElementMatrix2(
|
||||
|
||||
const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
|
||||
const FiniteElement &test_fe,
|
||||
const ElementTransformation &Trans,
|
||||
const bool stroud)
|
||||
const ElementTransformation &Trans)
|
||||
{
|
||||
// int order = trial_fe.GetOrder() + test_fe.GetOrder();
|
||||
const int order = trial_fe.GetOrder() + test_fe.GetOrder() + Trans.OrderW();
|
||||
@@ -1468,15 +1458,7 @@ const IntegrationRule &MassIntegrator::GetRule(const FiniteElement &trial_fe,
|
||||
{
|
||||
return RefinedIntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
|
||||
if (stroud)
|
||||
{
|
||||
return StroudIntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
else
|
||||
{
|
||||
return IntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
return IntRules.Get(trial_fe.GetGeomType(), order);
|
||||
}
|
||||
|
||||
|
||||
|
||||
+334
-534
File diff suppressed because it is too large
Load Diff
+17
-1
@@ -41,9 +41,14 @@ void PLBound::Setup(const int nb_i, const int ncp_i,
|
||||
tol = tol_i;
|
||||
lbound.SetSize(ncp, nb);
|
||||
ubound.SetSize(ncp, nb);
|
||||
lbound_t.SetSize(nb, ncp);
|
||||
ubound_t.SetSize(nb, ncp);
|
||||
nodes.SetSize(nb);
|
||||
weights.SetSize(nb);
|
||||
control_points.SetSize(ncp);
|
||||
xhat.SetSize(nb);
|
||||
what.SetSize(nb);
|
||||
cphat.SetSize(ncp);
|
||||
|
||||
auto scalenodes = [](const Vector &in, const real_t a, const real_t b) -> Vector
|
||||
{
|
||||
@@ -90,6 +95,10 @@ void PLBound::Setup(const int nb_i, const int ncp_i,
|
||||
MFEM_ABORT("Unsupported interval points. Use [0,1].\n");
|
||||
}
|
||||
control_points = scalenodes(control_points, 0.0, 1.0); // rescale to [0,1]
|
||||
for (int i = 0; i < ncp; i++)
|
||||
{
|
||||
cphat(i) = 2.0*control_points(i) - 1.0;
|
||||
}
|
||||
|
||||
Poly_1D::Basis &basis1d(poly1d.GetBasis(nb-1, b_type));
|
||||
|
||||
@@ -145,6 +154,8 @@ void PLBound::Setup(const int nb_i, const int ncp_i,
|
||||
lbound(j,i) = std::max(lbound(j,i),0_r);
|
||||
}
|
||||
}
|
||||
lbound_t(i,j) = lbound(j,i);
|
||||
ubound_t(i,j) = ubound(j,i);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -176,6 +187,11 @@ void PLBound::Setup(const int nb_i, const int ncp_i,
|
||||
nodes(i) = irule.IntPoint(i).x;
|
||||
}
|
||||
}
|
||||
for (int i = 0; i < nb; i++)
|
||||
{
|
||||
xhat(i) = 2.0*nodes(i) - 1.0;
|
||||
what(i) = 2.0*weights(i);
|
||||
}
|
||||
|
||||
if (b_type == 2)
|
||||
{
|
||||
@@ -755,4 +771,4 @@ void PLBound::Print(std::ostream &outp) const
|
||||
ubound.Print(outp);
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
+615
-1
@@ -13,6 +13,7 @@
|
||||
#define MFEM_BOUNDS
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#include "../general/forall.hpp"
|
||||
#include "fespace.hpp"
|
||||
|
||||
namespace mfem
|
||||
@@ -60,7 +61,9 @@ private:
|
||||
bool proj = true; // Use linear projection to compute bounds.
|
||||
real_t tol = 0.0; // offset bounds to avoid round-off errors
|
||||
Vector nodes, weights, control_points;
|
||||
Vector xhat, what, cphat;
|
||||
DenseMatrix lbound, ubound; // ncp x nb matrices with bounds of all bases
|
||||
DenseMatrix lbound_t, ubound_t; // nb x ncp transposes for device kernel
|
||||
// Some auxillary storage for computing the bounds with Bernstein
|
||||
DenseMatrix basisMatNodes; // Bernstein bases at equispaced nodes
|
||||
DenseMatrix basisMatInt; // Bernstein bases at GLL nodes
|
||||
@@ -113,7 +116,10 @@ public:
|
||||
* @details This projection increases the computational cost but results in
|
||||
* tighter bounds.
|
||||
*/
|
||||
void SetProjectionFlagForBounding(bool proj_) { proj = proj_; }
|
||||
void SetProjectionFlagForBounding(bool proj_)
|
||||
{
|
||||
proj = proj_;
|
||||
}
|
||||
|
||||
/** @brief Compute piecewise linear bounds for the lexicographically-ordered
|
||||
* nodal coefficients in @a coeff in 1D/2D/3D.
|
||||
@@ -137,9 +143,23 @@ public:
|
||||
/// Get number of control points used to compute the bounds.
|
||||
int GetNControlPoints() const { return ncp; }
|
||||
|
||||
/// Get the underlying 1D basis type.
|
||||
int GetBasisType() const { return b_type; }
|
||||
|
||||
/// Get 1D control point locations (lexicographic order) in [0,1].
|
||||
const Vector &GetControlPoints() const { return control_points; }
|
||||
|
||||
/** @brief Compute element-wise bounds from a lexicographic E-vector.
|
||||
*
|
||||
* @details The expected layout of @a e_vec is `ND x VDIM x NE`, where
|
||||
* `ND = nb^rdim`, `VDIM = fes_vdim`, and `NE` is the number of elements.
|
||||
* The output layout matches GridFunction::GetElementBounds:
|
||||
* `NE x active_vdim`, with the element index varying fastest.
|
||||
*/
|
||||
void GetElementBoundsKernel(const int rdim, const int fes_vdim,
|
||||
const Vector &e_vec, Vector &lower,
|
||||
Vector &upper, const int vdim = 0) const;
|
||||
|
||||
/** @brief Get lower and upper bounding matrix (ncp^dim x nb^dim)
|
||||
*
|
||||
* @details The matrices can be used to compute the bounds at control points
|
||||
@@ -183,6 +203,600 @@ private:
|
||||
const int cp_type_i, const real_t tol_i);
|
||||
};
|
||||
|
||||
namespace internal
|
||||
{
|
||||
|
||||
struct PLBoundDeviceData
|
||||
{
|
||||
int nb;
|
||||
int ncp;
|
||||
const real_t *xhat;
|
||||
const real_t *what;
|
||||
const real_t *cphat;
|
||||
const real_t *lbound;
|
||||
const real_t *ubound;
|
||||
};
|
||||
|
||||
template<int T_NB = 0, bool T_PROJ = true>
|
||||
inline void GetElementBoundsKernel1D(const PLBoundDeviceData &data,
|
||||
const int fes_vdim,
|
||||
const int ne,
|
||||
const Vector &e_vec,
|
||||
Vector &lower,
|
||||
Vector &upper,
|
||||
const int comp0,
|
||||
const int ncomp)
|
||||
{
|
||||
constexpr int GENERIC_MAX_ND = 32;
|
||||
constexpr int MAX_ND = T_NB ? T_NB : GENERIC_MAX_ND;
|
||||
constexpr int BLOCK_X = 2*MAX_ND;
|
||||
|
||||
const int nd = T_NB ? T_NB : data.nb;
|
||||
MFEM_VERIFY(nd <= MAX_ND,
|
||||
"Device element bounds kernel supports up to 32 "
|
||||
"1D degrees of freedom.");
|
||||
|
||||
const auto E = Reshape(e_vec.Read(), nd, fes_vdim, ne);
|
||||
auto L = Reshape(lower.Write(), ne, ncomp);
|
||||
auto U = Reshape(upper.Write(), ne, ncomp);
|
||||
|
||||
mfem::forall_2D<BLOCK_X>(ne*ncomp, BLOCK_X, 1,
|
||||
[=] MFEM_HOST_DEVICE (int ec)
|
||||
{
|
||||
const int e = ec % ne;
|
||||
const int c = ec / ne;
|
||||
const int vc = comp0 + c;
|
||||
const real_t *coeff = &E(0, vc, e);
|
||||
const int tid = MFEM_THREAD_ID(x);
|
||||
|
||||
MFEM_SHARED real_t sproj[MAX_ND];
|
||||
MFEM_SHARED real_t ssum0[MAX_ND];
|
||||
MFEM_SHARED real_t ssum1[MAX_ND];
|
||||
MFEM_SHARED real_t smin[BLOCK_X];
|
||||
MFEM_SHARED real_t smax[BLOCK_X];
|
||||
MFEM_SHARED real_t sa0;
|
||||
MFEM_SHARED real_t sa1;
|
||||
|
||||
MFEM_FOREACH_THREAD(i, x, nd)
|
||||
{
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
const real_t x = data.xhat[i];
|
||||
const real_t w = data.what[i];
|
||||
ssum0[i] = 0.5*coeff[i]*w;
|
||||
ssum1[i] = 1.5*coeff[i]*w*x;
|
||||
}
|
||||
else
|
||||
{
|
||||
ssum0[i] = 0.0;
|
||||
ssum1[i] = 0.0;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(ii, x, 1)
|
||||
{
|
||||
sa0 = 0.0;
|
||||
sa1 = 0.0;
|
||||
for (int i = 0; i < nd; i++)
|
||||
{
|
||||
sa0 += ssum0[i];
|
||||
sa1 += ssum1[i];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(i, x, nd)
|
||||
{
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
const real_t x = data.xhat[i];
|
||||
sproj[i] = coeff[i] - sa0 - sa1*x;
|
||||
}
|
||||
else
|
||||
{
|
||||
sproj[i] = coeff[i];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
real_t lower_local = HUGE_VAL;
|
||||
real_t upper_local = -HUGE_VAL;
|
||||
MFEM_FOREACH_THREAD(j, x, data.ncp)
|
||||
{
|
||||
real_t lo = 0.0;
|
||||
real_t hi = 0.0;
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
const real_t xcp = data.cphat[j];
|
||||
lo = sa0 + sa1*xcp;
|
||||
hi = lo;
|
||||
}
|
||||
|
||||
for (int i = 0; i < nd; i++)
|
||||
{
|
||||
const real_t val = sproj[i];
|
||||
const real_t lv = data.lbound[j + i*data.ncp]*val;
|
||||
const real_t uv = data.ubound[j + i*data.ncp]*val;
|
||||
lo += lv < uv ? lv : uv;
|
||||
hi += lv > uv ? lv : uv;
|
||||
}
|
||||
lower_local = lower_local < lo ? lower_local : lo;
|
||||
upper_local = upper_local > hi ? upper_local : hi;
|
||||
}
|
||||
|
||||
smin[tid] = lower_local;
|
||||
smax[tid] = upper_local;
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(ii, x, 1)
|
||||
{
|
||||
real_t lower_ec = smin[0];
|
||||
real_t upper_ec = smax[0];
|
||||
const int nthreads = MFEM_THREAD_SIZE(x);
|
||||
const int nactive = data.ncp < nthreads ? data.ncp : nthreads;
|
||||
for (int t = 1; t < nactive; t++)
|
||||
{
|
||||
lower_ec = lower_ec < smin[t] ? lower_ec : smin[t];
|
||||
upper_ec = upper_ec > smax[t] ? upper_ec : smax[t];
|
||||
}
|
||||
L(e, c) = lower_ec;
|
||||
U(e, c) = upper_ec;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
template<int T_NB = 0, int T_NCP = 0, bool T_PROJ = true>
|
||||
inline void GetElementBoundsKernel2D(const PLBoundDeviceData &data,
|
||||
const int fes_vdim,
|
||||
const int ne,
|
||||
const Vector &e_vec,
|
||||
Vector &lower,
|
||||
Vector &upper,
|
||||
const int comp0,
|
||||
const int ncomp)
|
||||
{
|
||||
constexpr int DEFAULT_MAX_NB = 8;
|
||||
constexpr int DEFAULT_MAX_CP = 3*DEFAULT_MAX_NB;
|
||||
constexpr int MAX_NB = T_NB ? T_NB : DEFAULT_MAX_NB;
|
||||
constexpr int MAX_CP = T_NCP ? T_NCP : DEFAULT_MAX_CP;
|
||||
constexpr int MAX_THREADS = MAX_CP*MAX_CP;
|
||||
|
||||
const int nb = data.nb;
|
||||
const int ncp = data.ncp;
|
||||
const int nd = nb*nb;
|
||||
MFEM_VERIFY(nb <= MAX_NB,
|
||||
"Device 2D element bounds kernel exceeds its compile-time "
|
||||
"1D degree bound.");
|
||||
MFEM_VERIFY(ncp <= MAX_CP,
|
||||
"Device 2D element bounds kernel exceeds its compile-time "
|
||||
"control-point bound.");
|
||||
MFEM_VERIFY(ncp*ncp <= MAX_THREADS,
|
||||
"Device 2D element bounds kernel exceeds its compile-time "
|
||||
"thread-block bound.");
|
||||
|
||||
const auto E = Reshape(e_vec.Read(), nd, fes_vdim, ne);
|
||||
auto L = Reshape(lower.Write(), ne, ncomp);
|
||||
auto U = Reshape(upper.Write(), ne, ncomp);
|
||||
|
||||
mfem::forall_2D<MAX_THREADS>(ne*ncomp, ncp, ncp,
|
||||
[=] MFEM_HOST_DEVICE (int ec)
|
||||
{
|
||||
const int e = ec % ne;
|
||||
const int c = ec / ne;
|
||||
const int vc = comp0 + c;
|
||||
const real_t *coeff = &E(0, vc, e);
|
||||
const int tx = MFEM_THREAD_ID(x);
|
||||
const int ty = MFEM_THREAD_ID(y);
|
||||
|
||||
MFEM_SHARED real_t sproj[MAX_NB*MAX_NB];
|
||||
MFEM_SHARED real_t srow_min[MAX_NB*MAX_CP];
|
||||
MFEM_SHARED real_t srow_max[MAX_NB*MAX_CP];
|
||||
MFEM_SHARED real_t srow_a0[MAX_NB];
|
||||
MFEM_SHARED real_t srow_a1[MAX_NB];
|
||||
MFEM_SHARED real_t sa0[MAX_CP];
|
||||
MFEM_SHARED real_t sa1[MAX_CP];
|
||||
MFEM_SHARED real_t smin[MAX_THREADS];
|
||||
MFEM_SHARED real_t smax[MAX_THREADS];
|
||||
|
||||
// Stage 1a: for each nodal row, form the per-node contributions to the
|
||||
// row-wise linear fit used by the first 1D bounding solve.
|
||||
MFEM_FOREACH_THREAD(jrow, y, nb)
|
||||
{
|
||||
const real_t *row_coeff = coeff + jrow*nb;
|
||||
const int row_ncp_off = jrow*MAX_CP;
|
||||
MFEM_FOREACH_THREAD(i, x, nb)
|
||||
{
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
const real_t x = data.xhat[i];
|
||||
const real_t w = data.what[i];
|
||||
srow_min[row_ncp_off + i] = 0.5*row_coeff[i]*w;
|
||||
srow_max[row_ncp_off + i] = 1.5*row_coeff[i]*w*x;
|
||||
}
|
||||
else
|
||||
{
|
||||
srow_min[row_ncp_off + i] = 0.0;
|
||||
srow_max[row_ncp_off + i] = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Stage 1b: reduce the row-wise projection coefficients a0/a1.
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(jrow, y, nb)
|
||||
{
|
||||
const int row_ncp_off = jrow*MAX_CP;
|
||||
real_t a0 = 0.0;
|
||||
real_t a1 = 0.0;
|
||||
MFEM_FOREACH_THREAD(ii, x, 1)
|
||||
{
|
||||
for (int i = 0; i < nb; i++)
|
||||
{
|
||||
a0 += srow_min[row_ncp_off + i];
|
||||
a1 += srow_max[row_ncp_off + i];
|
||||
}
|
||||
srow_a0[jrow] = a0;
|
||||
srow_a1[jrow] = a1;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
// Stage 1c: subtract the row-wise linear fit once and cache the
|
||||
// projected row coefficients for reuse across all x-control points.
|
||||
MFEM_FOREACH_THREAD(jrow, y, nb)
|
||||
{
|
||||
const real_t *row_coeff = coeff + jrow*nb;
|
||||
MFEM_FOREACH_THREAD(i, x, nb)
|
||||
{
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
const real_t x = data.xhat[i];
|
||||
sproj[jrow*MAX_NB + i] = row_coeff[i]
|
||||
- srow_a0[jrow] - srow_a1[jrow]*x;
|
||||
}
|
||||
else
|
||||
{
|
||||
sproj[jrow*MAX_NB + i] = row_coeff[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Stage 1d: solve the first 1D bounding problem along each nodal row and
|
||||
// store bounds at every x-direction control point.
|
||||
MFEM_FOREACH_THREAD(icp, x, ncp)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(jrow, y, nb)
|
||||
{
|
||||
const int row_cp_off = jrow*ncp;
|
||||
real_t lo = 0.0;
|
||||
real_t hi = 0.0;
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
const real_t xcp = data.cphat[icp];
|
||||
lo = srow_a0[jrow] + srow_a1[jrow]*xcp;
|
||||
hi = lo;
|
||||
}
|
||||
for (int i = 0; i < nb; i++)
|
||||
{
|
||||
const real_t val = sproj[jrow*MAX_NB + i];
|
||||
const real_t lv = data.lbound[icp + i*data.ncp]*val;
|
||||
const real_t uv = data.ubound[icp + i*data.ncp]*val;
|
||||
lo += lv < uv ? lv : uv;
|
||||
hi += lv > uv ? lv : uv;
|
||||
}
|
||||
srow_min[row_cp_off + icp] = lo;
|
||||
srow_max[row_cp_off + icp] = hi;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Stage 2a: from the row bounds, form the per-row contributions to the
|
||||
// second 1D projection solve in the y-direction.
|
||||
MFEM_FOREACH_THREAD(icp, x, ncp)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(jrow, y, nb)
|
||||
{
|
||||
const int row_cp_off = jrow*ncp;
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
const real_t x = data.xhat[jrow];
|
||||
const real_t w = data.what[jrow];
|
||||
const real_t t = 0.5*(srow_min[row_cp_off + icp] +
|
||||
srow_max[row_cp_off + icp]);
|
||||
smin[row_cp_off + icp] = 0.5*t*w;
|
||||
smax[row_cp_off + icp] = 1.5*t*w*x;
|
||||
}
|
||||
else
|
||||
{
|
||||
smin[row_cp_off + icp] = 0.0;
|
||||
smax[row_cp_off + icp] = 0.0;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Stage 2b: reduce the y-direction projection coefficients for each
|
||||
// x-control-point column.
|
||||
MFEM_FOREACH_THREAD(jj, y, 1)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(icp, x, ncp)
|
||||
{
|
||||
real_t a0 = 0.0;
|
||||
real_t a1 = 0.0;
|
||||
for (int jrow = 0; jrow < nb; jrow++)
|
||||
{
|
||||
a0 += smin[jrow*ncp + icp];
|
||||
a1 += smax[jrow*ncp + icp];
|
||||
}
|
||||
sa0[icp] = a0;
|
||||
sa1[icp] = a1;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Stage 2c: subtract the y-direction linear fit from the intermediate
|
||||
// row bounds so the final tensor-product bound uses the perturbation.
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(icp, x, ncp)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(jrow, y, nb)
|
||||
{
|
||||
const int row_cp_off = jrow*ncp;
|
||||
const real_t x = data.xhat[jrow];
|
||||
const real_t t = sa0[icp] + sa1[icp]*x;
|
||||
srow_min[row_cp_off + icp] -= t;
|
||||
srow_max[row_cp_off + icp] -= t;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// Stage 3: each thread now owns one 2D control point (icp, kcp) and
|
||||
// accumulates its final lower/upper bound from the row-bound data.
|
||||
MFEM_FOREACH_THREAD(icp, x, ncp)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(kcp, y, ncp)
|
||||
{
|
||||
real_t lo = 0.0;
|
||||
real_t hi = 0.0;
|
||||
if constexpr (T_PROJ)
|
||||
{
|
||||
const real_t xcp = data.cphat[kcp];
|
||||
lo = sa0[icp] + sa1[icp]*xcp;
|
||||
hi = lo;
|
||||
}
|
||||
for (int jrow = 0; jrow < nb; jrow++)
|
||||
{
|
||||
const real_t w0 = srow_min[jrow*ncp + icp];
|
||||
const real_t w1 = srow_max[jrow*ncp + icp];
|
||||
const real_t lb = data.lbound[kcp + jrow*data.ncp];
|
||||
const real_t ub = data.ubound[kcp + jrow*data.ncp];
|
||||
const real_t v0 = lb*w0;
|
||||
const real_t v1 = ub*w0;
|
||||
const real_t v2 = lb*w1;
|
||||
const real_t v3 = ub*w1;
|
||||
real_t vlo = v0 < v1 ? v0 : v1;
|
||||
real_t vhi = v0 > v1 ? v0 : v1;
|
||||
vlo = vlo < v2 ? vlo : v2;
|
||||
vlo = vlo < v3 ? vlo : v3;
|
||||
vhi = vhi > v2 ? vhi : v2;
|
||||
vhi = vhi > v3 ? vhi : v3;
|
||||
lo += vlo;
|
||||
hi += vhi;
|
||||
}
|
||||
const int slot = kcp*ncp + icp;
|
||||
smin[slot] = lo;
|
||||
smax[slot] = hi;
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
const int lane = ty*ncp + tx;
|
||||
const int nactive = ncp*ncp;
|
||||
const int nthreads = MFEM_THREAD_SIZE(x)*MFEM_THREAD_SIZE(y);
|
||||
|
||||
// Reduce all 2D control-point bounds to one lower/upper pair per
|
||||
// (element, component).
|
||||
if (nthreads == 1)
|
||||
{
|
||||
if (tx == 0 && ty == 0)
|
||||
{
|
||||
real_t lower_ec = smin[0];
|
||||
real_t upper_ec = smax[0];
|
||||
for (int t = 1; t < nactive; t++)
|
||||
{
|
||||
lower_ec = lower_ec < smin[t] ? lower_ec : smin[t];
|
||||
upper_ec = upper_ec > smax[t] ? upper_ec : smax[t];
|
||||
}
|
||||
L(e, c) = lower_ec;
|
||||
U(e, c) = upper_ec;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int stride = (nactive + 1)/2; stride > 0;
|
||||
stride = (stride + 1)/2)
|
||||
{
|
||||
if (lane < stride && lane + stride < nactive)
|
||||
{
|
||||
smin[lane] = smin[lane] < smin[lane + stride] ?
|
||||
smin[lane] : smin[lane + stride];
|
||||
smax[lane] = smax[lane] > smax[lane + stride] ?
|
||||
smax[lane] : smax[lane + stride];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
if (stride == 1) { break; }
|
||||
}
|
||||
|
||||
if (lane == 0)
|
||||
{
|
||||
L(e, c) = smin[0];
|
||||
U(e, c) = smax[0];
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace internal
|
||||
|
||||
inline void PLBound::GetElementBoundsKernel(const int rdim, const int fes_vdim,
|
||||
const Vector &e_vec,
|
||||
Vector &lower, Vector &upper,
|
||||
const int vdim) const
|
||||
{
|
||||
MFEM_VERIFY(b_type != BasisType::Positive,
|
||||
"Bernstein device bounds are not implemented.");
|
||||
if (rdim == 3)
|
||||
{
|
||||
MFEM_ABORT("Device element bounds kernel currently only supports 1D/2D.");
|
||||
}
|
||||
MFEM_VERIFY(rdim == 1 || rdim == 2, "Invalid element dimension.");
|
||||
MFEM_VERIFY(vdim >= -1 && vdim <= fes_vdim, "Invalid vector component.");
|
||||
const int nd = static_cast<int>(std::pow(nb, rdim));
|
||||
const int ne = e_vec.Size()/(nd*fes_vdim);
|
||||
const int ncomp = (vdim > 0) ? 1 : fes_vdim;
|
||||
|
||||
lower.SetSize(ne*ncomp, e_vec);
|
||||
upper.SetSize(ne*ncomp, e_vec);
|
||||
lower.UseDevice(true);
|
||||
upper.UseDevice(true);
|
||||
|
||||
if (!proj)
|
||||
{
|
||||
MFEM_ABORT("Device element bounds kernel currently requires projection "
|
||||
"enabled.");
|
||||
}
|
||||
|
||||
const real_t *dxhat = xhat.Read();
|
||||
const real_t *dwhat = what.Read();
|
||||
const real_t *dcphat = cphat.Read();
|
||||
const real_t *dlbound = lbound.Read();
|
||||
const real_t *dubound = ubound.Read();
|
||||
|
||||
internal::PLBoundDeviceData data
|
||||
{
|
||||
nb,
|
||||
ncp,
|
||||
dxhat,
|
||||
dwhat,
|
||||
dcphat,
|
||||
dlbound,
|
||||
dubound
|
||||
};
|
||||
|
||||
const int comp0 = (vdim > 0) ? (vdim - 1) : 0;
|
||||
|
||||
if (rdim == 1)
|
||||
{
|
||||
switch (nb)
|
||||
{
|
||||
case 2: return internal::GetElementBoundsKernel1D<2, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
case 3: return internal::GetElementBoundsKernel1D<3, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
case 4: return internal::GetElementBoundsKernel1D<4, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
case 5: return internal::GetElementBoundsKernel1D<5, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
case 6: return internal::GetElementBoundsKernel1D<6, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
case 7: return internal::GetElementBoundsKernel1D<7, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
case 8: return internal::GetElementBoundsKernel1D<8, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
case 9: return internal::GetElementBoundsKernel1D<9, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
case 10: return internal::GetElementBoundsKernel1D<10, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
default: return internal::GetElementBoundsKernel1D<0, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
}
|
||||
}
|
||||
#define MFEM_PLBOUND_2D_DISPATCH(NB, NCP) \
|
||||
return internal::GetElementBoundsKernel2D<NB, NCP, true>(data, fes_vdim, ne, \
|
||||
e_vec, lower, upper, \
|
||||
comp0, ncomp)
|
||||
switch (nb)
|
||||
{
|
||||
case 2:
|
||||
switch (ncp)
|
||||
{
|
||||
case 4: MFEM_PLBOUND_2D_DISPATCH(2, 4);
|
||||
case 6: MFEM_PLBOUND_2D_DISPATCH(2, 6);
|
||||
case 8: MFEM_PLBOUND_2D_DISPATCH(2, 8);
|
||||
}
|
||||
break;
|
||||
case 3:
|
||||
switch (ncp)
|
||||
{
|
||||
case 6: MFEM_PLBOUND_2D_DISPATCH(3, 6);
|
||||
case 9: MFEM_PLBOUND_2D_DISPATCH(3, 9);
|
||||
case 12: MFEM_PLBOUND_2D_DISPATCH(3, 12);
|
||||
}
|
||||
break;
|
||||
case 4:
|
||||
switch (ncp)
|
||||
{
|
||||
case 8: MFEM_PLBOUND_2D_DISPATCH(4, 8);
|
||||
case 12: MFEM_PLBOUND_2D_DISPATCH(4, 12);
|
||||
case 16: MFEM_PLBOUND_2D_DISPATCH(4, 16);
|
||||
}
|
||||
break;
|
||||
case 5:
|
||||
switch (ncp)
|
||||
{
|
||||
case 10: MFEM_PLBOUND_2D_DISPATCH(5, 10);
|
||||
case 15: MFEM_PLBOUND_2D_DISPATCH(5, 15);
|
||||
case 20: MFEM_PLBOUND_2D_DISPATCH(5, 20);
|
||||
}
|
||||
break;
|
||||
case 6:
|
||||
switch (ncp)
|
||||
{
|
||||
case 12: MFEM_PLBOUND_2D_DISPATCH(6, 12);
|
||||
case 18: MFEM_PLBOUND_2D_DISPATCH(6, 18);
|
||||
case 24: MFEM_PLBOUND_2D_DISPATCH(6, 24);
|
||||
}
|
||||
break;
|
||||
case 7:
|
||||
switch (ncp)
|
||||
{
|
||||
case 14: MFEM_PLBOUND_2D_DISPATCH(7, 14);
|
||||
case 21: MFEM_PLBOUND_2D_DISPATCH(7, 21);
|
||||
case 28: MFEM_PLBOUND_2D_DISPATCH(7, 28);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
switch (ncp)
|
||||
{
|
||||
case 16: MFEM_PLBOUND_2D_DISPATCH(8, 16);
|
||||
case 24: MFEM_PLBOUND_2D_DISPATCH(8, 24);
|
||||
case 32: MFEM_PLBOUND_2D_DISPATCH(8, 32);
|
||||
}
|
||||
break;
|
||||
}
|
||||
#undef MFEM_PLBOUND_2D_DISPATCH
|
||||
return internal::GetElementBoundsKernel2D<0, 0, true>(data, fes_vdim, ne,
|
||||
e_vec, lower, upper,
|
||||
comp0, ncomp);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_BOUNDS
|
||||
|
||||
@@ -54,8 +54,6 @@ void Coefficient::Project(QuadratureFunction &qf)
|
||||
QuadratureSpaceBase &qspace = *qf.GetSpace();
|
||||
const int ne = qspace.GetNE();
|
||||
Vector values;
|
||||
// GetValues makes a reference, but we need it to be valid on Host
|
||||
qf.HostWrite();
|
||||
for (int iel = 0; iel < ne; ++iel)
|
||||
{
|
||||
qf.GetValues(iel, values);
|
||||
@@ -329,8 +327,6 @@ void VectorCoefficient::Project(QuadratureFunction &qf)
|
||||
const int ne = qspace.GetNE();
|
||||
DenseMatrix values;
|
||||
Vector col;
|
||||
// GetValues makes a reference, but we need it to be valid on Host
|
||||
qf.HostWrite();
|
||||
for (int iel = 0; iel < ne; ++iel)
|
||||
{
|
||||
qf.GetValues(iel, values);
|
||||
@@ -699,8 +695,6 @@ void MatrixCoefficient::Project(QuadratureFunction &qf, bool transpose)
|
||||
QuadratureSpaceBase &qspace = *qf.GetSpace();
|
||||
const int ne = qspace.GetNE();
|
||||
DenseMatrix values, matrix;
|
||||
// GetValues makes a reference, but we need it to be valid on Host
|
||||
qf.HostWrite();
|
||||
for (int iel = 0; iel < ne; ++iel)
|
||||
{
|
||||
qf.GetValues(iel, values);
|
||||
|
||||
+1
-5
@@ -1055,8 +1055,7 @@ public:
|
||||
|
||||
typedef VectorCoefficient DiagonalMatrixCoefficient;
|
||||
|
||||
/** Base class for matrix-valued coefficients that optionally depend on time
|
||||
and space. */
|
||||
/// Base class for Matrix Coefficients that optionally depend on time and space.
|
||||
class MatrixCoefficient
|
||||
{
|
||||
protected:
|
||||
@@ -1103,9 +1102,6 @@ public:
|
||||
/// the quadrature points. The matrix will be transposed or not according to
|
||||
/// the boolean argument @a transpose.
|
||||
///
|
||||
/// The stored entries use the same row/column convention as `Eval()`,
|
||||
/// unless `transpose == true`, in which case `K^T` is stored instead.
|
||||
///
|
||||
/// The @a vdim of the QuadratureFunction should be equal to the height times
|
||||
/// the width of the matrix.
|
||||
virtual void Project(QuadratureFunction &qf, bool transpose=false);
|
||||
|
||||
+164
-1049
File diff suppressed because it is too large
Load Diff
@@ -166,75 +166,6 @@ public:
|
||||
return sqrt(err_r * err_r + err_i * err_i);
|
||||
}
|
||||
|
||||
/// @brief Returns Max|u_ex - u_h| error for complex-valued H1 or L2 elements
|
||||
///
|
||||
/// Compute the $L_\infty$ error across the entire domain.
|
||||
///
|
||||
/// @param[in] exsolr Coefficient object reproducing the real part of the
|
||||
/// anticipated values of the scalar field, Re(u_ex).
|
||||
/// @param[in] exsoli Coefficient object reproducing the imaginary part of
|
||||
/// the anticipated values of the scalar field, Im(u_ex).
|
||||
/// @param[in] irs Optional pointer to an array of custom integration
|
||||
/// rules e.g. higher order than the default rules. If
|
||||
/// present the array will be indexed by
|
||||
/// Geometry::Type.
|
||||
///
|
||||
/// @note Uses ComputeLpError internally. See the ComputeLpError
|
||||
/// documentation for generalizations of this error computation.
|
||||
///
|
||||
/// @note If an array of integration rules is provided through @a irs, be
|
||||
/// sure to include valid rules for each element type that may occur
|
||||
/// in the list of elements.
|
||||
///
|
||||
virtual real_t ComputeMaxError(Coefficient &exsolr,
|
||||
Coefficient &exsoli,
|
||||
const IntegrationRule *irs[] = NULL) const
|
||||
{
|
||||
return ComputeLpError(infinity(), exsolr, exsoli, NULL, irs);
|
||||
}
|
||||
|
||||
/// @brief Returns ||u_ex - u_h||_Lp for complex-valued H1 or L2 elements
|
||||
///
|
||||
/// Computes:
|
||||
/// $$(\sum_{elems} \int_{elem} w \, |u_{ex} - u_h|^p)^{1/p}$$
|
||||
/// Where:
|
||||
/// $$|u_{ex} - u_h| = \sqrt{Re(u_{ex} - u_h)^2 + Im(u_{ex} - u_h)^2}$$
|
||||
///
|
||||
/// @param[in] p Real value indicating the exponent of the $L^p$ norm.
|
||||
/// To avoid domain errors p should have a positive value,
|
||||
/// either finite or infinite.
|
||||
/// @param[in] exsolr Coefficient object reproducing the real part of the
|
||||
/// anticipated values of the scalar field, Re(u_ex).
|
||||
/// @param[in] exsoli Coefficient object reproducing the imaginary part of
|
||||
/// the anticipated values of the scalar field, Im(u_ex).
|
||||
/// @param[in] weight Optional pointer to a Coefficient object reproducing
|
||||
/// a weighting function, w.
|
||||
/// @param[in] irs Optional pointer to an array of custom integration
|
||||
/// rules e.g. higher order than the default rules. If
|
||||
/// present the array will be indexed by Geometry::Type.
|
||||
/// @param[in] elems Optional pointer to a marker array, with a length
|
||||
/// equal to the number of local elements, indicating
|
||||
/// which elements to integrate over. Only those elements
|
||||
/// corresponding to non-zero entries in @a elems will
|
||||
/// contribute to the computed L2 error.
|
||||
///
|
||||
/// @note If an array of integration rules is provided through @a irs, be
|
||||
/// sure to include valid rules for each element type that may occur
|
||||
/// in the list of elements.
|
||||
///
|
||||
/// @note Quadratures with negative weights (as in some simplex integration
|
||||
/// rules in MFEM) can produce negative integrals even with
|
||||
/// non-negative integrands. To avoid returning negative errors this
|
||||
/// function uses the absolute values of the element-wise integrals.
|
||||
/// This may lead to results which are not entirely consistent with
|
||||
/// such integration rules.
|
||||
virtual real_t ComputeLpError(const real_t p,
|
||||
Coefficient &exsolr,
|
||||
Coefficient &exsoli,
|
||||
Coefficient *weight = NULL,
|
||||
const IntegrationRule *irs[] = NULL,
|
||||
const Array<int> *elems = NULL) const;
|
||||
|
||||
/// Save the ComplexGridFunction to an output stream.
|
||||
virtual void Save(std::ostream &out) const;
|
||||
|
||||
@@ -392,9 +323,6 @@ private:
|
||||
bool RealInteg();
|
||||
bool ImagInteg();
|
||||
|
||||
void BuildComplexOperator(OperatorHandle &A_r, OperatorHandle &A_i,
|
||||
OperatorHandle &A) const;
|
||||
|
||||
public:
|
||||
SesquilinearForm(FiniteElementSpace *fes,
|
||||
ComplexOperator::Convention
|
||||
@@ -508,186 +436,6 @@ public:
|
||||
virtual ~SesquilinearForm();
|
||||
};
|
||||
|
||||
/** Class for a mixed sesquilinear form
|
||||
|
||||
A mixed sesquilinear form is a generalization of a mixed bilinear form to
|
||||
complex-valued fields. Mixed sesquilinear forms are linear in the second
|
||||
argument but the first argument involves a complex conjugate in the sense
|
||||
that:
|
||||
|
||||
a(alpha u, beta v) = conj(alpha) beta a(u, v)
|
||||
|
||||
The @a convention argument in the class's constructor is documented in the
|
||||
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
|
||||
|
||||
When supplying integrators to the MixedSesquilinearForm either the real or
|
||||
imaginary integrator can be NULL. This indicates that the corresponding
|
||||
portion of the complex-valued material coefficient is equal to zero.
|
||||
*/
|
||||
class MixedSesquilinearForm
|
||||
{
|
||||
private:
|
||||
ComplexOperator::Convention conv;
|
||||
|
||||
MixedBilinearForm * mblfr;
|
||||
MixedBilinearForm * mblfi;
|
||||
|
||||
/* These methods check if the real/imag parts of the sesqulinear form are not
|
||||
empty */
|
||||
bool RealInteg();
|
||||
bool ImagInteg();
|
||||
|
||||
public:
|
||||
MixedSesquilinearForm(
|
||||
FiniteElementSpace * trial_fes,
|
||||
FiniteElementSpace * test_fes,
|
||||
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
|
||||
|
||||
/** @brief Create a MixedSesquilinearForm on the given trial and test
|
||||
FiniteElementSpaces, using the same integrators as the
|
||||
MixedBilinearForms @a bfr and @a bfi.
|
||||
|
||||
The FiniteElementSpace pointers are not owned by the newly constructed
|
||||
object.
|
||||
|
||||
The integrators are copied as pointers and they are not owned by the
|
||||
newly constructed MixedSesquilinearForm. */
|
||||
MixedSesquilinearForm(
|
||||
FiniteElementSpace * trial_fes,
|
||||
FiniteElementSpace * test_fes,
|
||||
MixedBilinearForm * bfr,
|
||||
MixedBilinearForm * bfi,
|
||||
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
|
||||
|
||||
ComplexOperator::Convention GetConvention() const { return conv; }
|
||||
void SetConvention(const ComplexOperator::Convention & convention) { conv = convention; }
|
||||
|
||||
/// Set the desired assembly level.
|
||||
/** Valid choices are:
|
||||
|
||||
- AssemblyLevel::LEGACY (default)
|
||||
- AssemblyLevel::FULL
|
||||
- AssemblyLevel::PARTIAL
|
||||
- AssemblyLevel::ELEMENT
|
||||
- AssemblyLevel::NONE
|
||||
|
||||
This method must be called before assembly. */
|
||||
void SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
{
|
||||
mblfr->SetAssemblyLevel(assembly_level);
|
||||
mblfi->SetAssemblyLevel(assembly_level);
|
||||
}
|
||||
|
||||
MixedBilinearForm & real() { return *mblfr; }
|
||||
MixedBilinearForm & imag() { return *mblfi; }
|
||||
const MixedBilinearForm & real() const { return *mblfr; }
|
||||
const MixedBilinearForm & imag() const { return *mblfi; }
|
||||
|
||||
/// Adds new Domain Integrator.
|
||||
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds new Domain Integrator, restricted to specific attributes.
|
||||
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & elem_marker);
|
||||
|
||||
/// Adds new Boundary Integrator.
|
||||
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/** @brief Adds new boundary Integrator, restricted to specific boundary
|
||||
attributes.
|
||||
|
||||
Assumes ownership of @a bfi.
|
||||
|
||||
The mfem::array @a bdr_marker is stored internally as a pointer to the given
|
||||
mfem::Array<int> object. */
|
||||
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & bdr_marker);
|
||||
|
||||
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
|
||||
void AddInteriorFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
|
||||
attributes.
|
||||
|
||||
Assumes ownership of @a bfi.
|
||||
|
||||
The mfem::array @a bdr_marker is stored internally as a pointer to the given
|
||||
mfem::Array<int> object. */
|
||||
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & bdr_marker);
|
||||
|
||||
/** @brief Add a trace face integrator. Assumes ownership of @a bfi.
|
||||
|
||||
This type of integrator assembles terms over all faces of the mesh using
|
||||
the face FE from the trial space and the two adjacent volume FEs from
|
||||
the test space. */
|
||||
void AddTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> &bdr_marker);
|
||||
|
||||
/// Assemble the local matrix
|
||||
void Assemble(int skip_zeros = 1);
|
||||
|
||||
/// Finalizes the matrix initialization.
|
||||
void Finalize(int skip_zeros = 1);
|
||||
|
||||
/// Updates the internal mixed forms with the new finite element space.
|
||||
virtual void Update();
|
||||
|
||||
/** @brief Return a ComplexSparseMatrix wrapping the local (L-dof) real
|
||||
and imaginary matrices of the form.
|
||||
|
||||
The returned wrapper has to be deleted by the caller, but it does not
|
||||
own the wrapped real and imaginary matrices, which remain owned by
|
||||
this form. */
|
||||
ComplexSparseMatrix *AssembleComplexSparseMatrix();
|
||||
|
||||
/// Return the trial FE space associated with the MixedSesquilinearForm.
|
||||
FiniteElementSpace *TrialFESpace() { return mblfr->TrialFESpace(); }
|
||||
|
||||
/// Read-only access to the associated trial FiniteElementSpace.
|
||||
const FiniteElementSpace *TrialFESpace() const { return mblfr->TrialFESpace(); }
|
||||
|
||||
/// Return the test FE space associated with the MixedSesquilinearForm.
|
||||
FiniteElementSpace *TestFESpace() { return mblfr->TestFESpace(); }
|
||||
|
||||
/// Read-only access to the associated test FiniteElementSpace.
|
||||
const FiniteElementSpace *TestFESpace() const { return mblfr->TestFESpace(); }
|
||||
|
||||
|
||||
void FormRectangularLinearSystem(const Array<int> & ess_trial_tdof_list,
|
||||
const Array<int> & ess_test_tdof_list,
|
||||
Vector & x,
|
||||
Vector & b,
|
||||
OperatorHandle & A,
|
||||
Vector & X,
|
||||
Vector & B);
|
||||
|
||||
void FormRectangularSystemMatrix(const Array<int> & ess_trial_tdof_list,
|
||||
const Array<int> & ess_test_tdof_list,
|
||||
OperatorHandle & A);
|
||||
|
||||
virtual ~MixedSesquilinearForm();
|
||||
};
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
/// Class for parallel complex-valued grid function - real + imaginary part
|
||||
@@ -989,12 +737,6 @@ private:
|
||||
bool RealInteg();
|
||||
bool ImagInteg();
|
||||
|
||||
void SetImaginaryEssentialDiagonalToZero(
|
||||
const Array<int> &ess_tdof_list, OperatorHandle &A);
|
||||
|
||||
void BuildComplexOperator(OperatorHandle &A_r, OperatorHandle &A_i,
|
||||
OperatorHandle &A) const;
|
||||
|
||||
public:
|
||||
ParSesquilinearForm(ParFiniteElementSpace *pf,
|
||||
ComplexOperator::Convention
|
||||
@@ -1110,169 +852,6 @@ public:
|
||||
virtual ~ParSesquilinearForm();
|
||||
};
|
||||
|
||||
/** Class for a parallel mixed sesquilinear form
|
||||
|
||||
A mixed sesquilinear form is a generalization of a mixed bilinear form to
|
||||
complex-valued fields. Mixed sesquilinear forms are linear in the second
|
||||
argument but the first argument involves a complex conjugate in the sense
|
||||
that:
|
||||
|
||||
a(alpha u, beta v) = conj(alpha) beta a(u, v)
|
||||
|
||||
The @a convention argument in the class's constructor is documented in the
|
||||
mfem::ComplexOperator class found in linalg/complex_operator.hpp.
|
||||
|
||||
When supplying integrators to the ParMixedSesquilinearForm either the real
|
||||
or imaginary integrator can be NULL. This indicates that the corresponding
|
||||
portion of the complex-valued material coefficient is equal to zero.
|
||||
*/
|
||||
class ParMixedSesquilinearForm
|
||||
{
|
||||
private:
|
||||
ComplexOperator::Convention conv;
|
||||
|
||||
ParMixedBilinearForm * pmblfr;
|
||||
ParMixedBilinearForm * pmblfi;
|
||||
|
||||
/* These methods check if the real/imag parts of the sesqulinear form are
|
||||
not empty */
|
||||
bool RealInteg();
|
||||
bool ImagInteg();
|
||||
|
||||
public:
|
||||
ParMixedSesquilinearForm(
|
||||
ParFiniteElementSpace * trial_fes,
|
||||
ParFiniteElementSpace * test_fes,
|
||||
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
|
||||
|
||||
/** @brief Create a ParMixedSesquilinearForm on the given trial and test
|
||||
ParFiniteElementSpaces, using the same integrators as the
|
||||
ParMixedBilinearForms @a pbfr and @a pbfi.
|
||||
|
||||
The ParFiniteElementSpace pointers are not owned by the newly
|
||||
constructed object.
|
||||
|
||||
The integrators are copied as pointers and they are not owned by the
|
||||
newly constructed ParMixedSesquilinearForm. */
|
||||
ParMixedSesquilinearForm(
|
||||
ParFiniteElementSpace * trial_fes,
|
||||
ParFiniteElementSpace * test_fes,
|
||||
ParMixedBilinearForm * pbfr,
|
||||
ParMixedBilinearForm * pbfi,
|
||||
ComplexOperator::Convention convention = ComplexOperator::HERMITIAN);
|
||||
|
||||
ComplexOperator::Convention GetConvention() const { return conv; }
|
||||
void SetConvention(const ComplexOperator::Convention & convention) { conv = convention; }
|
||||
|
||||
/// Set the desired assembly level.
|
||||
/** Valid choices are:
|
||||
|
||||
- AssemblyLevel::LEGACY (default)
|
||||
- AssemblyLevel::FULL
|
||||
- AssemblyLevel::PARTIAL
|
||||
- AssemblyLevel::ELEMENT
|
||||
- AssemblyLevel::NONE
|
||||
|
||||
This method must be called before assembly. */
|
||||
void SetAssemblyLevel(AssemblyLevel assembly_level)
|
||||
{
|
||||
pmblfr->SetAssemblyLevel(assembly_level);
|
||||
pmblfi->SetAssemblyLevel(assembly_level);
|
||||
}
|
||||
|
||||
ParMixedBilinearForm & real() { return *pmblfr; }
|
||||
ParMixedBilinearForm & imag() { return *pmblfi; }
|
||||
const ParMixedBilinearForm & real() const { return *pmblfr; }
|
||||
const ParMixedBilinearForm & imag() const { return *pmblfi; }
|
||||
|
||||
/// Adds new Domain Integrator.
|
||||
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds new Domain Integrator, restricted to specific attributes.
|
||||
void AddDomainIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & elem_marker);
|
||||
|
||||
/// Adds new Boundary Integrator.
|
||||
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/** @brief Adds new boundary Integrator, restricted to specific boundary
|
||||
attributes.
|
||||
|
||||
Assumes ownership of @a bfi.
|
||||
|
||||
The mfem::array @a bdr_marker is stored internally as a pointer to the given
|
||||
mfem::Array<int> object. */
|
||||
void AddBoundaryIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & bdr_marker);
|
||||
|
||||
/// Adds new interior Face Integrator. Assumes ownership of @a bfi.
|
||||
void AddInteriorFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds new boundary Face Integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/** @brief Adds new boundary Face Integrator, restricted to specific boundary
|
||||
attributes.
|
||||
|
||||
Assumes ownership of @a bfi.
|
||||
|
||||
The mfem::array @a bdr_marker is stored internally as a pointer to the given
|
||||
mfem::Array<int> object. */
|
||||
void AddBdrFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> & bdr_marker);
|
||||
|
||||
/** @brief Add a trace face integrator. Assumes ownership of @a bfi.
|
||||
|
||||
This type of integrator assembles terms over all faces of the mesh using
|
||||
the face FE from the trial space and the two adjacent volume FEs from
|
||||
the test space. */
|
||||
void AddTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag);
|
||||
|
||||
/// Adds a boundary trace face integrator. Assumes ownership of @a bfi.
|
||||
void AddBdrTraceFaceIntegrator(BilinearFormIntegrator * bfi_real,
|
||||
BilinearFormIntegrator * bfi_imag,
|
||||
Array<int> &bdr_marker);
|
||||
|
||||
/// Assemble the local matrix
|
||||
void Assemble(int skip_zeros = 1);
|
||||
|
||||
/// Finalizes the matrix initialization.
|
||||
void Finalize(int skip_zeros = 1);
|
||||
|
||||
/// Updates the internal mixed forms with the new finite element space.
|
||||
virtual void Update();
|
||||
|
||||
/// Returns the matrix assembled on the true dofs, i.e. P^t A P.
|
||||
/** The returned matrix has to be deleted by the caller. */
|
||||
ComplexHypreParMatrix * ParallelAssemble();
|
||||
|
||||
void FormRectangularLinearSystem(const Array<int> & ess_trial_tdof_list,
|
||||
const Array<int> & ess_test_tdof_list,
|
||||
Vector & x,
|
||||
Vector & b,
|
||||
OperatorHandle & A,
|
||||
Vector & X,
|
||||
Vector & B);
|
||||
|
||||
void FormRectangularSystemMatrix(const Array<int> & ess_trial_tdof_list,
|
||||
const Array<int> & ess_test_tdof_list,
|
||||
OperatorHandle & A);
|
||||
|
||||
virtual ~ParMixedSesquilinearForm();
|
||||
};
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
}
|
||||
|
||||
@@ -114,10 +114,6 @@ void ConduitDataCollection::Save()
|
||||
n_mesh["fields"][name]);
|
||||
}
|
||||
|
||||
// TODO: in parallel, we need to call ParFiniteElementSpace::ApplyDofSigns
|
||||
// for all ParGridFunction objects before and after saving, see
|
||||
// ParGridFunction::Save.
|
||||
|
||||
// save mesh data
|
||||
SaveMeshAndFields(myid,
|
||||
n_mesh,
|
||||
|
||||
+5
-23
@@ -38,24 +38,9 @@ int DataCollection::create_directory(const std::string &dir_name,
|
||||
// create directories recursively
|
||||
const char path_delim = '/';
|
||||
std::string::size_type pos = 0;
|
||||
int err_flag = 0;
|
||||
int err_flag;
|
||||
#ifdef MFEM_USE_MPI
|
||||
const ParMesh *pmesh = dynamic_cast<const ParMesh*>(mesh);
|
||||
// In addition to the global root, let the lowest rank on each shared-memory
|
||||
// node create the directory too, so that node-local (non-shared) filesystems
|
||||
// get it on every node rather than only where the global root lives. On a
|
||||
// shared filesystem the extra mkdir() hits EEXIST and is tolerated below.
|
||||
bool node_root = true;
|
||||
if (pmesh)
|
||||
{
|
||||
MPI_Comm node_comm;
|
||||
MPI_Comm_split_type(pmesh->GetComm(), MPI_COMM_TYPE_SHARED, myid,
|
||||
MPI_INFO_NULL, &node_comm);
|
||||
int node_rank;
|
||||
MPI_Comm_rank(node_comm, &node_rank);
|
||||
node_root = (node_rank == 0);
|
||||
MPI_Comm_free(&node_comm);
|
||||
}
|
||||
#endif
|
||||
|
||||
do
|
||||
@@ -67,7 +52,7 @@ int DataCollection::create_directory(const std::string &dir_name,
|
||||
err_flag = mkdir(subdir.c_str(), 0777);
|
||||
err_flag = (err_flag && (errno != EEXIST)) ? 1 : 0;
|
||||
#else
|
||||
if (node_root || pmesh == NULL)
|
||||
if (myid == 0 || pmesh == NULL)
|
||||
{
|
||||
err_flag = mkdir(subdir.c_str(), 0777);
|
||||
err_flag = (err_flag && (errno != EEXIST)) ? 1 : 0;
|
||||
@@ -79,8 +64,7 @@ int DataCollection::create_directory(const std::string &dir_name,
|
||||
#ifdef MFEM_USE_MPI
|
||||
if (pmesh)
|
||||
{
|
||||
MPI_Allreduce(MPI_IN_PLACE, &err_flag, 1, MPI_INT, MPI_MAX,
|
||||
pmesh->GetComm());
|
||||
MPI_Bcast(&err_flag, 1, MPI_INT, 0, pmesh->GetComm());
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -825,7 +809,7 @@ ParaViewDataCollectionBase::ParaViewDataCollectionBase(
|
||||
|
||||
void ParaViewDataCollectionBase::SetLevelsOfDetail(int levels_of_detail_)
|
||||
{
|
||||
levels_of_detail = std::max(levels_of_detail_, 1);
|
||||
levels_of_detail = levels_of_detail_;
|
||||
}
|
||||
|
||||
void ParaViewDataCollectionBase::SetHighOrderOutput(bool high_order_output_)
|
||||
@@ -1197,14 +1181,12 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
|
||||
DenseMatrix vval, pmat;
|
||||
std::vector<char> buf;
|
||||
int vec_dim = it->second->VectorDim();
|
||||
int map_type = it->second->FESpace()->GetTypicalFE()->GetMapType();
|
||||
os << "<DataArray type=\"" << GetDataTypeString()
|
||||
<< "\" Name=\"" << it->first
|
||||
<< "\" NumberOfComponents=\"" << vec_dim << "\" "
|
||||
<< VTKComponentLabels(vec_dim) << " "
|
||||
<< "format=\"" << GetDataFormatString() << "\" >" << '\n';
|
||||
if (vec_dim == 1 && (map_type == FiniteElement::VALUE ||
|
||||
map_type == FiniteElement::INTEGRAL))
|
||||
if (vec_dim == 1)
|
||||
{
|
||||
for (int i = 0; i < mesh->GetNE(); i++)
|
||||
{
|
||||
|
||||
@@ -51,52 +51,4 @@ DifferentiableOperator::DifferentiableOperator(
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void FDJacobian::Mult(const Vector &v, Vector &y) const
|
||||
{
|
||||
// See [1] for choice of eps.
|
||||
//
|
||||
// [1] Woodward, C.S., Gardner, D.J. and Evans, K.J., 2015. On the use of
|
||||
// finite difference matrix-vector products in Newton-Krylov solvers for
|
||||
// implicit climate dynamics with spectral elements. Procedia Computer
|
||||
// Science, 51, pp.2036-2045.
|
||||
real_t eps;
|
||||
if (fixed_eps > 0.0)
|
||||
{
|
||||
eps = fixed_eps;
|
||||
}
|
||||
else
|
||||
{
|
||||
const real_t vnorm_local = v.Norml2();
|
||||
real_t vnorm;
|
||||
MPI_Allreduce(&vnorm_local, &vnorm, 1, MPITypeMap<real_t>::mpi_type, MPI_SUM,
|
||||
MPI_COMM_WORLD);
|
||||
eps = lambda * (lambda + xnorm / vnorm);
|
||||
}
|
||||
|
||||
// x + eps * v
|
||||
{
|
||||
const auto d_v = v.Read();
|
||||
const auto d_x = x.Read();
|
||||
auto d_xpev = xpev.Write();
|
||||
mfem::forall(x.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_xpev[i] = d_x[i] + eps * d_v[i];
|
||||
});
|
||||
}
|
||||
|
||||
// y = f(x + eps * v)
|
||||
op.Mult(xpev, y);
|
||||
|
||||
// y = (f(x + eps * v) - f(x)) / eps
|
||||
{
|
||||
const auto d_f = f.Read();
|
||||
auto d_y = y.ReadWrite();
|
||||
mfem::forall(f.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_y[i] = (d_y[i] - d_f[i]) / eps;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
+22
-23
@@ -697,18 +697,17 @@ void DifferentiableOperator::AddIntegrator(
|
||||
|
||||
// The explicit captures are necessary to avoid dependency on
|
||||
// the specific instance of this class (this pointer).
|
||||
restriction_callback = [element_dof_ordering,
|
||||
solutions_ = this->solutions,
|
||||
parameters_ = this->parameters]
|
||||
(std::vector<Vector> &sol,
|
||||
const std::vector<Vector> &par,
|
||||
std::vector<Vector> &f)
|
||||
restriction_callback =
|
||||
[=, solutions = this->solutions, parameters = this->parameters]
|
||||
(std::vector<Vector> &sol,
|
||||
const std::vector<Vector> &par,
|
||||
std::vector<Vector> &f)
|
||||
{
|
||||
restriction<entity_t>(solutions_, sol, f,
|
||||
restriction<entity_t>(solutions, sol, f,
|
||||
element_dof_ordering);
|
||||
restriction<entity_t>(parameters_, par, f,
|
||||
restriction<entity_t>(parameters, par, f,
|
||||
element_dof_ordering,
|
||||
solutions_.size());
|
||||
solutions.size());
|
||||
};
|
||||
|
||||
prolongation_transpose = get_prolongation_transpose(
|
||||
@@ -836,19 +835,19 @@ void DifferentiableOperator::AddIntegrator(
|
||||
|
||||
// capture by ref:
|
||||
&restriction_cb = this->restriction_callback,
|
||||
&fields_e_ = this->fields_e,
|
||||
&residual_e_ = this->residual_e,
|
||||
&output_restriction_transpose_ = this->output_restriction_transpose
|
||||
&fields_e = this->fields_e,
|
||||
&residual_e = this->residual_e,
|
||||
&output_restriction_transpose = this->output_restriction_transpose
|
||||
]
|
||||
(std::vector<Vector> &sol, const std::vector<Vector> &par, Vector &res)
|
||||
mutable // mutable: needed to modify 'shmem_cache'
|
||||
{
|
||||
restriction_cb(sol, par, fields_e_);
|
||||
restriction_cb(sol, par, fields_e);
|
||||
|
||||
residual_e_ = 0.0;
|
||||
auto ye = Reshape(residual_e_.ReadWrite(), test_vdim, num_test_dof, num_entities);
|
||||
residual_e = 0.0;
|
||||
auto ye = Reshape(residual_e.ReadWrite(), test_vdim, num_test_dof, num_entities);
|
||||
|
||||
auto wrapped_fields_e = wrap_fields(fields_e_,
|
||||
auto wrapped_fields_e = wrap_fields(fields_e,
|
||||
action_shmem_info.field_sizes,
|
||||
num_entities);
|
||||
|
||||
@@ -879,7 +878,7 @@ void DifferentiableOperator::AddIntegrator(
|
||||
y, fhat, output_fop, output_dtq_shmem[0],
|
||||
scratch_shmem, dimension, use_sum_factorization);
|
||||
}, num_entities, thread_blocks, action_shmem_info.total_size, shmem_cache.ReadWrite());
|
||||
output_restriction_transpose_(residual_e_, res);
|
||||
output_restriction_transpose(residual_e, res);
|
||||
});
|
||||
|
||||
// Without this compile-time check, some valid instantiations of this method
|
||||
@@ -1194,7 +1193,7 @@ void DifferentiableOperator::AddIntegrator(
|
||||
|
||||
// capture by ref:
|
||||
&qpdc_mem = derivative_qp_caches_ref,
|
||||
&fields_ = fields_ref
|
||||
&fields = fields_ref
|
||||
](std::vector<Vector> &f_e, SparseMatrix *&A) mutable
|
||||
{
|
||||
auto wrapped_fields_e = wrap_fields(f_e, shmem_info.field_sizes,
|
||||
@@ -1242,14 +1241,14 @@ void DifferentiableOperator::AddIntegrator(
|
||||
{
|
||||
if (input_is_dependent[s])
|
||||
{
|
||||
trial_field = &fields_[input_to_field[s]];
|
||||
trial_field = &fields[input_to_field[s]];
|
||||
}
|
||||
}
|
||||
|
||||
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&trial_field->data);
|
||||
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&fields_[output_to_field[0]].data);
|
||||
(&fields[output_to_field[0]].data);
|
||||
|
||||
A = new SparseMatrix(test_fes->GetVSize(), trial_fes->GetVSize());
|
||||
|
||||
@@ -1335,7 +1334,7 @@ void DifferentiableOperator::AddIntegrator(
|
||||
input_to_field,
|
||||
output_to_field,
|
||||
&spmatcb = assemble_derivative_sparsematrix_callbacks_ref,
|
||||
&fields_ = fields_ref
|
||||
&fields = fields_ref
|
||||
](std::vector<Vector> &f_e, HypreParMatrix *&A) mutable
|
||||
{
|
||||
SparseMatrix *spmat = nullptr;
|
||||
@@ -1367,14 +1366,14 @@ void DifferentiableOperator::AddIntegrator(
|
||||
{
|
||||
if (input_is_dependent[s])
|
||||
{
|
||||
trial_field = &fields_[input_to_field[s]];
|
||||
trial_field = &fields[input_to_field[s]];
|
||||
}
|
||||
}
|
||||
|
||||
auto trial_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&trial_field->data);
|
||||
auto test_fes = *std::get_if<const ParFiniteElementSpace *>
|
||||
(&fields_[output_to_field[0]].data);
|
||||
(&fields[output_to_field[0]].data);
|
||||
|
||||
if (same_test_and_trial)
|
||||
{
|
||||
|
||||
+768
-742
File diff suppressed because it is too large
Load Diff
+52
-9
@@ -597,7 +597,7 @@ struct ThreadBlocks
|
||||
int z = 1;
|
||||
};
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
template <typename func_t>
|
||||
__global__ void forall_kernel_shmem(func_t f, int n)
|
||||
{
|
||||
@@ -617,11 +617,10 @@ void forall(func_t f,
|
||||
int num_shmem = 0,
|
||||
real_t *shmem = nullptr)
|
||||
{
|
||||
internal::RequireKernelCompilation();
|
||||
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP_LANG)
|
||||
if (Device::Allows(Backend::CUDA_MASK | Backend::HIP_MASK))
|
||||
if (Device::Allows(Backend::CUDA_MASK) ||
|
||||
Device::Allows(Backend::HIP_MASK))
|
||||
{
|
||||
#if defined(MFEM_USE_CUDA_OR_HIP)
|
||||
// int gridsize = (N + Z - 1) / Z;
|
||||
int num_bytes = num_shmem * sizeof(decltype(shmem));
|
||||
dim3 block_size(blocks.x, blocks.y, blocks.z);
|
||||
@@ -632,10 +631,9 @@ void forall(func_t f,
|
||||
MFEM_GPU_CHECK(hipGetLastError());
|
||||
#endif
|
||||
MFEM_DEVICE_SYNC;
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
if (Device::Allows(Backend::CPU_MASK))
|
||||
}
|
||||
else if (Device::Allows(Backend::CPU_MASK))
|
||||
{
|
||||
MFEM_ASSERT(!((bool)num_shmem != (bool)shmem),
|
||||
"Backend::CPU needs a pre-allocated shared memory block");
|
||||
@@ -673,7 +671,52 @@ public:
|
||||
MPI_COMM_WORLD);
|
||||
}
|
||||
|
||||
void Mult(const Vector &v, Vector &y) const override;
|
||||
void Mult(const Vector &v, Vector &y) const override
|
||||
{
|
||||
// See [1] for choice of eps.
|
||||
//
|
||||
// [1] Woodward, C.S., Gardner, D.J. and Evans, K.J., 2015. On the use of
|
||||
// finite difference matrix-vector products in Newton-Krylov solvers for
|
||||
// implicit climate dynamics with spectral elements. Procedia Computer
|
||||
// Science, 51, pp.2036-2045.
|
||||
real_t eps;
|
||||
if (fixed_eps > 0.0)
|
||||
{
|
||||
eps = fixed_eps;
|
||||
}
|
||||
else
|
||||
{
|
||||
const real_t vnorm_local = v.Norml2();
|
||||
real_t vnorm;
|
||||
MPI_Allreduce(&vnorm_local, &vnorm, 1, MPITypeMap<real_t>::mpi_type, MPI_SUM,
|
||||
MPI_COMM_WORLD);
|
||||
eps = lambda * (lambda + xnorm / vnorm);
|
||||
}
|
||||
|
||||
// x + eps * v
|
||||
{
|
||||
const auto d_v = v.Read();
|
||||
const auto d_x = x.Read();
|
||||
auto d_xpev = xpev.Write();
|
||||
mfem::forall(x.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_xpev[i] = d_x[i] + eps * d_v[i];
|
||||
});
|
||||
}
|
||||
|
||||
// y = f(x + eps * v)
|
||||
op.Mult(xpev, y);
|
||||
|
||||
// y = (f(x + eps * v) - f(x)) / eps
|
||||
{
|
||||
const auto d_f = f.Read();
|
||||
auto d_y = y.ReadWrite();
|
||||
mfem::forall(f.Size(), [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
d_y[i] = (d_y[i] - d_f[i]) / eps;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
virtual MemoryClass GetMemoryClass() const override
|
||||
{
|
||||
|
||||
+5
-6
@@ -1316,14 +1316,13 @@ void VectorFiniteElement::Project_RT(
|
||||
}
|
||||
}
|
||||
|
||||
void VectorFiniteElement::ProjectCurl2D_RT(
|
||||
void VectorFiniteElement::ProjectGrad_RT(
|
||||
const real_t *nk, const Array<int> &d2n, const FiniteElement &fe,
|
||||
ElementTransformation &Trans, DenseMatrix &grad) const
|
||||
{
|
||||
// 2D "ProjectCurl_RT"
|
||||
if (dim != 2)
|
||||
{
|
||||
mfem_error("VectorFiniteElement::ProjectCurl2D_RT works only in 2D!");
|
||||
mfem_error("VectorFiniteElement::ProjectGrad_RT works only in 2D!");
|
||||
}
|
||||
|
||||
DenseMatrix dshape(fe.GetDof(), fe.GetDim());
|
||||
@@ -1334,8 +1333,8 @@ void VectorFiniteElement::ProjectCurl2D_RT(
|
||||
for (int k = 0; k < dof; k++)
|
||||
{
|
||||
fe.CalcDShape(Nodes.IntPoint(k), dshape);
|
||||
tk[0] = -nk[d2n[k]*dim+1];
|
||||
tk[1] = nk[d2n[k]*dim];
|
||||
tk[0] = nk[d2n[k]*dim+1];
|
||||
tk[1] = -nk[d2n[k]*dim];
|
||||
dshape.Mult(tk, grad_k);
|
||||
for (int j = 0; j < grad_k.Size(); j++)
|
||||
{
|
||||
@@ -1382,7 +1381,7 @@ void VectorFiniteElement::ProjectCurl_ND(
|
||||
}
|
||||
}
|
||||
|
||||
void VectorFiniteElement::ProjectCurl3D_RT(
|
||||
void VectorFiniteElement::ProjectCurl_RT(
|
||||
const real_t *nk, const Array<int> &d2n, const FiniteElement &fe,
|
||||
ElementTransformation &Trans, DenseMatrix &curl) const
|
||||
{
|
||||
|
||||
+8
-52
@@ -167,15 +167,7 @@ public:
|
||||
/** @brief Full multidimensional representation which does not use tensor
|
||||
product structure. The ordering of the degrees of freedom is the
|
||||
same as TENSOR, but the sizes of B and G are the same as FULL.*/
|
||||
LEXICOGRAPHIC_FULL,
|
||||
|
||||
/** @brief Ragged tensor product representation using 1D matrices/tensors
|
||||
with dimensions using 1D number of quadrature points and ragged tensor degrees of
|
||||
freedom. */
|
||||
/** Used only for partial assembly of the H1 positive basis. The
|
||||
size of B is d1d x qnpt x dim. Since different Gauss-Jacobi quadrature rules
|
||||
are employed in each dimension, we need to store dim arrays. */
|
||||
RAGGED_TENSOR
|
||||
LEXICOGRAPHIC_FULL
|
||||
};
|
||||
|
||||
/// Describes the contents of the #B, #Bt, #G, and #Gt arrays, see #Mode.
|
||||
@@ -236,39 +228,6 @@ public:
|
||||
const Array<DofToQuad*> &dof2quad_array,
|
||||
const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode);
|
||||
|
||||
virtual ~DofToQuad() = default;
|
||||
};
|
||||
|
||||
/** @brief Structure representing the matrices/tensors needed to evaluate (in
|
||||
reference space) the values, gradients, divergences, or curls of a positive
|
||||
FiniteElement on simplices at the quadrature points of Stroud conical quadrature. */
|
||||
class RaggedDofToQuad : public DofToQuad
|
||||
{
|
||||
public:
|
||||
/** @brief Special basis function structures for positive (Bernstein) basis with
|
||||
partial assembly. The storage layout of Ba1 is ndof x nqpt for scalar elements.
|
||||
The storage layout of Ba2 is ndof x ndof x nqpt. In particular, we have
|
||||
Ba2(iqpt, a1, a2) = B^{p-a1}_{a2}(x_{iqpt}). */
|
||||
Array<real_t> Ba1, Ba2, Ba3;
|
||||
Array<real_t> Ba1t, Ba2t, Ba3t;
|
||||
|
||||
/** @brief Special structures for gradients of positive basis with partial assembly.
|
||||
The gradient arrays exploit properties of the Bernstein basis which allow grad(B^p_alpha)
|
||||
to be expressed as the sum of products of B^{p-1}_alpha and the barycentric coordinates.
|
||||
Thus, Ga1 and Ga2 simply contain the ragged tensor product components of B^{p-1}_alpha */
|
||||
Array<real_t> Ga1, Ga2, Ga3;
|
||||
Array<real_t> Ga1t, Ga2t, Ga3t;
|
||||
|
||||
/** @brief Mapping from the Bernstein multi-index (a_1, ..., a_d) to the lexicographic
|
||||
dof index. */
|
||||
Array<int> lex_map;
|
||||
|
||||
Array<int> forward_map2d_diff, forward_map3d_diff;
|
||||
Array<int> inverse_map2d_diff, inverse_map3d_diff;
|
||||
|
||||
Array<int> forward_map2d_mass, forward_map3d_mass;
|
||||
Array<int> inverse_map2d_mass, inverse_map3d_mass;
|
||||
};
|
||||
|
||||
/// Describes the function space on each element
|
||||
@@ -957,11 +916,10 @@ protected:
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &I) const;
|
||||
|
||||
// Input is a scalar representing the Z (out of plane) component, Output is
|
||||
// the X-Y (in-plane) RT curl
|
||||
void ProjectCurl2D_RT(const real_t *nk, const Array<int> &d2n,
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const;
|
||||
// rotated gradient in 2D
|
||||
void ProjectGrad_RT(const real_t *nk, const Array<int> &d2n,
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const;
|
||||
|
||||
// Compute the curl as a discrete operator from ND FE (fe) to ND FE (this).
|
||||
// The natural FE for the range is RT, so this is an approximation.
|
||||
@@ -969,9 +927,9 @@ protected:
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const;
|
||||
|
||||
void ProjectCurl3D_RT(const real_t *nk, const Array<int> &d2n,
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const;
|
||||
void ProjectCurl_RT(const real_t *nk, const Array<int> &d2n,
|
||||
const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const;
|
||||
|
||||
/** @brief Project a vector coefficient onto the ND basis functions
|
||||
@param tk Edge tangent vectors for this element type
|
||||
@@ -1447,8 +1405,6 @@ public:
|
||||
dof2quad_array_open);
|
||||
}
|
||||
|
||||
const Poly_1D::Basis &GetOpenBasis1D() const { return obasis1d; }
|
||||
|
||||
virtual ~VectorTensorFiniteElement();
|
||||
};
|
||||
|
||||
|
||||
@@ -307,12 +307,12 @@ public:
|
||||
|
||||
/** @brief virtual function which evaluates the values of all
|
||||
shape functions at a given point ip and stores
|
||||
them in the vector shape of dimension Dof (6) */
|
||||
them in the vector shape of dimension Dof (4) */
|
||||
void CalcShape(const IntegrationPoint &ip, Vector &shape) const override;
|
||||
|
||||
/** @brief virtual function which evaluates the values of all
|
||||
partial derivatives of all shape functions at a given
|
||||
point ip and stores them in the matrix dshape (Dof x Dim) (6 x 3)
|
||||
point ip and stores them in the matrix dshape (Dof x Dim) (4 x 3)
|
||||
so that each row contains the derivatives of one shape function */
|
||||
void CalcDShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &dshape) const override;
|
||||
@@ -336,12 +336,12 @@ public:
|
||||
|
||||
/** @brief virtual function which evaluates the values of all
|
||||
shape functions at a given point ip and stores
|
||||
them in the vector shape of dimension Dof (5) */
|
||||
them in the vector shape of dimension Dof (4) */
|
||||
void CalcShape(const IntegrationPoint &ip, Vector &shape) const override;
|
||||
|
||||
/** @brief virtual function which evaluates the values of all
|
||||
partial derivatives of all shape functions at a given
|
||||
point ip and stores them in the matrix dshape (Dof x Dim) (5 x 3)
|
||||
point ip and stores them in the matrix dshape (Dof x Dim) (4 x 3)
|
||||
so that each row contains the derivatives of one shape function */
|
||||
void CalcDShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &dshape) const override;
|
||||
|
||||
+57
-130
@@ -1757,45 +1757,22 @@ H1_BergotPyramidElement::H1_BergotPyramidElement(const int p, const int btype)
|
||||
real_t y = (ip.z < 1.0) ? (ip.y / (1.0 - ip.z)) : 0.0;
|
||||
real_t z = ip.z;
|
||||
|
||||
if (std::abs(z - 1.0) < apex_tol)
|
||||
{
|
||||
// Compute the limit of the basis functions as z->1 with x and y on the
|
||||
// line between the center of the base and the apex
|
||||
o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
if (i == 0 && j == 0)
|
||||
{
|
||||
T(o++, m) = ((k + 3.) * k + 2.) / 2.;
|
||||
}
|
||||
else
|
||||
{
|
||||
T(o++, m) = 0.;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData());
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData());
|
||||
|
||||
o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
{
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0),
|
||||
z, 1.0, shape_z);
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0),
|
||||
z, 1.0, shape_z);
|
||||
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
{
|
||||
T(o++, m) = shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij);
|
||||
}
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
{
|
||||
T(o++, m) = shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1816,44 +1793,25 @@ void H1_BergotPyramidElement::CalcShape(const IntegrationPoint &ip,
|
||||
Vector u(dof);
|
||||
#endif
|
||||
|
||||
const real_t x = (ip.z < 1.0) ? (ip.x / (1.0 - ip.z)) : 0.0;
|
||||
const real_t y = (ip.z < 1.0) ? (ip.y / (1.0 - ip.z)) : 0.0;
|
||||
const real_t z = ip.z;
|
||||
real_t x = (ip.z < 1.0) ? (ip.x / (1.0 - ip.z)) : 0.0;
|
||||
real_t y = (ip.z < 1.0) ? (ip.y / (1.0 - ip.z)) : 0.0;
|
||||
real_t z = ip.z;
|
||||
|
||||
if (std::abs(z - 1.0) < apex_tol)
|
||||
{
|
||||
// Compute the limit of the basis functions as z->1 with x and y on the
|
||||
// line between the center of the base and the apex
|
||||
u = 0.;
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
for (int k = 0; k <= p - maxij; k++, o++)
|
||||
if (i == 0 && j == 0)
|
||||
{
|
||||
u(o) = ((k + 3.) * k + 2.) / 2.;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData());
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData());
|
||||
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0), z, 1.0,
|
||||
shape_z);
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
u[o++] = shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij);
|
||||
}
|
||||
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0),
|
||||
z, 1.0, shape_z);
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
u[o++] = shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij);
|
||||
}
|
||||
}
|
||||
Ti.Mult(u, shape);
|
||||
}
|
||||
|
||||
@@ -1872,68 +1830,37 @@ void H1_BergotPyramidElement::CalcDShape(const IntegrationPoint &ip,
|
||||
Vector dshape_z(order+1);
|
||||
Vector dshape_z_dt(order+1);
|
||||
#endif
|
||||
const real_t x = (ip.z < 1.0) ? (ip.x / (1.0 - ip.z)) : 0.0;
|
||||
const real_t y = (ip.z < 1.0) ? (ip.y / (1.0 - ip.z)) : 0.0;
|
||||
const real_t z = ip.z;
|
||||
real_t x = (ip.z < 1.0) ? (ip.x / (1.0 - ip.z)) : 0.0;
|
||||
real_t y = (ip.z < 1.0) ? (ip.y / (1.0 - ip.z)) : 0.0;
|
||||
real_t z = ip.z;
|
||||
|
||||
if (std::abs(z - 1.0) < apex_tol)
|
||||
{
|
||||
// Compute the limit of the gradients of the basis functions as
|
||||
// z->1 with x and y on the line between the center of the base and the
|
||||
// apex
|
||||
du = 0.;
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData(), dshape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData(), dshape_y.GetData());
|
||||
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0), z, 1.0,
|
||||
shape_z, dshape_z, dshape_z_dt);
|
||||
|
||||
for (int k = 0; k <= p - maxij; k++, o++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
for (int k = 0; k <= p - maxij; k++, o++)
|
||||
{
|
||||
if (i == 0 && j == 0)
|
||||
{
|
||||
du(o,2) = (((k + 6.) * k + 11.) * k + 6.) * k / 6.;
|
||||
}
|
||||
else if (i == 1 && j == 0)
|
||||
{
|
||||
du(o,0) = ((((k + 10.) * k + 35.) * k + 50.) * k + 24.) / 24.;
|
||||
}
|
||||
else if (i == 0 && j == 1)
|
||||
{
|
||||
du(o,1) = ((((k + 10.) * k + 35.) * k + 50.) * k + 24.) / 24.;
|
||||
}
|
||||
}
|
||||
du(o,0) = dshape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
du(o,1) = shape_x(i) * dshape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
du(o,2) = shape_x(i) * shape_y(j) * dshape_z(k) *
|
||||
pow(1.0 - ip.z, maxij) +
|
||||
(ip.x * dshape_x(i) * shape_y(j) +
|
||||
ip.y * shape_x(i) * dshape_y(j)) *
|
||||
shape_z(k) * pow(1.0 - ip.z, maxij - 2) -
|
||||
maxij * shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData(), dshape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData(), dshape_y.GetData());
|
||||
}
|
||||
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0),
|
||||
z, 1.0,
|
||||
shape_z, dshape_z, dshape_z_dt);
|
||||
|
||||
for (int k = 0; k <= p - maxij; k++, o++)
|
||||
{
|
||||
du(o,0) = dshape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
du(o,1) = shape_x(i) * dshape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
du(o,2) = shape_x(i) * shape_y(j) * dshape_z(k) *
|
||||
pow(1.0 - ip.z, maxij) +
|
||||
(ip.x * dshape_x(i) * shape_y(j) +
|
||||
ip.y * shape_x(i) * dshape_y(j)) *
|
||||
shape_z(k) * pow(1.0 - ip.z, maxij - 2) -
|
||||
maxij * shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
(maxij > 0 ? pow(1.0 - ip.z, maxij - 1) : 0.0);
|
||||
}
|
||||
}
|
||||
}
|
||||
Ti.Mult(du, dshape);
|
||||
}
|
||||
|
||||
|
||||
@@ -208,8 +208,6 @@ private:
|
||||
#endif
|
||||
DenseMatrixInverse Ti;
|
||||
|
||||
static constexpr real_t apex_tol = 1e-8;
|
||||
|
||||
public:
|
||||
H1_BergotPyramidElement(const int p,
|
||||
const int btype = BasisType::GaussLobatto);
|
||||
|
||||
+57
-131
@@ -1106,16 +1106,9 @@ L2_BergotPyramidElement::L2_BergotPyramidElement(const int p, const int btype)
|
||||
{
|
||||
const real_t wik = op[i] + op[k] + op[p-i-k];
|
||||
const real_t w = wik * wjk * op[p-k];
|
||||
if (std::abs(w) < apex_tol)
|
||||
{
|
||||
Nodes.IntPoint(o++).Set3(0.,0.,1.);
|
||||
}
|
||||
else
|
||||
{
|
||||
Nodes.IntPoint(o++).Set3(op[i] * (op[j] + op[p-j-k]) / w,
|
||||
op[j] * (op[i] + op[p-i-k]) / w,
|
||||
op[k] * op[p-k] / w);
|
||||
}
|
||||
Nodes.IntPoint(o++).Set3(op[i] * (op[j] + op[p-j-k]) / w,
|
||||
op[j] * (op[j] + op[p-j-k]) / w,
|
||||
op[k] * op[p-k] / w);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1132,45 +1125,22 @@ L2_BergotPyramidElement::L2_BergotPyramidElement(const int p, const int btype)
|
||||
const real_t y = (ip.z < 1.0) ? (ip.y / (1.0 - ip.z)) : 0.0;
|
||||
const real_t z = ip.z;
|
||||
|
||||
if (std::abs(z - 1.0) < apex_tol)
|
||||
{
|
||||
// Compute the limit of the basis functions as z->1 with x and y on the
|
||||
// line between the center of the base and the apex
|
||||
o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
if (i == 0 && j == 0)
|
||||
{
|
||||
T(o++, m) = ((k + 3.) * k + 2.) / 2.;
|
||||
}
|
||||
else
|
||||
{
|
||||
T(o++, m) = 0.;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData());
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData());
|
||||
|
||||
o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
{
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0),
|
||||
z, 1.0, shape_z);
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0),
|
||||
z, 1.0, shape_z);
|
||||
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
{
|
||||
T(o++, m) = shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij);
|
||||
}
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
{
|
||||
T(o++, m) = shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1195,41 +1165,26 @@ void L2_BergotPyramidElement::CalcShape(const IntegrationPoint &ip,
|
||||
const real_t y = (ip.z < 1.0) ? (ip.y / (1.0 - ip.z)) : 0.0;
|
||||
const real_t z = ip.z;
|
||||
|
||||
if (std::abs(z - 1.0) < apex_tol)
|
||||
{
|
||||
// Compute the limit of the basis functions as z->1 with x and y on the
|
||||
// line between the center of the base and the apex
|
||||
u = 0.;
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
for (int k = 0; k <= p - maxij; k++, o++)
|
||||
if (i == 0 && j == 0)
|
||||
{
|
||||
u(o) = ((k + 3.) * k + 2.) / 2.;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData());
|
||||
poly1d.CalcLegendre(p, x, shape_x.GetData());
|
||||
poly1d.CalcLegendre(p, y, shape_y.GetData());
|
||||
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0),
|
||||
z, 1.0, shape_z);
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
{
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0), z, 1.0,
|
||||
shape_z);
|
||||
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
u[o++] = shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij);
|
||||
for (int k = 0; k <= p - maxij; k++)
|
||||
{
|
||||
u[o++] = shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ti.Mult(u, shape);
|
||||
}
|
||||
|
||||
@@ -1253,64 +1208,35 @@ void L2_BergotPyramidElement::CalcDShape(const IntegrationPoint &ip,
|
||||
const real_t y = (ip.z < 1.0) ? (ip.y / (1.0 - ip.z)) : 0.0;
|
||||
const real_t z = ip.z;
|
||||
|
||||
if (std::abs(z - 1.0) < apex_tol)
|
||||
{
|
||||
// Compute the limit of the gradients of the basis functions as
|
||||
// z->1 with x and y on the line between the center of the base and the
|
||||
// apex
|
||||
du = 0.;
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
for (int k = 0; k <= p - maxij; k++, o++)
|
||||
{
|
||||
if (i == 0 && j == 0)
|
||||
{
|
||||
du(o,2) = (((k + 6.) * k + 11.) * k + 6.) * k / 6.;
|
||||
}
|
||||
else if (i == 1 && j == 0)
|
||||
{
|
||||
du(o,0) = ((((k + 10.) * k + 35.) * k + 50.) * k + 24.) / 24.;
|
||||
}
|
||||
else if (i == 0 && j == 1)
|
||||
{
|
||||
du(o,1) = ((((k + 10.) * k + 35.) * k + 50.) * k + 24.) / 24.;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
Poly_1D::CalcLegendre(p, x, shape_x.GetData(), dshape_x.GetData());
|
||||
Poly_1D::CalcLegendre(p, y, shape_y.GetData(), dshape_y.GetData());
|
||||
Poly_1D::CalcLegendre(p, x, shape_x.GetData(), dshape_x.GetData());
|
||||
Poly_1D::CalcLegendre(p, y, shape_y.GetData(), dshape_y.GetData());
|
||||
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0),
|
||||
z, 1.0,
|
||||
shape_z, dshape_z, dshape_z_dt);
|
||||
int o = 0;
|
||||
for (int i = 0; i <= p; i++)
|
||||
{
|
||||
for (int j = 0; j <= p; j++)
|
||||
{
|
||||
int maxij = std::max(i, j);
|
||||
FuentesPyramid::CalcScaledJacobi(p-maxij, 2.0 * (maxij + 1.0), z, 1.0,
|
||||
shape_z, dshape_z, dshape_z_dt);
|
||||
|
||||
for (int k = 0; k <= p - maxij; k++, o++)
|
||||
{
|
||||
du(o,0) = dshape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
du(o,1) = shape_x(i) * dshape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
du(o,2) = shape_x(i) * shape_y(j) * dshape_z(k) *
|
||||
pow(1.0 - ip.z, maxij) +
|
||||
(ip.x * dshape_x(i) * shape_y(j) +
|
||||
ip.y * shape_x(i) * dshape_y(j)) *
|
||||
shape_z(k) * pow(1.0 - ip.z, maxij - 2) -
|
||||
maxij * shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
(maxij > 0 ? pow(1.0 - ip.z, maxij - 1) : 0.0);
|
||||
}
|
||||
for (int k = 0; k <= p - maxij; k++, o++)
|
||||
{
|
||||
du(o,0) = dshape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
du(o,1) = shape_x(i) * dshape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1);
|
||||
du(o,2) = shape_x(i) * shape_y(j) * dshape_z(k) *
|
||||
pow(1.0 - ip.z, maxij) +
|
||||
(ip.x * dshape_x(i) * shape_y(j) +
|
||||
ip.y * shape_x(i) * dshape_y(j)) *
|
||||
shape_z(k) * pow(1.0 - ip.z, maxij - 2) -
|
||||
((maxij > 0) ? (maxij * shape_x(i) * shape_y(j) * shape_z(k) *
|
||||
pow(1.0 - ip.z, maxij - 1)) : 0.0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ti.Mult(du, dshape);
|
||||
}
|
||||
|
||||
|
||||
@@ -225,8 +225,6 @@ private:
|
||||
#endif
|
||||
DenseMatrixInverse Ti;
|
||||
|
||||
static constexpr real_t apex_tol = 1e-8;
|
||||
|
||||
public:
|
||||
/// Construct the L2_PyramidElement of order @a p and BasisType @a btype
|
||||
L2_BergotPyramidElement(const int p,
|
||||
|
||||
+1
-38
@@ -1282,49 +1282,12 @@ ND_SegmentElement::ND_SegmentElement(const int p, const int ob_type)
|
||||
}
|
||||
}
|
||||
|
||||
void ND_SegmentElement::CalcShape(const IntegrationPoint &ip,
|
||||
Vector &shape) const
|
||||
{
|
||||
if (obasis1d.IsIntegratedType()) { obasis1d.ScaleIntegrated(false); }
|
||||
obasis1d.Eval(ip.x, shape);
|
||||
}
|
||||
|
||||
void ND_SegmentElement::CalcVShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &shape) const
|
||||
{
|
||||
Vector vshape(shape.Data(), dof);
|
||||
|
||||
CalcShape(ip, vshape);
|
||||
}
|
||||
|
||||
void ND_SegmentElement::ProjectIntegrated(VectorCoefficient &vc,
|
||||
ElementTransformation &Trans,
|
||||
Vector &dofs) const
|
||||
{
|
||||
MFEM_ASSERT(obasis1d.IsIntegratedType(), "Not integrated type");
|
||||
real_t vk[Geometry::MaxDim];
|
||||
Vector xk(vk, vc.GetVDim());
|
||||
|
||||
const real_t *cp = poly1d.ClosedPoints(dof, BasisType::GaussLobatto);
|
||||
const IntegrationRule &ir = IntRules.Get(Geometry::SEGMENT, dof);
|
||||
IntegrationPoint ip;
|
||||
|
||||
for (int i = 0; i < dof; i++)
|
||||
{
|
||||
const real_t h = cp[i+1] - cp[i];
|
||||
real_t val = 0.0;
|
||||
|
||||
for (int q = 0; q < ir.GetNPoints(); q++)
|
||||
{
|
||||
const IntegrationPoint &ip1d = ir.IntPoint(q);
|
||||
ip.x = cp[i] + h*ip1d.x;
|
||||
Trans.SetIntPoint(&ip);
|
||||
vc.Eval(xk, Trans, ip);
|
||||
val += ip1d.weight*Trans.Jacobian().InnerProduct(tk, vk);
|
||||
}
|
||||
|
||||
dofs(i) = val*h;
|
||||
}
|
||||
obasis1d.Eval(ip.x, vshape);
|
||||
}
|
||||
|
||||
const real_t ND_WedgeElement::tk[15] =
|
||||
|
||||
+3
-10
@@ -303,7 +303,8 @@ public:
|
||||
/** @brief Construct the ND_SegmentElement of order @a p and open
|
||||
BasisType @a ob_type */
|
||||
ND_SegmentElement(const int p, const int ob_type = BasisType::GaussLegendre);
|
||||
void CalcShape(const IntegrationPoint &ip, Vector &shape) const override;
|
||||
void CalcShape(const IntegrationPoint &ip, Vector &shape) const override
|
||||
{ obasis1d.Eval(ip.x, shape); }
|
||||
void CalcVShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &shape) const override;
|
||||
void CalcVShape(ElementTransformation &Trans,
|
||||
@@ -324,10 +325,7 @@ public:
|
||||
using FiniteElement::Project;
|
||||
void Project(VectorCoefficient &vc,
|
||||
ElementTransformation &Trans, Vector &dofs) const override
|
||||
{
|
||||
if (obasis1d.IsIntegratedType()) { ProjectIntegrated(vc, Trans, dofs); }
|
||||
else { Project_ND(tk, dof2tk, vc, Trans, dofs); }
|
||||
}
|
||||
{ Project_ND(tk, dof2tk, vc, Trans, dofs); }
|
||||
void ProjectMatrixCoefficient(MatrixCoefficient &mc,
|
||||
ElementTransformation &T,
|
||||
Vector &dofs) const override
|
||||
@@ -340,11 +338,6 @@ public:
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const override
|
||||
{ ProjectGrad_ND(tk, dof2tk, fe, Trans, grad); }
|
||||
|
||||
protected:
|
||||
void ProjectIntegrated(VectorCoefficient &vc,
|
||||
ElementTransformation &Trans,
|
||||
Vector &dofs) const;
|
||||
};
|
||||
|
||||
class ND_WedgeElement : public VectorFiniteElement
|
||||
|
||||
@@ -557,101 +557,6 @@ H1Pos_TriangleElement::H1Pos_TriangleElement(const int p)
|
||||
}
|
||||
}
|
||||
|
||||
const DofToQuad &H1Pos_TriangleElement::GetRaggedTensorDofToQuad(
|
||||
const FiniteElement &fe, const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode,
|
||||
Array<DofToQuad*> &dof2quad_array)
|
||||
{
|
||||
DofToQuad *d2q = nullptr;
|
||||
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
|
||||
}
|
||||
if (!d2q)
|
||||
{
|
||||
d2q = new RaggedDofToQuad;
|
||||
const int ndof = fe.GetOrder() + 1; // verify
|
||||
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
|
||||
d2q->FE = &fe;
|
||||
d2q->IntRule = &ir;
|
||||
d2q->mode = mode;
|
||||
d2q->ndof = ndof;
|
||||
d2q->nqpt = nqpt;
|
||||
|
||||
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
|
||||
rd2q->Ba1.SetSize(nqpt*ndof);
|
||||
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
|
||||
rd2q->Ba2.SetSize((int)nqpt*ndof*ndof);
|
||||
rd2q->Ba1t.SetSize(nqpt*ndof);
|
||||
rd2q->Ba2t.SetSize((int)nqpt*ndof*ndof);
|
||||
// stores first component of ragged tensor basis with order p-1, for gradients only
|
||||
rd2q->Ga1.SetSize(nqpt*(ndof -1));
|
||||
// stores second component of ragged tensor basis with order p-1
|
||||
rd2q->Ga2.SetSize(nqpt*(ndof-1)*(ndof -1));
|
||||
rd2q->Ga1t.SetSize(nqpt*(ndof -1));
|
||||
rd2q->Ga2t.SetSize(nqpt*(ndof-1)*(ndof -1));
|
||||
rd2q->lex_map.SetSize(ndof * ndof);
|
||||
Vector shape_a1(ndof), shape_a2(ndof * ndof);
|
||||
Vector shape_Ga1(ndof-1), shape_Ga2((ndof-1) * (ndof-1));
|
||||
for (int i = 0; i < nqpt; i++)
|
||||
{
|
||||
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
|
||||
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
|
||||
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
|
||||
// Gauss-Jacobi rule). Additionally, the Bernstein PA algorithms expect evaluation of the
|
||||
// component 1D bases at the Stroud nodes pulled back to the unit square, so perform the pullback
|
||||
// on the fly.
|
||||
const real_t x = ir.IntPoint(i).x;
|
||||
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
|
||||
|
||||
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
|
||||
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
|
||||
for (int j = 0; j < ndof; j++)
|
||||
{
|
||||
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
|
||||
if (j < ndof-1)
|
||||
{
|
||||
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
|
||||
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
|
||||
}
|
||||
|
||||
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
|
||||
for (int k = 0; k < ndof-j; k++)
|
||||
{
|
||||
rd2q->Ba2t[i + nqpt*(j + ndof*k)] = rd2q->Ba2[k + ndof*(j + ndof*i)] = shape_a2(
|
||||
k);
|
||||
if (j < ndof-1 && k < ndof-j-1)
|
||||
{
|
||||
rd2q->Ga2t[i + nqpt*(j + (ndof-1)*k)] = rd2q->Ga2[k + (ndof-1)*(j +
|
||||
(ndof-1)*i)] = shape_Ga2(k);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// stores the mapping from 2D Bernstein multi-index (i,j,p-i-j) to the
|
||||
// lexicographic DOF ordering
|
||||
for (int i = 0; i < ndof; i++)
|
||||
{
|
||||
for (int j = 0; j < ndof-i; j++)
|
||||
{
|
||||
int idx = ((2 * (ndof-1) + 3) - j) * j / 2 + i;
|
||||
rd2q->lex_map[j + ndof*i] = idx;
|
||||
}
|
||||
}
|
||||
dof2quad_array.Append(d2q);
|
||||
}
|
||||
}
|
||||
return *d2q;
|
||||
}
|
||||
|
||||
// static method
|
||||
void H1Pos_TriangleElement::CalcShape(
|
||||
const int p, const real_t l1, const real_t l2, real_t *shape)
|
||||
@@ -844,213 +749,6 @@ H1Pos_TetrahedronElement::H1Pos_TetrahedronElement(const int p)
|
||||
}
|
||||
}
|
||||
|
||||
const DofToQuad &H1Pos_TetrahedronElement::GetRaggedTensorDofToQuad(
|
||||
const FiniteElement &fe, const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode,
|
||||
Array<DofToQuad*> &dof2quad_array)
|
||||
{
|
||||
DofToQuad *d2q = nullptr;
|
||||
MFEM_VERIFY(mode == DofToQuad::RAGGED_TENSOR, "invalid mode requested");
|
||||
|
||||
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
|
||||
#pragma omp critical (DofToQuad)
|
||||
#endif
|
||||
{
|
||||
for (int i = 0; i < dof2quad_array.Size(); i++)
|
||||
{
|
||||
d2q = dof2quad_array[i];
|
||||
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
|
||||
}
|
||||
if (!d2q)
|
||||
{
|
||||
d2q = new RaggedDofToQuad;
|
||||
const int ndof = fe.GetOrder() + 1; // verify
|
||||
const int nqpt = (int)floor(pow(ir.GetNPoints(), 1.0/fe.GetDim()) + 0.5);
|
||||
const int basis_dim2d = ndof*(ndof+1) / 2;
|
||||
const int basis_dim3d = ndof*(ndof+1)*(ndof+2) / 6;
|
||||
const int basis_dim2d_diff = (ndof-1)*(ndof) / 2;
|
||||
const int basis_dim3d_diff = (ndof-1)*(ndof)*(ndof+1) / 6;
|
||||
d2q->FE = &fe;
|
||||
d2q->IntRule = &ir;
|
||||
d2q->mode = mode;
|
||||
d2q->ndof = ndof;
|
||||
d2q->nqpt = nqpt;
|
||||
|
||||
RaggedDofToQuad *rd2q = static_cast<RaggedDofToQuad*>(d2q);
|
||||
rd2q->Ba1.SetSize(nqpt * ndof);
|
||||
// second component of ragged tensor basis, technically dof*(dof-1)/2 entries
|
||||
rd2q->Ba2.SetSize(nqpt * basis_dim2d);
|
||||
// third component of ragged tensor basis, technically dof*(dof-1)/2 entries
|
||||
rd2q->Ba3.SetSize(nqpt * basis_dim3d);
|
||||
rd2q->Ba1t.SetSize(nqpt * ndof);
|
||||
rd2q->Ba2t.SetSize(nqpt * basis_dim2d);
|
||||
rd2q->Ba3t.SetSize(nqpt * basis_dim3d);
|
||||
// stores first component of ragged tensor basis with order p-1, for gradients only
|
||||
rd2q->Ga1.SetSize(nqpt * (ndof-1));
|
||||
// stores second component of ragged tensor basis with order p-1
|
||||
rd2q->Ga2.SetSize(nqpt * basis_dim2d_diff);
|
||||
// stores third component of ragged tensor basis with order p-1
|
||||
rd2q->Ga3.SetSize(nqpt * basis_dim3d_diff);
|
||||
rd2q->Ga1t.SetSize(nqpt * (ndof-1));
|
||||
rd2q->Ga2t.SetSize(nqpt * basis_dim2d_diff);
|
||||
rd2q->Ga3t.SetSize(nqpt * basis_dim3d_diff);
|
||||
rd2q->lex_map.SetSize(ndof * ndof * ndof);
|
||||
|
||||
rd2q->forward_map2d_diff.SetSize((ndof-1) * (ndof-1));
|
||||
rd2q->forward_map3d_diff.SetSize((ndof-1) * (ndof-1) * (ndof-1));
|
||||
rd2q->inverse_map2d_diff.SetSize(2 * basis_dim2d_diff);
|
||||
rd2q->inverse_map3d_diff.SetSize(3 * basis_dim3d_diff);
|
||||
|
||||
rd2q->forward_map2d_mass.SetSize(ndof * ndof);
|
||||
rd2q->forward_map3d_mass.SetSize(ndof * ndof * ndof);
|
||||
rd2q->inverse_map2d_mass.SetSize(2 * basis_dim2d);
|
||||
rd2q->inverse_map3d_mass.SetSize(2 * basis_dim3d);
|
||||
|
||||
// forward and inverse maps for multi-index to collpased 1d index for diffusion, can combine
|
||||
// these four loops, but need four idx's and clause for shorter diff loops
|
||||
int idx = 0;
|
||||
for (int i = 0; i < ndof-1; i++)
|
||||
{
|
||||
for (int j = 0; j < ndof-i-1; j++)
|
||||
{
|
||||
rd2q->forward_map2d_diff[j + (ndof-1)*i] = idx;
|
||||
rd2q->inverse_map2d_diff[2*idx] = i;
|
||||
rd2q->inverse_map2d_diff[1 + 2*idx] = j;
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
|
||||
idx = 0;
|
||||
for (int k = 0; k < ndof-1; k++)
|
||||
{
|
||||
for (int j = 0; j < ndof-k-1; j++)
|
||||
{
|
||||
for (int i = 0; i < ndof-k-j-1; i++)
|
||||
{
|
||||
rd2q->forward_map3d_diff[k + (ndof-1)*(j + (ndof-1)*i)] = idx;
|
||||
rd2q->inverse_map3d_diff[3*idx] = i;
|
||||
rd2q->inverse_map3d_diff[1 + 3*idx] = j;
|
||||
rd2q->inverse_map3d_diff[2 + 3*idx] = k;
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// forward and inverse maps for multi-index to collpased 1d index for mass
|
||||
idx = 0;
|
||||
for (int j = 0; j < ndof; j++)
|
||||
{
|
||||
for (int i = 0; i < ndof-j; i++)
|
||||
{
|
||||
rd2q->forward_map2d_mass[j + ndof*i] = idx;
|
||||
rd2q->inverse_map2d_mass[2*idx] = i;
|
||||
rd2q->inverse_map2d_mass[1 + 2*idx] = j;
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
|
||||
idx = 0;
|
||||
for (int k = 0; k < ndof; k++)
|
||||
{
|
||||
for (int j = 0; j < ndof-k; j++)
|
||||
{
|
||||
for (int i = 0; i < ndof-k-j; i++)
|
||||
{
|
||||
rd2q->forward_map3d_mass[k + ndof*(j + ndof*i)] = idx;
|
||||
rd2q->inverse_map3d_mass[2*idx] = i;
|
||||
rd2q->inverse_map3d_mass[1 + 2*idx] = j;
|
||||
// d2q->inverse_map3d_mass[2 + 3*idx] = k;
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vector shape_a1(ndof), shape_a2(ndof * ndof), shape_a3(ndof * ndof * ndof);
|
||||
Vector shape_Ga1(ndof-1), shape_Ga2(ndof-1), shape_Ga3(ndof-1);
|
||||
for (int i = 0; i < nqpt; i++)
|
||||
{
|
||||
// The first 'nqpt' points in the first dimension 'ir' have the same x-coordinates as those
|
||||
// of the 1D rule (ie. (2,0) Gauss-Jacobi rule). The first 'nqpt' points in the second dimension
|
||||
// 'ir' have the same y-coordinates as those of the 1D rule for second dimension (i.e. (1,0)
|
||||
// Gauss-Jacobi rule). The first 'nqpt' points in the third dimension have the same z-coordinates
|
||||
// as those of the 1D rule for the third dimension (i.e. Gauss-Legendre rule). Additionally,
|
||||
// the Bernstein PA algorithms expect evaluation of the component 1D bases at the Stroud nodes
|
||||
// pulled back to the unit cube, so perform the pullback on the fly.
|
||||
const real_t x = ir.IntPoint(i).x;
|
||||
const real_t y = ir.IntPoint(nqpt*i).y / (1.0 - ir.IntPoint(nqpt*i).x);
|
||||
const real_t z = ir.IntPoint(nqpt*nqpt*i).z / (1.0 - ir.IntPoint(
|
||||
nqpt*nqpt*i).x - ir.IntPoint(nqpt*nqpt*i).y);
|
||||
Poly_1D::CalcBernstein(ndof-1, x, shape_a1);
|
||||
Poly_1D::CalcBernstein(ndof-2, x, shape_Ga1);
|
||||
for (int j = 0; j < ndof; j++)
|
||||
{
|
||||
rd2q->Ba1t[i+nqpt*j] = rd2q->Ba1[j+ndof*i] = shape_a1(j);
|
||||
if (j < ndof-1)
|
||||
{
|
||||
rd2q->Ga1t[i+nqpt*j] = rd2q->Ga1[j+(ndof-1)*i] = shape_Ga1(j);
|
||||
Poly_1D::CalcBernstein(ndof-2-j, y, shape_Ga2);
|
||||
}
|
||||
|
||||
Poly_1D::CalcBernstein(ndof-1-j, y, shape_a2);
|
||||
for (int k = 0; k < ndof-j; k++)
|
||||
{
|
||||
const int a_2d_mass = rd2q->forward_map2d_mass[k + ndof*j];
|
||||
rd2q->Ba2t[i + nqpt*a_2d_mass] = rd2q->Ba2[a_2d_mass + basis_dim2d*i] =
|
||||
shape_a2(
|
||||
k);
|
||||
if (j < ndof-1 && k < ndof-j-1)
|
||||
{
|
||||
const int a_2d_diff = rd2q->forward_map2d_diff[k + (ndof-1)*j];
|
||||
rd2q->Ga2t[i + nqpt*a_2d_diff] = rd2q->Ga2[a_2d_diff + basis_dim2d_diff*i] =
|
||||
shape_Ga2(k);
|
||||
Poly_1D::CalcBernstein(ndof-2-j-k, z, shape_Ga3);
|
||||
}
|
||||
|
||||
Poly_1D::CalcBernstein(ndof-1-j-k, z, shape_a3);
|
||||
for (int m = 0; m < ndof-j-k; m++)
|
||||
{
|
||||
const int a_3d_mass = rd2q->forward_map3d_mass[m + ndof*(k + ndof*j)];
|
||||
rd2q->Ba3t[i + nqpt*a_3d_mass] = rd2q->Ba3[a_3d_mass + basis_dim3d*i] =
|
||||
shape_a3(
|
||||
m);
|
||||
if (j < ndof-1 && k < ndof-j-1 && m < ndof-j-k-1)
|
||||
{
|
||||
// // collapsed 1D access
|
||||
// d2q->Ga3[i + nqpt*(m + d2q->offset3d[k + (ndof-1)*j])] = shape_Ga3(m);
|
||||
// collapsed 1D access with forward mapping
|
||||
const int a_3d_diff = rd2q->forward_map3d_diff[m + (ndof-1)*(k + (ndof-1)*j)];
|
||||
rd2q->Ga3t[i + nqpt*a_3d_diff] = rd2q->Ga3[a_3d_diff + basis_dim3d_diff*i] =
|
||||
shape_Ga3(m);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// stores the mapping from 3D Bernstein multi-index (i,j,k,p-i-j-k) to the
|
||||
// lexicographic DOF ordering
|
||||
int p = ndof - 1;
|
||||
for (int i = 0; i < ndof; i++)
|
||||
{
|
||||
for (int j = 0; j < ndof-i; j++)
|
||||
{
|
||||
for (int k = 0; k < ndof-i-j; k++)
|
||||
{
|
||||
int dof = (p+1)*(p+2)*(p+3) / 6;
|
||||
int tet = (p-k)*(p-k+1)*(p-k+2) / 6;
|
||||
int tri = (p+1-k-j)*(p+2-k-j)/2;
|
||||
int multi_idx = dof - tet - tri + i;
|
||||
rd2q->lex_map[k + ndof*(j + ndof*i)] = multi_idx;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dof2quad_array.Append(d2q);
|
||||
}
|
||||
}
|
||||
return *d2q;
|
||||
}
|
||||
|
||||
// static method
|
||||
void H1Pos_TetrahedronElement::CalcShape(
|
||||
const int p, const real_t l1, const real_t l2, const real_t l3,
|
||||
|
||||
@@ -191,21 +191,6 @@ public:
|
||||
/// Construct the H1Pos_TriangleElement of order @a p
|
||||
H1Pos_TriangleElement(const int p);
|
||||
|
||||
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode) const override
|
||||
{
|
||||
return (mode == DofToQuad::RAGGED_TENSOR) ?
|
||||
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
|
||||
FiniteElement::GetDofToQuad(ir, mode);
|
||||
}
|
||||
|
||||
static const DofToQuad &GetRaggedTensorDofToQuad(
|
||||
const FiniteElement &fe, const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode,
|
||||
Array<DofToQuad*> &dof2quad_array);
|
||||
|
||||
const Array<int> &GetDofMap() const { return dof_map; }
|
||||
|
||||
// The size of shape is (p+1)(p+2)/2 (dof).
|
||||
static void CalcShape(const int p, const real_t x, const real_t y,
|
||||
real_t *shape);
|
||||
@@ -235,21 +220,6 @@ public:
|
||||
/// Construct the H1Pos_TetrahedronElement of order @a p
|
||||
H1Pos_TetrahedronElement(const int p);
|
||||
|
||||
const DofToQuad &GetDofToQuad(const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode) const override
|
||||
{
|
||||
return (mode == DofToQuad::RAGGED_TENSOR) ?
|
||||
GetRaggedTensorDofToQuad(*this, ir, mode, dof2quad_array) :
|
||||
FiniteElement::GetDofToQuad(ir, mode);
|
||||
}
|
||||
|
||||
static const DofToQuad &GetRaggedTensorDofToQuad(
|
||||
const FiniteElement &fe, const IntegrationRule &ir,
|
||||
DofToQuad::Mode mode,
|
||||
Array<DofToQuad*> &dof2quad_array);
|
||||
|
||||
const Array<int> &GetDofMap() const { return dof_map; }
|
||||
|
||||
// The size of shape is (p+1)(p+2)(p+3)/6 (dof).
|
||||
static void CalcShape(const int p, const real_t x, const real_t y,
|
||||
const real_t z, real_t *shape);
|
||||
|
||||
@@ -17,12 +17,6 @@
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
struct ScalarPyramid
|
||||
{
|
||||
// Default basis type for H1 and L2 pyramids
|
||||
static inline int DefaultType = 1; // Bergot(0) or Fuentes(1)
|
||||
};
|
||||
|
||||
/** Base class for arbitrary order basis functions on pyramid-shaped elements
|
||||
|
||||
This base class provides a common class to store temporary vectors,
|
||||
|
||||
+16
-6
@@ -73,11 +73,16 @@ public:
|
||||
void Project(const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &I) const override
|
||||
{ Project_RT(nk, dof2nk, fe, Trans, I); }
|
||||
// Gradient + rotation = Curl: H1 -> H(div)
|
||||
void ProjectGrad(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const override
|
||||
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, grad); }
|
||||
// Curl = Gradient + rotation: H1 -> H(div)
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectCurl2D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
|
||||
void GetFaceMap(const int face_id, Array<int> &face_map) const override;
|
||||
|
||||
@@ -143,7 +148,7 @@ public:
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
|
||||
/// @brief Return the mapping from lexicographically ordered face DOFs to
|
||||
/// lexicographically ordered element DOFs corresponding to local face
|
||||
@@ -205,11 +210,16 @@ public:
|
||||
void Project(const FiniteElement &fe, ElementTransformation &Trans,
|
||||
DenseMatrix &I) const override
|
||||
{ Project_RT(nk, dof2nk, fe, Trans, I); }
|
||||
// Gradient + rotation = Curl: H1 -> H(div)
|
||||
void ProjectGrad(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &grad) const override
|
||||
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, grad); }
|
||||
// Curl = Gradient + rotation: H1 -> H(div)
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectCurl2D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectGrad_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
};
|
||||
|
||||
|
||||
@@ -264,7 +274,7 @@ public:
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
};
|
||||
|
||||
class RT_WedgeElement : public VectorFiniteElement
|
||||
@@ -322,7 +332,7 @@ public:
|
||||
void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const override
|
||||
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
};
|
||||
|
||||
/** Arbitrary order H(Div) basis functions defined on pyramid-shaped elements
|
||||
@@ -418,7 +428,7 @@ public:
|
||||
virtual void ProjectCurl(const FiniteElement &fe,
|
||||
ElementTransformation &Trans,
|
||||
DenseMatrix &curl) const
|
||||
{ ProjectCurl3D_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
{ ProjectCurl_RT(nk, dof2nk, fe, Trans, curl); }
|
||||
|
||||
void CalcRawVShape(const IntegrationPoint &ip,
|
||||
DenseMatrix &shape) const;
|
||||
|
||||
+30
-88
@@ -228,19 +228,7 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
|
||||
}
|
||||
else if (!strncmp(name, "H1_", 3))
|
||||
{
|
||||
// Parse pyramid basis type if included in the name
|
||||
const char *pyr = strstr(name, "Pyr");
|
||||
if (pyr == NULL)
|
||||
{
|
||||
// Use default pyramid type elements
|
||||
fec = new H1_FECollection(atoi(name + 7), atoi(name + 3));
|
||||
}
|
||||
else
|
||||
{
|
||||
// Use specific pyramid type elements
|
||||
fec = new H1_FECollection(atoi(name + 7), atoi(name + 3),
|
||||
BasisType::GaussLobatto, atoi(pyr + 3));
|
||||
}
|
||||
fec = new H1_FECollection(atoi(name + 7), atoi(name + 3));
|
||||
}
|
||||
else if (!strncmp(name, "H1Pos_Trace_", 12))
|
||||
{
|
||||
@@ -257,44 +245,26 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
|
||||
}
|
||||
else if (!strncmp(name, "H1@", 3))
|
||||
{
|
||||
// Parse pyramid basis type if included in the name
|
||||
const char *pyr = strstr(name, "Pyr");
|
||||
if (pyr == NULL)
|
||||
{
|
||||
// Use default pyramid type elements
|
||||
fec = new H1_FECollection(atoi(name + 9), atoi(name + 5),
|
||||
BasisType::GetType(name[3]));
|
||||
}
|
||||
else
|
||||
{
|
||||
// Use specific pyramid type elements
|
||||
fec = new H1_FECollection(atoi(name + 9), atoi(name + 5),
|
||||
BasisType::GetType(name[3]),
|
||||
atoi(pyr + 3));
|
||||
}
|
||||
fec = new H1_FECollection(atoi(name + 9), atoi(name + 5),
|
||||
BasisType::GetType(name[3]));
|
||||
}
|
||||
else if (!strncmp(name, "L2", 2))
|
||||
else if (!strncmp(name, "L2_T", 4))
|
||||
fec = new L2_FECollection(atoi(name + 10), atoi(name + 6),
|
||||
atoi(name + 4));
|
||||
else if (!strncmp(name, "L2_", 3))
|
||||
{
|
||||
// Parse Map Type
|
||||
const int mtype = strstr(name, "Int") == NULL ?
|
||||
FiniteElement::VALUE : FiniteElement::INTEGRAL;
|
||||
|
||||
// Parse the base order
|
||||
const int p = atoi(strstr(name, "_P") + 2);
|
||||
|
||||
// Parse the mesh dimension
|
||||
const int dim = atoi(strstr(name, "D") - 1);
|
||||
|
||||
// Parse basis type if specified
|
||||
const char *t = strstr(name, "_T");
|
||||
const int btype = t == NULL ? BasisType::GaussLegendre : atoi(t + 2);
|
||||
|
||||
// Parse the pyramid type if specified
|
||||
const char *pyr = strstr(name, "Pyr");
|
||||
const int ptype = pyr == NULL ? 1 : atoi(pyr + 3);
|
||||
|
||||
// Create collection
|
||||
fec = new L2_FECollection(p, dim, btype, mtype, ptype);
|
||||
fec = new L2_FECollection(atoi(name + 7), atoi(name + 3));
|
||||
}
|
||||
else if (!strncmp(name, "L2Int_T", 7))
|
||||
{
|
||||
fec = new L2_FECollection(atoi(name + 13), atoi(name + 9),
|
||||
atoi(name + 7), FiniteElement::INTEGRAL);
|
||||
}
|
||||
else if (!strncmp(name, "L2Int_", 6))
|
||||
{
|
||||
fec = new L2_FECollection(atoi(name + 10), atoi(name + 6),
|
||||
BasisType::GaussLegendre,
|
||||
FiniteElement::INTEGRAL);
|
||||
}
|
||||
else if (!strncmp(name, "RT_Trace_", 9))
|
||||
{
|
||||
@@ -1739,10 +1709,9 @@ const int *RT1_3DFECollection::DofOrderForOrientation(Geometry::Type GeomType,
|
||||
|
||||
|
||||
H1_FECollection::H1_FECollection(const int p, const int dim, const int btype,
|
||||
const int pyr_type)
|
||||
const int pyrtype)
|
||||
: FiniteElementCollection(p)
|
||||
, dim(dim)
|
||||
, p_type(pyr_type)
|
||||
{
|
||||
MFEM_VERIFY(p >= 1, "H1_FECollection requires order >= 1.");
|
||||
MFEM_VERIFY(dim >= 0 && dim <= 3, "H1_FECollection requires 0 <= dim <= 3.");
|
||||
@@ -1755,14 +1724,7 @@ H1_FECollection::H1_FECollection(const int p, const int dim, const int btype,
|
||||
{
|
||||
case BasisType::GaussLobatto:
|
||||
{
|
||||
if (pyr_type == ScalarPyramid::DefaultType)
|
||||
{
|
||||
snprintf(h1_name, 32, "H1_%dD_P%d", dim, p);
|
||||
}
|
||||
else
|
||||
{
|
||||
snprintf(h1_name, 32, "H1_%dD_P%d_Pyr%d", dim, p, pyr_type);
|
||||
}
|
||||
snprintf(h1_name, 32, "H1_%dD_P%d", dim, p);
|
||||
break;
|
||||
}
|
||||
case BasisType::Positive:
|
||||
@@ -1948,11 +1910,11 @@ H1_FECollection::H1_FECollection(const int p, const int dim, const int btype,
|
||||
H1_dof[Geometry::TETRAHEDRON] = (TriDof*pm3)/3;
|
||||
H1_dof[Geometry::CUBE] = QuadDof*pm1;
|
||||
H1_dof[Geometry::PRISM] = TriDof*pm1;
|
||||
if (pyr_type == 0 || b_type == BasisType::Positive)
|
||||
if (pyrtype == 0 || b_type == BasisType::Positive)
|
||||
{
|
||||
H1_dof[Geometry::PYRAMID] = pm2*pm1*(2*p-3)/6; // Bergot (JSC)
|
||||
}
|
||||
else if (pyr_type == 1)
|
||||
else if (pyrtype == 1)
|
||||
{
|
||||
H1_dof[Geometry::PYRAMID] = pm1*pm1*pm1; // Fuentes
|
||||
}
|
||||
@@ -1973,15 +1935,13 @@ H1_FECollection::H1_FECollection(const int p, const int dim, const int btype,
|
||||
new H1_TetrahedronElement(p, btype);
|
||||
H1_Elements[Geometry::CUBE] = new H1_HexahedronElement(p, btype);
|
||||
H1_Elements[Geometry::PRISM] = new H1_WedgeElement(p, btype);
|
||||
if (pyr_type == 0)
|
||||
if (pyrtype == 0)
|
||||
{
|
||||
H1_Elements[Geometry::PYRAMID] =
|
||||
new H1_BergotPyramidElement(p, btype);
|
||||
H1_Elements[Geometry::PYRAMID] = new H1_BergotPyramidElement(p, btype);
|
||||
}
|
||||
else
|
||||
{
|
||||
H1_Elements[Geometry::PYRAMID] =
|
||||
new H1_FuentesPyramidElement(p, btype);
|
||||
H1_Elements[Geometry::PYRAMID] = new H1_FuentesPyramidElement(p, btype);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2188,7 +2148,6 @@ L2_FECollection::L2_FECollection(const int p, const int dim, const int btype,
|
||||
: FiniteElementCollection(p)
|
||||
, dim(dim)
|
||||
, m_type(map_type)
|
||||
, p_type(pyr_type)
|
||||
{
|
||||
MFEM_VERIFY(p >= 0, "L2_FECollection requires order >= 0.");
|
||||
|
||||
@@ -2204,25 +2163,10 @@ L2_FECollection::L2_FECollection(const int p, const int dim, const int btype,
|
||||
switch (btype)
|
||||
{
|
||||
case BasisType::GaussLegendre:
|
||||
if (pyr_type == ScalarPyramid::DefaultType)
|
||||
{
|
||||
snprintf(d_name, 32, "%s_%dD_P%d", prefix, dim, p);
|
||||
}
|
||||
else
|
||||
{
|
||||
snprintf(d_name, 32, "%s_%dD_P%d_Pyr%d", prefix, dim, p, pyr_type);
|
||||
}
|
||||
snprintf(d_name, 32, "%s_%dD_P%d", prefix, dim, p);
|
||||
break;
|
||||
default:
|
||||
if (pyr_type == ScalarPyramid::DefaultType)
|
||||
{
|
||||
snprintf(d_name, 32, "%s_T%d_%dD_P%d", prefix, btype, dim, p);
|
||||
}
|
||||
else
|
||||
{
|
||||
snprintf(d_name, 32, "%s_T%d_%dD_P%d_Pyr%d",
|
||||
prefix, btype, dim, p, pyr_type);
|
||||
}
|
||||
snprintf(d_name, 32, "%s_T%d_%dD_P%d", prefix, btype, dim, p);
|
||||
}
|
||||
|
||||
for (int g = 0; g < Geometry::NumGeom; g++)
|
||||
@@ -2341,13 +2285,11 @@ L2_FECollection::L2_FECollection(const int p, const int dim, const int btype,
|
||||
L2_Elements[Geometry::PRISM] = new L2_WedgeElement(p, btype);
|
||||
if (pyr_type == 0)
|
||||
{
|
||||
L2_Elements[Geometry::PYRAMID] =
|
||||
new L2_BergotPyramidElement(p, btype);
|
||||
L2_Elements[Geometry::PYRAMID] = new L2_BergotPyramidElement(p, btype);
|
||||
}
|
||||
else
|
||||
{
|
||||
L2_Elements[Geometry::PYRAMID] =
|
||||
new L2_FuentesPyramidElement(p, btype);
|
||||
L2_Elements[Geometry::PYRAMID] = new L2_FuentesPyramidElement(p, btype);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+5
-37
@@ -100,10 +100,6 @@ public:
|
||||
return FiniteElementForGeometry(GeomType);
|
||||
}
|
||||
|
||||
/** @brief Returns a collection of the trace elements.
|
||||
|
||||
@note The collection is owned by the caller and is NOT deleted in the
|
||||
destructor. */
|
||||
virtual FiniteElementCollection *GetTraceCollection() const;
|
||||
|
||||
virtual ~FiniteElementCollection();
|
||||
@@ -254,14 +250,6 @@ public:
|
||||
its GetOrder() method. */
|
||||
virtual FiniteElementCollection *Clone(int p) const;
|
||||
|
||||
/** @brief Return the order parameter used to construct this collection.
|
||||
* This differs from GetOrder() depending on the collection type. */
|
||||
virtual int GetConstructorOrder() const
|
||||
{
|
||||
MFEM_ABORT("Collection " << Name() << " does not support GetConstructorOrder");
|
||||
return -1;
|
||||
}
|
||||
|
||||
protected:
|
||||
const int base_p; ///< Order as returned by GetOrder().
|
||||
|
||||
@@ -290,7 +278,7 @@ protected:
|
||||
class H1_FECollection : public FiniteElementCollection
|
||||
{
|
||||
protected:
|
||||
int dim, b_type, p_type;
|
||||
int dim, b_type;
|
||||
char h1_name[32];
|
||||
FiniteElement *H1_Elements[Geometry::NumGeom];
|
||||
int H1_dof[Geometry::NumGeom];
|
||||
@@ -299,7 +287,7 @@ protected:
|
||||
public:
|
||||
explicit H1_FECollection(const int p, const int dim = 3,
|
||||
const int btype = BasisType::GaussLobatto,
|
||||
const int pyr_type = ScalarPyramid::DefaultType);
|
||||
const int pyrtype = 1);
|
||||
|
||||
const FiniteElement *
|
||||
FiniteElementForGeometry(Geometry::Type GeomType) const override;
|
||||
@@ -324,10 +312,7 @@ public:
|
||||
const int *GetDofMap(Geometry::Type GeomType, int p) const;
|
||||
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new H1_FECollection(p, dim, b_type, p_type); }
|
||||
|
||||
int GetConstructorOrder() const override
|
||||
{ return base_p; }
|
||||
{ return new H1_FECollection(p, dim, b_type); }
|
||||
|
||||
virtual ~H1_FECollection();
|
||||
};
|
||||
@@ -358,10 +343,6 @@ class H1_Trace_FECollection : public H1_FECollection
|
||||
public:
|
||||
H1_Trace_FECollection(const int p, const int dim,
|
||||
const int btype = BasisType::GaussLobatto);
|
||||
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new H1_Trace_FECollection(p, dim+1, b_type); }
|
||||
|
||||
};
|
||||
|
||||
/// Arbitrary order "L2-conforming" discontinuous finite elements.
|
||||
@@ -371,7 +352,6 @@ private:
|
||||
int dim;
|
||||
int b_type; // BasisType
|
||||
int m_type; // map type
|
||||
int p_type; // Pyramid type (0 -> Bergot, 1 -> Fuentes)
|
||||
char d_name[32];
|
||||
ScalarFiniteElement *L2_Elements[Geometry::NumGeom];
|
||||
ScalarFiniteElement *Tr_Elements[Geometry::NumGeom];
|
||||
@@ -384,7 +364,7 @@ public:
|
||||
L2_FECollection(const int p, const int dim,
|
||||
const int btype = BasisType::GaussLegendre,
|
||||
const int map_type = FiniteElement::VALUE,
|
||||
const int pyr_type = ScalarPyramid::DefaultType);
|
||||
const int pyrtype = 1);
|
||||
|
||||
const FiniteElement *
|
||||
FiniteElementForGeometry(Geometry::Type GeomType) const override;
|
||||
@@ -414,10 +394,7 @@ public:
|
||||
int GetBasisType() const { return b_type; }
|
||||
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new L2_FECollection(p, dim, b_type, m_type, p_type); }
|
||||
|
||||
int GetConstructorOrder() const override
|
||||
{ return base_p; }
|
||||
{ return new L2_FECollection(p, dim, b_type, m_type); }
|
||||
|
||||
virtual ~L2_FECollection();
|
||||
};
|
||||
@@ -479,9 +456,6 @@ public:
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new RT_FECollection(p, dim, cb_type, ob_type); }
|
||||
|
||||
int GetConstructorOrder() const override
|
||||
{ return base_p-1; }
|
||||
|
||||
virtual ~RT_FECollection();
|
||||
};
|
||||
|
||||
@@ -562,9 +536,6 @@ public:
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new ND_FECollection(p, dim, cb_type, ob_type); }
|
||||
|
||||
int GetConstructorOrder() const override
|
||||
{ return dim>1 ? base_p : base_p+1; }
|
||||
|
||||
virtual ~ND_FECollection();
|
||||
};
|
||||
|
||||
@@ -577,9 +548,6 @@ public:
|
||||
ND_Trace_FECollection(const int p, const int dim,
|
||||
const int cb_type = BasisType::GaussLobatto,
|
||||
const int ob_type = BasisType::GaussLegendre);
|
||||
|
||||
FiniteElementCollection *Clone(int p) const override
|
||||
{ return new ND_Trace_FECollection(p, dim+1, cb_type, ob_type); }
|
||||
};
|
||||
|
||||
/// Arbitrary order 3D H(curl)-conforming Nedelec finite elements in 1D.
|
||||
|
||||
+2
-207
@@ -22,8 +22,6 @@
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstdarg>
|
||||
#include <unordered_map>
|
||||
#include <unordered_set>
|
||||
|
||||
using namespace std;
|
||||
|
||||
@@ -4529,210 +4527,6 @@ void FiniteElementSpace
|
||||
}
|
||||
}
|
||||
|
||||
void FiniteElementSpace::GetBoundaryLoopEdgeDofs(
|
||||
const Array<int> &boundary_element_indices,
|
||||
Array<int> &boundary_edge_dofs,
|
||||
Array<int> *dof_edges,
|
||||
Array<int> *dof_boundary_elements) const
|
||||
{
|
||||
MFEM_VERIFY(mesh->Dimension() >= 2,
|
||||
"GetBoundaryLoopEdgeDofs requires 2D or 3D meshes to find edge objects");
|
||||
|
||||
boundary_edge_dofs.SetSize(0);
|
||||
if (dof_edges) { dof_edges->SetSize(0); }
|
||||
if (dof_boundary_elements) { dof_boundary_elements->SetSize(0); }
|
||||
|
||||
// A DOF that appears in exactly one selected boundary element lies on the
|
||||
// bounding loop; one appearing in two or more is interior to the boundary
|
||||
// region and is dropped. Count occurrences of each DOF (using scratch maps,
|
||||
// exposed only as parallel-indexed Array<int> below) and record, on first
|
||||
// sight, the local edge and boundary element carrying it.
|
||||
//
|
||||
// The count is over GetEdgeDofs, which returns endpoint vertex DOFs as well
|
||||
// as edge-interior DOFs (relevant for collections such as ND_R2D that carry
|
||||
// vertex DOFs). Edge-interior DOFs occur once per edge, so the count mainly
|
||||
// resolves vertex DOFs: a vertex shared by several elements is interior and
|
||||
// dropped, while a genuine loop-corner (open-curve endpoint) vertex is kept.
|
||||
// This is why we count GetEdgeDofs rather than collecting GetEdgeInteriorDofs,
|
||||
// which would omit the endpoint vertex DOFs the method is documented to keep.
|
||||
// The 3D removal criterion (any edge in two or more faces) matches the
|
||||
// parallel version rather than a parity toggle.
|
||||
std::unordered_map<int, int> dof_count, dof_edge, dof_belem;
|
||||
Array<int> edge_dofs, edges, edge_orientations;
|
||||
|
||||
const int dim = mesh->Dimension();
|
||||
for (int i = 0; i < boundary_element_indices.Size(); ++i)
|
||||
{
|
||||
const int boundary_element_idx = boundary_element_indices[i];
|
||||
std::unordered_set<int> boundary_element_dofs;
|
||||
|
||||
if (dim == 3)
|
||||
{
|
||||
// Boundary elements are 2D faces; extract their 1D edges.
|
||||
int face_index, face_orientation;
|
||||
mesh->GetBdrElementFace(boundary_element_idx, &face_index,
|
||||
&face_orientation);
|
||||
mesh->GetFaceEdges(face_index, edges, edge_orientations);
|
||||
}
|
||||
else
|
||||
{
|
||||
// Boundary elements are 1D segments, each being a single edge.
|
||||
mesh->GetBdrElementEdges(boundary_element_idx, edges, edge_orientations);
|
||||
MFEM_VERIFY(edges.Size() == 1,
|
||||
"2D boundary element should have exactly one edge");
|
||||
}
|
||||
|
||||
for (int j = 0; j < edges.Size(); ++j)
|
||||
{
|
||||
GetEdgeDofs(edges[j], edge_dofs);
|
||||
for (int k = 0; k < edge_dofs.Size(); ++k)
|
||||
{
|
||||
const int dof = edge_dofs[k];
|
||||
// Count each DOF once per boundary element and record metadata the
|
||||
// first time it is seen, so H1 DOFs shared by multiple edges of the
|
||||
// same element are not double counted.
|
||||
if (boundary_element_dofs.insert(dof).second &&
|
||||
dof_count[dof]++ == 0)
|
||||
{
|
||||
dof_edge[dof] = edges[j];
|
||||
dof_belem[dof] = boundary_element_idx;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Emit the DOFs seen in exactly one selected boundary element, in a
|
||||
// deterministic (increasing DOF index) order shared by all output arrays.
|
||||
std::vector<int> kept;
|
||||
kept.reserve(dof_count.size());
|
||||
for (const auto &[dof, count] : dof_count)
|
||||
{
|
||||
if (count == 1) { kept.push_back(dof); }
|
||||
}
|
||||
std::sort(kept.begin(), kept.end());
|
||||
|
||||
boundary_edge_dofs.Reserve(static_cast<int>(kept.size()));
|
||||
if (dof_edges) { dof_edges->Reserve(static_cast<int>(kept.size())); }
|
||||
if (dof_boundary_elements)
|
||||
{
|
||||
dof_boundary_elements->Reserve(static_cast<int>(kept.size()));
|
||||
}
|
||||
for (int dof : kept)
|
||||
{
|
||||
boundary_edge_dofs.Append(dof);
|
||||
if (dof_edges) { dof_edges->Append(dof_edge[dof]); }
|
||||
if (dof_boundary_elements) { dof_boundary_elements->Append(dof_belem[dof]); }
|
||||
}
|
||||
}
|
||||
|
||||
void FiniteElementSpace::GetBoundaryElementsByAttribute(
|
||||
const Array<int> &bdr_attrs,
|
||||
std::vector<Array<int>> &attr_to_elements)
|
||||
{
|
||||
// One (initially empty) list of boundary elements per requested attribute,
|
||||
// indexed to match bdr_attrs.
|
||||
attr_to_elements.assign(bdr_attrs.Size(), Array<int>());
|
||||
|
||||
// Map attribute value -> position in bdr_attrs for quick lookup.
|
||||
std::unordered_map<int, int> attr_to_index;
|
||||
for (int i = 0; i < bdr_attrs.Size(); ++i)
|
||||
{
|
||||
attr_to_index[bdr_attrs[i]] = i;
|
||||
}
|
||||
|
||||
// Bucket boundary elements by their attribute.
|
||||
for (int i = 0; i < mesh->GetNBE(); ++i)
|
||||
{
|
||||
int attr = mesh->GetBdrElement(i)->GetAttribute();
|
||||
auto it = attr_to_index.find(attr);
|
||||
if (it != attr_to_index.end())
|
||||
{
|
||||
attr_to_elements[it->second].Append(i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void FiniteElementSpace::GetBoundaryElementsByAttribute(int bdr_attr,
|
||||
Array<int> &boundary_elements)
|
||||
{
|
||||
boundary_elements.SetSize(0);
|
||||
|
||||
for (int i = 0; i < mesh->GetNBE(); ++i)
|
||||
{
|
||||
if (mesh->GetBdrElement(i)->GetAttribute() == bdr_attr)
|
||||
{
|
||||
boundary_elements.Append(i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void FiniteElementSpace::ComputeLoopEdgeOrientations(
|
||||
const Array<int> &dof_edges,
|
||||
const Array<int> &dof_boundary_elements,
|
||||
const Vector &loop_normal,
|
||||
Array<int> &dof_orientations) const
|
||||
{
|
||||
MFEM_VERIFY(dof_edges.Size() == dof_boundary_elements.Size(),
|
||||
"dof_edges and dof_boundary_elements must be parallel-indexed");
|
||||
|
||||
const int ndof = dof_edges.Size();
|
||||
dof_orientations.SetSize(ndof);
|
||||
|
||||
Array<int> edge_verts, bdr_elem_verts;
|
||||
Vector edge_vec(3), to_edge_vec(3), cross_product(3);
|
||||
for (int i = 0; i < ndof; i++)
|
||||
{
|
||||
const int edge_id = dof_edges[i];
|
||||
const int bdr_elem_idx = dof_boundary_elements[i];
|
||||
|
||||
// Get edge vertices
|
||||
mesh->GetEdgeVertices(edge_id, edge_verts);
|
||||
|
||||
const real_t *v0 = mesh->GetVertex(edge_verts[0]);
|
||||
const real_t *v1 = mesh->GetVertex(edge_verts[1]);
|
||||
|
||||
// Get boundary element vertices
|
||||
mesh->GetBdrElement(bdr_elem_idx)->GetVertices(bdr_elem_verts);
|
||||
|
||||
// Find the third vertex (not part of the edge)
|
||||
int third_vertex = -1;
|
||||
for (int j = 0; j < bdr_elem_verts.Size(); j++)
|
||||
{
|
||||
int v = bdr_elem_verts[j];
|
||||
if (v != edge_verts[0] && v != edge_verts[1])
|
||||
{
|
||||
third_vertex = v;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (third_vertex == -1)
|
||||
{
|
||||
MFEM_ABORT("Boundary element " << bdr_elem_idx << " has only 2 vertices, "
|
||||
"but 3D boundary elements must have at least 3 vertices");
|
||||
}
|
||||
|
||||
const real_t *v2 = mesh->GetVertex(third_vertex);
|
||||
|
||||
// Edge vector
|
||||
for (int j = 0; j < 3; j++) { edge_vec[j] = v1[j] - v0[j]; }
|
||||
|
||||
// Vector from third vertex to edge (use edge midpoint)
|
||||
for (int j = 0; j < 3; j++)
|
||||
{
|
||||
real_t edge_midpoint = (v0[j] + v1[j]) * 0.5;
|
||||
to_edge_vec[j] = edge_midpoint - v2[j];
|
||||
}
|
||||
|
||||
// Cross product: to_edge × edge
|
||||
to_edge_vec.cross3D(edge_vec, cross_product);
|
||||
|
||||
// Check alignment with loop normal
|
||||
real_t dot_product = cross_product * loop_normal;
|
||||
dof_orientations[i] = (dot_product > 0) ? 1 : -1;
|
||||
}
|
||||
}
|
||||
|
||||
FiniteElementCollection *FiniteElementSpace::Load(Mesh *m, std::istream &input)
|
||||
{
|
||||
string buff;
|
||||
@@ -4837,8 +4631,9 @@ FiniteElementCollection *FiniteElementSpace::Load(Mesh *m, std::istream &input)
|
||||
|
||||
ElementDofOrdering GetEVectorOrdering(const FiniteElementSpace& fes)
|
||||
{
|
||||
return (UsesTensorBasis(fes) || fes.UsesRaggedTensorBasis()) ?
|
||||
return UsesTensorBasis(fes)?
|
||||
ElementDofOrdering::LEXICOGRAPHIC:
|
||||
ElementDofOrdering::NATIVE;
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
@@ -22,7 +22,6 @@
|
||||
#include "restriction.hpp"
|
||||
#include <iostream>
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -1390,80 +1389,6 @@ public:
|
||||
virtual void GetExteriorTrueDofs(Array<int> &exterior_dofs,
|
||||
int component = -1) const;
|
||||
|
||||
/** @brief Extract the edge degrees of freedom of a boundary "loop".
|
||||
|
||||
Here a "loop" is the set of boundary edges bounding the region covered by
|
||||
@a boundary_element_indices: in 3D the outer edges of a patch of boundary
|
||||
faces, in 2D the boundary segments themselves. An edge that is shared by
|
||||
two (or more) of the selected boundary elements is interior to that region
|
||||
rather than on its bounding loop, so its DOFs are excluded from the result.
|
||||
This exclusion of interior DOFs is the defining feature of the method.
|
||||
|
||||
The three output arrays share a single indexing: for each valid index @a i,
|
||||
@a dof_edges[i] and @a dof_boundary_elements[i] describe the DOF
|
||||
@a boundary_edge_dofs[i].
|
||||
|
||||
@param[in] boundary_element_indices Boundary element indices spanning a
|
||||
boundary surface (3D) or curve (2D).
|
||||
@param[out] boundary_edge_dofs Local DOF indices on the boundary loop.
|
||||
@param[out] dof_edges Optional; local edge index carrying each DOF.
|
||||
@param[out] dof_boundary_elements Optional; a boundary element containing
|
||||
each DOF.
|
||||
|
||||
@note In 3D the edge DOFs are extracted from the 1D edges of the 2D
|
||||
boundary faces; in 2D they come directly from the 1D boundary segments, so
|
||||
@a dof_edges then holds the boundary element (segment) edge indices.
|
||||
@note This method uses GetEdgeDofs internally, which returns both vertex and
|
||||
edge DOFs. Standard Nédélec elements (ND_FECollection) have no vertex DOFs,
|
||||
so only genuine edge DOFs appear. Collections that carry vertex DOFs (e.g.
|
||||
ND_R2D_FECollection) additionally contribute the vertex DOFs at loop
|
||||
endpoints.
|
||||
@note This is the serial version. For parallel meshes, use the parallel
|
||||
version in ParFiniteElementSpace which handles processor boundaries
|
||||
correctly.
|
||||
@note Requires a 2D or 3D mesh to identify edge objects. The method will
|
||||
assert if called on 1D meshes.
|
||||
@note Only supports conforming meshes; non-conforming meshes are not
|
||||
supported. */
|
||||
void GetBoundaryLoopEdgeDofs(const Array<int> &boundary_element_indices,
|
||||
Array<int> &boundary_edge_dofs,
|
||||
Array<int> *dof_edges = nullptr,
|
||||
Array<int> *dof_boundary_elements = nullptr) const;
|
||||
|
||||
/** @brief Get boundary elements grouped by attribute.
|
||||
|
||||
For each attribute in @a bdr_attrs, collect the indices of all boundary
|
||||
elements carrying that attribute. The result is indexed to match
|
||||
@a bdr_attrs: @a attr_to_elements[i] holds the boundary elements with
|
||||
attribute @a bdr_attrs[i]. */
|
||||
void GetBoundaryElementsByAttribute(
|
||||
const Array<int> &bdr_attrs,
|
||||
std::vector<Array<int>> &attr_to_elements);
|
||||
|
||||
/** @brief Get all boundary elements with a specific attribute. */
|
||||
void GetBoundaryElementsByAttribute(int bdr_attr,
|
||||
Array<int> &boundary_elements);
|
||||
|
||||
/** @brief Compute edge orientations relative to a boundary loop direction.
|
||||
|
||||
For each boundary-loop DOF described by @a dof_edges and
|
||||
@a dof_boundary_elements (see GetBoundaryLoopEdgeDofs), determine whether
|
||||
the carrying edge is
|
||||
traversed in the direction consistent with @a loop_normal, following the
|
||||
right-hand rule. Intended for 3D meshes.
|
||||
|
||||
@param[in] dof_edges Local edge index of each DOF (parallel-indexed with
|
||||
the boundary_edge_dofs output of GetBoundaryLoopEdgeDofs).
|
||||
@param[in] dof_boundary_elements A boundary element containing each DOF,
|
||||
using the same indexing as @a dof_edges.
|
||||
@param[in] loop_normal Normal vector defining the loop orientation.
|
||||
@param[out] dof_orientations Orientation (+1 or -1) for each DOF, using the
|
||||
same indexing as @a dof_edges. */
|
||||
void ComputeLoopEdgeOrientations(const Array<int> &dof_edges,
|
||||
const Array<int> &dof_boundary_elements,
|
||||
const Vector &loop_normal,
|
||||
Array<int> &dof_orientations) const;
|
||||
|
||||
/// Convert a Boolean marker array to a list containing all marked indices.
|
||||
static void MarkerToList(const Array<int> &marker, Array<int> &list);
|
||||
|
||||
@@ -1589,18 +1514,6 @@ public:
|
||||
return dynamic_cast<const L2_FECollection*>(fec) != NULL;
|
||||
}
|
||||
|
||||
/// @brief Return true if the mesh contains only one topology, the elements are
|
||||
/// all triangles or tetrahedrons, and the elements are ragged tensor elements
|
||||
/// i.e. Bernstein/positive basis.
|
||||
bool UsesRaggedTensorBasis() const
|
||||
{
|
||||
bool simplex = this->GetMesh()->IsSimplexMesh();
|
||||
bool positive =
|
||||
dynamic_cast<const mfem::H1Pos_TriangleElement *>(this->GetTypicalFE()) ||
|
||||
dynamic_cast<const mfem::H1Pos_TetrahedronElement *>(this->GetTypicalFE());
|
||||
return simplex && positive;
|
||||
}
|
||||
|
||||
/** In variable-order spaces on nonconforming (NC) meshes, this function
|
||||
controls whether strict conformity is enforced in cases where coarse
|
||||
edges/faces have higher polynomial order than their fine NC neighbors.
|
||||
|
||||
+24
-167
@@ -2256,104 +2256,6 @@ void GridFunction::AccumulateAndCountBdrTangentValues(
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::AccumulateAndCountTraceValues(
|
||||
Coefficient *coeff[], VectorCoefficient *vcoeff,
|
||||
Array<int> &values_counter)
|
||||
{
|
||||
if (vcoeff)
|
||||
{
|
||||
MFEM_VERIFY(fes->GetVDim() == vcoeff->GetVDim(),
|
||||
"vcoeff vdim != fes VDim");
|
||||
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetMapType() ==
|
||||
FiniteElement::VALUE &&
|
||||
fes->GetTypicalTraceElement()->GetRangeType() ==
|
||||
FiniteElement::SCALAR,
|
||||
"Can only call ProjectTraceCoefficient on scalar value-type "
|
||||
"trace elements. "
|
||||
"Use ProjectTraceCoefficientNormal for RT and "
|
||||
"ProjectTraceCoefficientTangent for ND finite elements.");
|
||||
}
|
||||
|
||||
Array<int> vdofs;
|
||||
Vector vc;
|
||||
|
||||
values_counter.SetSize(Size());
|
||||
values_counter = 0;
|
||||
|
||||
const int vdim = fes->GetVDim();
|
||||
HostReadWrite();
|
||||
|
||||
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
|
||||
{
|
||||
|
||||
const FiniteElement *fe = fes->GetFaceElement(i);
|
||||
const int fdof = fe->GetDof();
|
||||
ElementTransformation *transf = fes->GetMesh()->GetFaceTransformation(i);
|
||||
const IntegrationRule &ir = fe->GetNodes();
|
||||
fes->GetFaceVDofs(i, vdofs);
|
||||
|
||||
for (int j = 0; j < fdof; j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(j);
|
||||
transf->SetIntPoint(&ip);
|
||||
if (vcoeff) { vcoeff->Eval(vc, *transf, ip); }
|
||||
for (int d = 0; d < vdim; d++)
|
||||
{
|
||||
if (!vcoeff && !coeff[d]) { continue; }
|
||||
|
||||
real_t val = vcoeff ? vc(d) : coeff[d]->Eval(*transf, ip);
|
||||
int ind = vdofs[fdof*d+j];
|
||||
if ( ind < 0 )
|
||||
{
|
||||
val = -val, ind = -1-ind;
|
||||
}
|
||||
if (++values_counter[ind] == 1)
|
||||
{
|
||||
(*this)(ind) = val;
|
||||
}
|
||||
else
|
||||
{
|
||||
(*this)(ind) += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::AccumulateAndCountTraceTangentValues(
|
||||
VectorCoefficient &vcoeff, Array<int> &values_counter)
|
||||
{
|
||||
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
|
||||
MFEM_VERIFY(fes->GetTypicalTraceElement()
|
||||
->GetRangeType() == FiniteElement::VECTOR &&
|
||||
fes->GetTypicalTraceElement()
|
||||
->GetMapType() == FiniteElement::H_CURL,
|
||||
"Not an ND FE space!");
|
||||
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetPhysRangeDim(
|
||||
fes->GetMesh()->SpaceDimension()) == vcoeff.GetVDim(),
|
||||
"vcoeff vdim != PhysRangeDim");
|
||||
|
||||
const FiniteElement *fe;
|
||||
ElementTransformation *T;
|
||||
Array<int> dofs;
|
||||
Vector lvec;
|
||||
|
||||
values_counter.SetSize(Size());
|
||||
values_counter = 0;
|
||||
|
||||
HostReadWrite();
|
||||
|
||||
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
|
||||
{
|
||||
fe = fes->GetFaceElement(i);
|
||||
T = fes->GetMesh()->GetFaceTransformation(i);
|
||||
fes->GetFaceVDofs(i, dofs);
|
||||
lvec.SetSize(fe->GetDof());
|
||||
fe->Project(vcoeff, *T, lvec);
|
||||
accumulate_dofs(dofs, lvec, *this, values_counter);
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::ComputeMeans(AvgType type, Array<int> &zones_per_vdof)
|
||||
{
|
||||
switch (type)
|
||||
@@ -2796,74 +2698,6 @@ void GridFunction::ProjectCoefficient(VectorCoefficient &vcoeff,
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficient(Coefficient *coeff[])
|
||||
{
|
||||
Array<int> values_counter;
|
||||
AccumulateAndCountTraceValues(coeff, NULL, values_counter);
|
||||
ComputeMeans(ARITHMETIC, values_counter);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficient(Coefficient &coeff)
|
||||
{
|
||||
MFEM_VERIFY(FESpace()->GetVDim() == 1, "ProjectTraceCoefficient(Coefficient&)"
|
||||
"is only valid for scalar GridFunction");
|
||||
Coefficient *coeff_p = &coeff;
|
||||
ProjectTraceCoefficient(&coeff_p);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficient(VectorCoefficient &vcoeff)
|
||||
{
|
||||
MFEM_VERIFY(FESpace()->GetVDim() == vcoeff.GetVDim(),
|
||||
"Incompatible vcoeff vdim and fes vdim");
|
||||
Array<int> values_counter;
|
||||
AccumulateAndCountTraceValues(NULL, &vcoeff, values_counter);
|
||||
ComputeMeans(ARITHMETIC, values_counter);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff)
|
||||
{
|
||||
MFEM_VERIFY(fes->GetVDim() == 1, "fespace VDim != 1");
|
||||
MFEM_VERIFY(fes->GetTypicalTraceElement()->GetRangeType() ==
|
||||
FiniteElement::SCALAR &&
|
||||
fes->GetTypicalTraceElement()->GetMapType() ==
|
||||
FiniteElement::INTEGRAL, "Not an RT FE space!");
|
||||
MFEM_VERIFY(vcoeff.GetVDim() == fes->GetMesh()->SpaceDimension(),
|
||||
"vcoeff vdim (" << vcoeff.GetVDim()
|
||||
<< ") != SpaceDimension ("
|
||||
<< fes->GetMesh()->SpaceDimension() << ")");
|
||||
|
||||
const FiniteElement *fe;
|
||||
ElementTransformation *T;
|
||||
Array<int> dofs;
|
||||
int dim = vcoeff.GetVDim();
|
||||
Vector vc(dim), nor(dim), lvec;
|
||||
|
||||
for (int i = 0; i < fes->GetMesh()->GetNumFaces(); i++)
|
||||
{
|
||||
fe = fes->GetFaceElement(i);
|
||||
T = fes->GetMesh()->GetFaceTransformation(i);
|
||||
const IntegrationRule &ir = fe->GetNodes();
|
||||
lvec.SetSize(fe->GetDof());
|
||||
for (int j = 0; j < ir.GetNPoints(); j++)
|
||||
{
|
||||
const IntegrationPoint &ip = ir.IntPoint(j);
|
||||
T->SetIntPoint(&ip);
|
||||
vcoeff.Eval(vc, *T, ip);
|
||||
CalcOrtho(T->Jacobian(), nor);
|
||||
lvec(j) = (vc * nor);
|
||||
}
|
||||
fes->GetFaceVDofs(i, dofs);
|
||||
SetSubVector(dofs, lvec);
|
||||
}
|
||||
}
|
||||
|
||||
void GridFunction::ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff)
|
||||
{
|
||||
Array<int> values_counter;
|
||||
AccumulateAndCountTraceTangentValues(vcoeff, values_counter);
|
||||
ComputeMeans(ARITHMETIC, values_counter);
|
||||
}
|
||||
|
||||
void GridFunction::ProjectCoefficientGlobalL2(VectorCoefficient &vcoeff,
|
||||
real_t rtol, int iter)
|
||||
{
|
||||
@@ -5418,6 +5252,30 @@ void GridFunction::GetElementBounds(const PLBound &plb,
|
||||
Vector &lower, Vector &upper,
|
||||
const int vdim) const
|
||||
{
|
||||
if (UseDevice() && Device::Allows(Backend::DEVICE_MASK) &&
|
||||
plb.GetBasisType() != BasisType::Positive &&
|
||||
UsesTensorBasis(*fes))
|
||||
{
|
||||
const FiniteElement &fe = *fes->GetTypicalFE();
|
||||
const int rdim = fe.GetDim();
|
||||
const int fes_dim = fes->GetVDim();
|
||||
const int nel = fes->GetNE();
|
||||
const int nd = fe.GetDof();
|
||||
|
||||
Vector e_vec(nd*fes_dim*nel, Device::GetDeviceMemoryType());
|
||||
e_vec.UseDevice(true);
|
||||
const ElementRestrictionOperator *elem_restr =
|
||||
fes->GetElementRestriction(ElementDofOrdering::LEXICOGRAPHIC);
|
||||
MFEM_VERIFY(elem_restr != nullptr,
|
||||
"Element restriction is required for device bounds.");
|
||||
elem_restr->Mult(*this, e_vec);
|
||||
|
||||
plb.GetElementBoundsKernel(rdim, fes_dim, e_vec, lower, upper, vdim);
|
||||
lower.HostRead();
|
||||
upper.HostRead();
|
||||
return;
|
||||
}
|
||||
|
||||
int nel = fes->GetNE();
|
||||
int fes_dim = fes->GetVDim();
|
||||
lower.SetSize(nel*(vdim > 0 ? 1 :fes_dim));
|
||||
@@ -5452,7 +5310,6 @@ PLBound GridFunction::GetBounds(Vector &lower, Vector &upper,
|
||||
{
|
||||
int max_order = fes->GetMaxElementOrder();
|
||||
PLBound plb(fes, ref_factor*(max_order+1));
|
||||
|
||||
Vector lel, uel;
|
||||
GetElementBounds(plb, lel, uel, vdim);
|
||||
|
||||
|
||||
+3
-27
@@ -578,13 +578,6 @@ protected:
|
||||
const Array<int> &bdr_attr,
|
||||
Array<int> &values_counter);
|
||||
|
||||
void AccumulateAndCountTraceValues(Coefficient *coeff[],
|
||||
VectorCoefficient *vcoeff,
|
||||
Array<int> &values_counter);
|
||||
|
||||
void AccumulateAndCountTraceTangentValues(VectorCoefficient &vcoeff,
|
||||
Array<int> &values_counter);
|
||||
|
||||
// Complete the computation of averages; called e.g. after
|
||||
// AccumulateAndCountZones().
|
||||
void ComputeMeans(AvgType type, Array<int> &zones_per_vdof);
|
||||
@@ -670,23 +663,6 @@ public:
|
||||
ProjectBdrCoefficient(&coeff_p, attr);
|
||||
}
|
||||
|
||||
/// Project a Coefficient on a GridFunction defined on H1 trace space
|
||||
void ProjectTraceCoefficient(Coefficient *coeff[]);
|
||||
void ProjectTraceCoefficient(Coefficient &coeff);
|
||||
|
||||
/** @brief Project a VectorCoefficient @a vcoeff on a GridFunction
|
||||
defined on a Vector H1 trace space. Note that this also works
|
||||
for a scalar H1 trace space, where only the first component of
|
||||
@a vcoeff is used. */
|
||||
void ProjectTraceCoefficient(VectorCoefficient &vcoeff);
|
||||
/** @brief Project a VectorCoefficient on a GridFunction
|
||||
defined on an RT trace space */
|
||||
void ProjectTraceCoefficientNormal(VectorCoefficient &vcoeff);
|
||||
/** @brief Project a VectorCoefficient on a GridFunction
|
||||
defined on an ND trace space */
|
||||
void ProjectTraceCoefficientTangent(VectorCoefficient &vcoeff);
|
||||
|
||||
|
||||
/** @brief Project a VectorCoefficient on the GridFunction, modifying only
|
||||
DOFs on the boundary associated with the boundary attributes marked in
|
||||
the @a attr array. */
|
||||
@@ -1791,8 +1767,8 @@ public:
|
||||
const int ref_factor=1, const int vdim=-1) const;
|
||||
|
||||
/// Computes the \ref PLBound for the gridfunction with number of control
|
||||
/// points based on @a ref_factor, and returns the bounds for each element
|
||||
/// ordered byNODES:
|
||||
/// points based on \p ref_factor, and returns the bounds for each element
|
||||
/// ordered byNodes:
|
||||
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
|
||||
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}. We also return the
|
||||
/// PLBound object used to compute the bounds.
|
||||
@@ -1826,7 +1802,7 @@ public:
|
||||
const int vdim = -1) const;
|
||||
|
||||
/// Compute bounds on the grid function for all the elements. The bounds
|
||||
/// are returned in @b lower and @b upper, ordered byNODES:
|
||||
/// are returned in @b lower and @b upper, ordered byNodes:
|
||||
/// lower_{0,0}, lower_{1,0}, ..., lower_{ne-1,0},
|
||||
/// lower_{0,1}, ..., lower_{ne-1,vdim-1}
|
||||
void GetElementBounds(const PLBound &plb, Vector &lower, Vector &upper,
|
||||
|
||||
+384
-3128
File diff suppressed because it is too large
Load Diff
+103
-494
@@ -12,9 +12,6 @@
|
||||
#ifndef MFEM_GSLIB
|
||||
#define MFEM_GSLIB
|
||||
|
||||
#include <map>
|
||||
#include <vector>
|
||||
|
||||
#include "../config/config.hpp"
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include "pgridfunc.hpp"
|
||||
@@ -24,45 +21,6 @@
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
/* gslib license and copyright statement for code adapted from gslib:
|
||||
|
||||
Copyright (c) 2008-2024, UCHICAGO ARGONNE, LLC.
|
||||
|
||||
The UChicago Argonne, LLC as Operator of Argonne National
|
||||
Laboratory holds copyright in the Software. The copyright holder
|
||||
reserves all rights except those expressly granted to licensees,
|
||||
and U.S. Government license rights.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions
|
||||
are met:
|
||||
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the disclaimer below.
|
||||
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the disclaimer (as noted below)
|
||||
in the documentation and/or other materials provided with the
|
||||
distribution.
|
||||
|
||||
3. Neither the name of ANL nor the names of its contributors
|
||||
may be used to endorse or promote products derived from this software
|
||||
without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
|
||||
FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL
|
||||
UCHICAGO ARGONNE, LLC, THE U.S. DEPARTMENT OF
|
||||
ENERGY OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||
SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED
|
||||
TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||
DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||
THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
namespace gslib
|
||||
{
|
||||
struct comm;
|
||||
@@ -122,18 +80,13 @@ protected:
|
||||
// IntegrationRules for simplex->Quad/Hex and to project to p_max in-case of
|
||||
// p-refinement.
|
||||
Array<IntegrationRule *> ir_split;
|
||||
/// Integration rules built at the field polynomial order (only for surface
|
||||
/// meshes when mesh order is not the same as gridfunction order).
|
||||
Array<IntegrationRule *> ir_split_sol;
|
||||
/// Order at which #ir_split_sol was built; -1 means not built.
|
||||
int ir_split_sol_order = -1;
|
||||
Array<FiniteElementSpace *> fes_rst_map; //FESpaces to map Quad/Hex->Simplex
|
||||
Array<GridFunction *> gf_rst_map; // GridFunctions to map Quad/Hex->Simplex
|
||||
FiniteElementCollection *fec_map_lin;
|
||||
void *fdataD;
|
||||
struct gslib::crystal *cr; // gslib's internal data
|
||||
struct gslib::comm *gsl_comm; // gslib's internal data
|
||||
int dim, spacedim, points_cnt; // mesh dimension and number of points
|
||||
int dim, points_cnt; // mesh dimension and number of points
|
||||
Array<unsigned int> gsl_code, gsl_proc, gsl_elem, gsl_mfem_elem;
|
||||
Vector gsl_mesh, gsl_ref, gsl_dist, gsl_mfem_ref;
|
||||
Array<unsigned int> recv_proc, recv_index; // data for custom interpolation
|
||||
@@ -142,8 +95,6 @@ protected:
|
||||
AvgType avgtype; // average type used for L2 functions
|
||||
Array<int> split_element_map;
|
||||
Array<int> split_element_index;
|
||||
// Geometry::Type (as int) of the original element for each split quad.
|
||||
Array<int> split_element_geom;
|
||||
int NE_split_total; // total number of elements after mesh splitting
|
||||
int mesh_points_cnt; // number of mesh nodes
|
||||
// Tolerance to ignore points found beyond the mesh boundary.
|
||||
@@ -151,235 +102,111 @@ protected:
|
||||
double bdr_tol;
|
||||
// Use CPU functions for Mesh/GridFunction on device for gslib1.0.7
|
||||
bool gpu_to_cpu_fallback = false;
|
||||
// Check if a point is inside the oriented bounding box of an
|
||||
// element before the Newton iteration.
|
||||
// Note: only used in MFEM implementation (not in gslib) which currently
|
||||
// supports GPU kernels for area meshes in 2D, volume meshes in 3D,
|
||||
// and surface meshes in 1D/2D/3D.
|
||||
bool obb_check = true;
|
||||
|
||||
// Device specific data used for FindPoints
|
||||
struct DEV_STRUCT
|
||||
struct
|
||||
{
|
||||
bool setup_device = false;
|
||||
bool find_device = false;
|
||||
int local_hash_size, dof1d, dof1d_sol, lh_nx, gh_nx;
|
||||
int local_hash_size, dof1d, dof1d_sol, h_o_size, h_nx;
|
||||
double newt_tol; // Tolerance specified during setup for Newton solve
|
||||
struct gslib::crystal *cr;
|
||||
struct gslib::hash_data_3 *hash3;
|
||||
struct gslib::hash_data_2 *hash2;
|
||||
mutable Vector bb, wtend, gll1d, lagcoeff, gll1d_sol, lagcoeff_sol;
|
||||
mutable Array<unsigned int> lh_offset, gh_offset;
|
||||
mutable Vector lh_min, lh_fac, gh_min, gh_fac;
|
||||
// Tolerance to mark points found on the surface as CODE_INTERNAL
|
||||
// or CODE_BORDER. This is needed because we cannot only use reference
|
||||
// space coordinates to determine if a point is located inside the
|
||||
// element or not.
|
||||
mutable double surf_dist_tol;
|
||||
mutable Array<unsigned int> loc_hash_offset;
|
||||
mutable Vector loc_hash_min, loc_hash_fac;
|
||||
} DEV;
|
||||
|
||||
// Helper function to setup and free gslib's crystal router.
|
||||
void SetupCrystal(); // Called inside Setup and SetupSurf_base
|
||||
void FreeCrystal(); // Called inside FreeData
|
||||
|
||||
/// Use GSLIB for communication and interpolation. Updates field_out on
|
||||
/// host.
|
||||
/// Use GSLIB for communication and interpolation
|
||||
virtual void InterpolateH1(const GridFunction &field_in, Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
/// Uses GSLIB Crystal Router for communication followed by MFEM's
|
||||
/// interpolation functions. Updates field_out on host.
|
||||
/// interpolation functions
|
||||
virtual void InterpolateGeneral(const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
|
||||
/** @brief Since GSLIB is designed to work with quads/hexes, we split every
|
||||
* triangle/tet/prism/pyramid element into quads/hexes. */
|
||||
/// Since GSLIB is designed to work with quads/hexes, we split every
|
||||
/// triangle/tet/prism/pyramid element into quads/hexes.
|
||||
virtual void SetupSplitMeshes();
|
||||
|
||||
/** @brief Setup integration points that will be used to interpolate the
|
||||
* nodal location at points expected by GSLIB. */
|
||||
/// Setup integration points that will be used to interpolate the nodal
|
||||
/// location at points expected by GSLIB.
|
||||
virtual void SetupIntegrationRuleForSplitMesh(Mesh *mesh,
|
||||
IntegrationRule *irule,
|
||||
int order);
|
||||
|
||||
/** @brief Build integration rules at the given @a order for each split mesh
|
||||
* and store them in @a ir_out. Requires that \ref SetupSplitMeshes has
|
||||
* already been called. */
|
||||
virtual void SetupIntegrationRules(const int order,
|
||||
Array<IntegrationRule *> &ir_out);
|
||||
|
||||
/** @brief Helper function that calls \ref SetupSplitMeshes and
|
||||
* \ref SetupIntegrationRules. */
|
||||
/// Helper function that calls \ref SetupSplitMeshes and
|
||||
/// \ref SetupIntegrationRuleForSplitMesh.
|
||||
virtual void SetupSplitMeshesAndIntegrationRules(const int order);
|
||||
|
||||
/** @brief Get GridFunction value at the points expected by GSLIB.
|
||||
* @param[in] gf_in Grid function to evaluate.
|
||||
* @param[out] node_vals Output values.
|
||||
* @param[in] ir_in If non-null, use these rules instead of #ir_split.
|
||||
* @param[in] by_element If true, output has element-major layout
|
||||
* [nel][vdim][ndofs]; otherwise component-major
|
||||
* layout [vdim][total_pts]. */
|
||||
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals,
|
||||
const Array<IntegrationRule *> *ir_in = nullptr,
|
||||
bool by_element = false) const;
|
||||
/// Get GridFunction value at the points expected by GSLIB.
|
||||
virtual void GetNodalValues(const GridFunction *gf_in, Vector &node_vals) const;
|
||||
|
||||
/** @brief Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For
|
||||
* simplices, find the original element number (that was split into
|
||||
* micro quads/hexes) during the setup phase. */
|
||||
/// Map {r,s,t} coordinates from [-1,1] to [0,1] for MFEM. For simplices,
|
||||
/// find the original element number (that was split into micro quads/hexes)
|
||||
/// during the setup phase.
|
||||
virtual void MapRefPosAndElemIndices();
|
||||
|
||||
/// FindPoints locally on device for 3D.
|
||||
// Device functions
|
||||
// FindPoints locally on device for 3D.
|
||||
void FindPointsLocal3(const Vector &point_pos, int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l, Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l, int npt);
|
||||
|
||||
/// FindPoints locally on device for 2D.
|
||||
// FindPoints locally on device for 2D.
|
||||
void FindPointsLocal2(const Vector &point_pos, int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l, Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l, int npt);
|
||||
|
||||
/// FindPoints locally on device for 3D surface elements.
|
||||
void FindPointsSurfLocal3(const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l,
|
||||
int npt);
|
||||
|
||||
/// FindPoints locally on device for 3D edge elements.
|
||||
void FindPointsEdgeLocal3(const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l,
|
||||
int npt);
|
||||
|
||||
/// FindPoints locally on device for 2D edge elements.
|
||||
void FindPointsEdgeLocal2(const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &gsl_code_dev_l,
|
||||
Array<unsigned int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &gsl_dist_l,
|
||||
int npt);
|
||||
|
||||
/// Interpolate on device for 3D.
|
||||
// Interpolate on device for 3D.
|
||||
void InterpolateLocal3(const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int dof1dsol);
|
||||
|
||||
/// Interpolate on device for 2D.
|
||||
int nel, int dof1dsol);
|
||||
// Interpolate on device for 2D.
|
||||
void InterpolateLocal2(const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int dof1dsol);
|
||||
int nel, int dof1dsol);
|
||||
|
||||
/// Interpolate on device for 1D.
|
||||
void InterpolateLocal1(const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp, int dof1dsol);
|
||||
|
||||
/// Prepare data for device execution for volume meshes.
|
||||
// Prepare data for device functions.
|
||||
void SetupDevice();
|
||||
|
||||
/** @brief Searches positions given in physical space by @a point_pos.
|
||||
/** Searches positions given in physical space by @a point_pos.
|
||||
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
|
||||
byVDim: (XYZ,XYZ,....XYZ) specified by @a point_pos_ordering. */
|
||||
void FindPointsOnDevice(const Vector &point_pos,
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/** @brief Interpolation of field values at prescribed reference space
|
||||
* positions.
|
||||
* @param[in] field_in_evec E-vector of grid function to be interpolated.
|
||||
* Assumed ordering is NDOFSxVDIMxNEL
|
||||
* @param[in] nel Number of elements in the mesh.
|
||||
* @param[in] ncomp Number of components in the field.
|
||||
* @param[in] dof1dsol Number of degrees of freedom in each reference
|
||||
* space direction.
|
||||
* @param[in] ordering Ordering of the out field values: byNodes/byVDIM
|
||||
*
|
||||
* @param[out] field_out Interpolated values. For points that are not
|
||||
* found the value is set to
|
||||
* #default_interp_value. */
|
||||
/** Interpolation of field values at prescribed reference space positions.
|
||||
@param[in] field_in_evec E-vector of grid function to be interpolated.
|
||||
Assumed ordering is NDOFSxVDIMxNEL
|
||||
@param[in] nel Number of elements in the mesh.
|
||||
@param[in] ncomp Number of components in the field.
|
||||
@param[in] dof1dsol Number of degrees of freedom in each reference
|
||||
space direction.
|
||||
@param[in] ordering Ordering of the out field values: byNodes/byVDIM
|
||||
|
||||
@param[out] field_out Interpolated values. For points that are not found
|
||||
the value is set to #default_interp_value. */
|
||||
void InterpolateOnDevice(const Vector &field_in_evec, Vector &field_out,
|
||||
const int nel, const int ncomp,
|
||||
const int dof1dsol, const int ordering);
|
||||
|
||||
/** @brief Interpolation of field values at prescribed reference space
|
||||
* positions for surface meshes. */
|
||||
void InterpolateSurfBase(const Vector &field_in, Vector &field_out,
|
||||
const int nel, const int ncomp,
|
||||
const int dof1dsol, const int field_out_ordering);
|
||||
|
||||
/// Preprocess 2D surface mesh needed for FindPoints.
|
||||
void findptsedge_setup_2(DEV_STRUCT &devs,
|
||||
const double *const elx[2],
|
||||
const unsigned n,
|
||||
const unsigned int nel,
|
||||
const unsigned m,
|
||||
const double bbox_rel_size_inc,
|
||||
const unsigned int local_hash_size,
|
||||
const unsigned int global_hash_size,
|
||||
const Vector *aabb_sz_inc);
|
||||
|
||||
/// Preprocess 3D surface mesh needed for FindPoints.
|
||||
void findptssurf_setup_3(DEV_STRUCT &devs,
|
||||
const double *const elx[3],
|
||||
const unsigned n,
|
||||
const unsigned int nel,
|
||||
const unsigned m,
|
||||
const double bbox_rel_size_inc,
|
||||
const unsigned int local_hash_size,
|
||||
const unsigned int global_hash_size,
|
||||
const int rD,
|
||||
const Vector *aabb_sz_inc);
|
||||
|
||||
/** @brief Shared implementation for the public surface-setup methods.
|
||||
*
|
||||
* @details Initializes the surface-search data structures, builds the
|
||||
* split-element representation expected by gslib, and constructs the
|
||||
* element bounding boxes used by the MFEM surface kernels.
|
||||
*
|
||||
* If @a aabb_sz_inc is null, the setup stores the default oriented
|
||||
* bounding boxes and uses @a bbox_rel_size_inc as their relative size
|
||||
* increase factor.
|
||||
*
|
||||
* If @a aabb_sz_inc is non-null, the setup stores axis-aligned bounding
|
||||
* boxes only, applies the requested absolute AABB expansion in each
|
||||
* physical direction, and adjusts the tolerance @a bdr_tol so points
|
||||
* found in the expanded region are classified as border points.
|
||||
*
|
||||
* @param[in] m Input surface mesh.
|
||||
* @param[in] bbox_rel_size_inc Relative size increase applied when
|
||||
* expanding each element bounding box during
|
||||
* setup.
|
||||
* @param[in] aabb_sz_inc Optional total absolute AABB expansion
|
||||
* applied to the stored axis-aligned
|
||||
* bounding boxes after construction.
|
||||
* @param[in] newt_tol Newton tolerance for the point-search
|
||||
* kernels.
|
||||
*/
|
||||
void SetupSurf_Base(Mesh &m,
|
||||
const double bbox_rel_size_inc,
|
||||
const Vector *aabb_sz_inc,
|
||||
const double newt_tol);
|
||||
public:
|
||||
/// Serial constructor
|
||||
FindPointsGSLIB();
|
||||
|
||||
/// Serial constructor + setup with given Mesh (see \ref Setup)
|
||||
FindPointsGSLIB(Mesh &mesh_in, const double bbox_rel_size_inc = 0.1,
|
||||
FindPointsGSLIB(Mesh &mesh_in, const double bb_t = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
@@ -388,7 +215,7 @@ public:
|
||||
FindPointsGSLIB(MPI_Comm comm_);
|
||||
|
||||
/// Constructor + setup with given ParMesh (see \ref Setup)
|
||||
FindPointsGSLIB(ParMesh &mesh_in, const double bbox_rel_size_inc = 0.1,
|
||||
FindPointsGSLIB(ParMesh &mesh_in, const double bb_t = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
#endif
|
||||
@@ -397,72 +224,25 @@ public:
|
||||
FindPointsGSLIB(const FindPointsGSLIB&) = delete;
|
||||
FindPointsGSLIB& operator=(const FindPointsGSLIB&) = delete;
|
||||
|
||||
/** @brief Preprocess the internal mesh in gslib.
|
||||
|
||||
@details Initializes the internal mesh in gslib, by sending the
|
||||
positions of the Gauss-Lobatto nodes of the input Mesh object \p m.
|
||||
/** Initializes the internal mesh in gslib, by sending the positions of the
|
||||
Gauss-Lobatto nodes of the input Mesh object \p m.
|
||||
Note: not tested with periodic (L2).
|
||||
Note: the input mesh \p m must have Nodes set.
|
||||
|
||||
@param[in] m Input mesh.
|
||||
@param[in] bbox_rel_size_inc (Optional) Relative size increase applied
|
||||
when expanding each element bounding box.
|
||||
@param[in] newt_tol (Optional) Newton tolerance for the gslib
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for
|
||||
simultaneous iteration. This alters
|
||||
performance and memory footprint.
|
||||
*/
|
||||
void Setup(Mesh &m, const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
@param[in] m Input mesh.
|
||||
@param[in] bb_t (Optional) Relative size of bounding box around
|
||||
each element.
|
||||
@param[in] newt_tol (Optional) Newton tolerance for the gslib
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for simultaneous
|
||||
iteration. This alters performance and
|
||||
memory footprint.*/
|
||||
|
||||
void Setup(Mesh &m, const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
/// Preprocess the surface mesh to compute data for FindPoints.
|
||||
void SetupSurf(Mesh &m,
|
||||
const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12);
|
||||
|
||||
/** @brief Preprocess the surface mesh to compute data for FindPoints using
|
||||
* absolute AABB expansion.
|
||||
*
|
||||
* @details This method computes only axis-aligned bounding boxes and
|
||||
* increases their total length by a user-specified amount in each
|
||||
* physical direction. The absolute AABB expansion is applied
|
||||
* symmetrically to the lower and upper bounds.
|
||||
*
|
||||
* The size of @a aabb_sz_inc determines how the expansion values are
|
||||
* interpreted:
|
||||
* - `1`: one expansion value used in every direction for every element
|
||||
* - `NElements`: one expansion value per element, reused in x/y/z
|
||||
* directions
|
||||
* - `SpaceDim`: one expansion value per physical direction, reused for
|
||||
* every element
|
||||
* - `NElements*SpaceDim`: one expansion value per element and direction,
|
||||
* ordered as `(dx1,dy1,dz1, ... dxN,dyN,dzN)`
|
||||
*
|
||||
* This method disables the oriented bounding-box precheck because the
|
||||
* stored boxes are modified only in their axis-aligned representation.
|
||||
*
|
||||
* @param[in] m Input surface mesh.
|
||||
* @param[in] aabb_sz_inc Total absolute AABB expansion applied in
|
||||
* each physical direction to the stored
|
||||
* axis-aligned bounding boxes.
|
||||
* @param[in] newt_tol Newton tolerance for the point-search
|
||||
* kernels.
|
||||
*
|
||||
* @note We disable the oriented bounding box check with this setup.
|
||||
* @a bdr_tol is also adjusted so that all points in the AABBs can
|
||||
* be found.
|
||||
*/
|
||||
void SetupSurfWithAABBExpansion(Mesh &m, const Vector &aabb_sz_inc,
|
||||
const double newt_tol = 1.0e-12);
|
||||
|
||||
|
||||
/** @brief Searches positions given in physical space by \p point_pos.
|
||||
|
||||
@details These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
|
||||
/** Searches positions given in physical space by \p point_pos.
|
||||
These positions can be ordered byNodes: (XXX...,YYY...,ZZZ) or
|
||||
byVDim: (XYZ,XYZ,....XYZ) specified by \p point_pos_ordering.
|
||||
|
||||
This function populates the following member variables:
|
||||
#gsl_code Return codes for each point: inside element (0),
|
||||
element boundary (1), not found (2).
|
||||
@@ -481,77 +261,40 @@ public:
|
||||
#gsl_dist Distance between the sought and the found point
|
||||
in physical space. */
|
||||
void FindPoints(const Vector &point_pos,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
/// Convenience function when point positions are in a ParticleVector
|
||||
void FindPoints(const ParticleVector &point_pos)
|
||||
{
|
||||
FindPoints(point_pos, point_pos.GetOrdering());
|
||||
}
|
||||
|
||||
/** @brief Searches positions given in physical space by \p point_pos on
|
||||
* surface mesh. */
|
||||
void FindPointsSurf(const Vector &point_pos,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/// Convenience function when point positions are in a ParticleVector
|
||||
void FindPointsSurf(const ParticleVector &point_pos)
|
||||
{
|
||||
FindPointsSurf(point_pos, point_pos.GetOrdering());
|
||||
}
|
||||
|
||||
/// Setup FindPoints and search positions
|
||||
void FindPoints(Mesh &m, const Vector &point_pos,
|
||||
const int point_pos_ordering = Ordering::byNODES,
|
||||
const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
/** @brief Interpolation of field values at prescribed reference space
|
||||
* positions.
|
||||
|
||||
/** Interpolation of field values at prescribed reference space positions.
|
||||
@param[in] field_in Function values that will be interpolated on the
|
||||
reference positions. Note: it is assumed that
|
||||
\p field_in is in H1 and in the same space as the
|
||||
mesh that was given to Setup().
|
||||
@param[out] field_out Interpolated values. For points that are not found
|
||||
the value is set to #default_interp_value.
|
||||
The output ordering is determined from field_in.
|
||||
|
||||
@note: field_out is moved to device if field_in is on device. Otherwise,
|
||||
field_out memory allocation is not changed.
|
||||
*/
|
||||
The output ordering is determined from field_in.*/
|
||||
virtual void Interpolate(const GridFunction &field_in, Vector &field_out);
|
||||
|
||||
/// Interpolation of field values, with output ordering specification.
|
||||
virtual void Interpolate(const GridFunction &field_in, Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
|
||||
/** @brief Same as Interpolate but for surface meshes */
|
||||
virtual void InterpolateSurf(const GridFunction &field_in,
|
||||
Vector &field_out);
|
||||
|
||||
/** @brief Same as Interpolate but for surface meshes with specified output
|
||||
ordering */
|
||||
virtual void InterpolateSurf(const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
const int field_out_ordering);
|
||||
|
||||
/** @brief Search positions and interpolate.
|
||||
*
|
||||
* @details The ordering (byNODES or byVDIM) of the output values in
|
||||
* \p field_out corresponds to the ordering used in the input
|
||||
* GridFunction \p field_in.
|
||||
*/
|
||||
/** Search positions and interpolate. The ordering (byNODES or byVDIM) of
|
||||
the output values in \p field_out corresponds to the ordering used
|
||||
in the input GridFunction \p field_in. */
|
||||
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
|
||||
Vector &field_out,
|
||||
int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
/// Search positions and interpolate with given point and output ordering.
|
||||
void Interpolate(const Vector &point_pos, const GridFunction &field_in,
|
||||
Vector &field_out, const int point_pos_ordering,
|
||||
const int field_out_ordering);
|
||||
|
||||
/** Setup FindPoints, search positions and interpolate. The ordering (byNODES
|
||||
or byVDIM) of the output values in \p field_out corresponds to the
|
||||
ordering used in the input GridFunction \p field_in. */
|
||||
@@ -559,41 +302,32 @@ public:
|
||||
const GridFunction &field_in, Vector &field_out,
|
||||
const int point_pos_ordering = Ordering::byNODES);
|
||||
|
||||
/** @brief Average type to be used for L2 functions in-case a point is
|
||||
* located at an element boundary where the function might be multi-valued.
|
||||
*/
|
||||
/// Average type to be used for L2 functions in-case a point is located at
|
||||
/// an element boundary where the function might be multi-valued.
|
||||
virtual void SetL2AvgType(AvgType avgtype_) { avgtype = avgtype_; }
|
||||
|
||||
/** @brief Set the default interpolation value for points that are not found in the mesh. */
|
||||
/// Set the default interpolation value for points that are not found in the
|
||||
/// mesh.
|
||||
virtual void SetDefaultInterpolationValue(double interp_value_)
|
||||
{
|
||||
default_interp_value = interp_value_;
|
||||
}
|
||||
|
||||
/** @brief Tolerance for detecting points outside the 'curvilinear' boundary.
|
||||
*
|
||||
* @details When using FindPoints, gslib may return points as found on the
|
||||
* boundary even when they are slightly outside the domain. This tolerance
|
||||
* is used to filter such points based on the distance^2 value and mark them
|
||||
* as not found.
|
||||
*
|
||||
* @note When the SetupSurfWithAABBExpansion method is used for surface
|
||||
* meshes, this tolerance is automatically computed based on the size of
|
||||
* expanded AABBs. Using this method will override that computed tolerance.
|
||||
* */
|
||||
/// Set the tolerance for detecting points outside the 'curvilinear' boundary
|
||||
/// that gslib may return as found on the boundary. Points found on boundary
|
||||
/// with distance greater than @ bdr_tol are marked as not found.
|
||||
virtual void SetDistanceToleranceForPointsFoundOnBoundary(double bdr_tol_)
|
||||
{
|
||||
bdr_tol = bdr_tol_;
|
||||
}
|
||||
|
||||
/** @brief Enable/Disable use of CPU functions for GPU data if the gslib
|
||||
* version is older. */
|
||||
/// Enable/Disable use of CPU functions for GPU data if the gslib version
|
||||
/// is older.
|
||||
virtual void SetGPUtoCPUFallback(bool mode) { gpu_to_cpu_fallback = mode; }
|
||||
|
||||
/** @brief Cleans up memory allocated internally by gslib.
|
||||
|
||||
@details Note that in parallel, this must be called before MPI_Finalize,
|
||||
as it calls MPI_Comm_free() for internal gslib communicators. FreeData is
|
||||
/** Cleans up memory allocated internally by gslib.
|
||||
Note that in parallel, this must be called before MPI_Finalize(), as it
|
||||
calls MPI_Comm_free() for internal gslib communicators. FreeData is
|
||||
also called by the class destructor and there are no memory leaks if the
|
||||
destructor is called before MPI_Finalize(). If the destructor is called
|
||||
after MPI_Finalize(), there will be an error because gslib will try to
|
||||
@@ -601,8 +335,8 @@ public:
|
||||
*/
|
||||
virtual void FreeData();
|
||||
|
||||
/** @brief Return code for each point searched by FindPoints:
|
||||
* inside element (0), element boundary (1), or not found (2). */
|
||||
/// Return code for each point searched by FindPoints: inside element (0), on
|
||||
/// element boundary (1), or not found (2).
|
||||
virtual const Array<unsigned int> &GetCode() const { return gsl_code; }
|
||||
/// Return element number for each point found by FindPoints.
|
||||
virtual const Array<unsigned int> &GetElem() const { return gsl_mfem_elem; }
|
||||
@@ -610,15 +344,15 @@ public:
|
||||
virtual const Array<unsigned int> &GetProc() const { return gsl_proc; }
|
||||
/// Return reference coordinates for each point found by FindPoints.
|
||||
virtual const Vector &GetReferencePosition() const { return gsl_mfem_ref; }
|
||||
/// Return distance between the sought and the found point in physical space.
|
||||
/// Return distance between the sought and the found point in physical space,
|
||||
/// for each point found by FindPoints.
|
||||
virtual const Vector &GetDist() const { return gsl_dist; }
|
||||
|
||||
/** @brief Return element number for each point found by FindPoints
|
||||
* corresponding to GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with
|
||||
* simplices. */
|
||||
/// Return element number for each point found by FindPoints corresponding to
|
||||
/// GSLIB mesh. gsl_mfem_elem != gsl_elem for mesh with simplices.
|
||||
virtual const Array<unsigned int> &GetGSLIBElem() const { return gsl_elem; }
|
||||
/** @brief Return reference coordinates in [-1,1] (internal range in GSLIB)
|
||||
* for each point found by FindPoints. */
|
||||
/// Return reference coordinates in [-1,1] (internal range in GSLIB) for each
|
||||
/// point found by FindPoints.
|
||||
virtual const Vector &GetGSLIBReferencePosition() const { return gsl_ref; }
|
||||
|
||||
/// Get array of indices of not-found points.
|
||||
@@ -661,7 +395,7 @@ public:
|
||||
|
||||
/// Return the axis-aligned bounding boxes (AABB) computed during \ref Setup.
|
||||
/// The size of the returned vector is (nel x nverts x dim), where nel is the
|
||||
/// number of elements (after splitting for simplicies), nverts is number of
|
||||
/// number of elements (after splitting for simplcies), nverts is number of
|
||||
/// vertices (4 in 2D, 8 in 3D), and dim is the spatial dimension.
|
||||
void GetAxisAlignedBoundingBoxes(Vector &aabb) const;
|
||||
|
||||
@@ -675,18 +409,6 @@ public:
|
||||
/// \p obbV, a vector of size (nel x nverts x dim) .
|
||||
void GetOrientedBoundingBoxes(DenseTensor &obbA, Vector &obbC,
|
||||
Vector &obbV) const;
|
||||
|
||||
/** @brief Return the bounding boxes as a mesh on rank 0.
|
||||
*
|
||||
* @param[in] type Bounding-box type: 0 - AABB, 1 - OBB.
|
||||
*
|
||||
* @return On rank 0, returns a newly allocated mesh containing the
|
||||
* bounding boxes. The caller owns the returned pointer and is responsible
|
||||
* for deleting it. On other ranks, returns nullptr.
|
||||
*/
|
||||
Mesh *GetBoundingBoxMesh(int type);
|
||||
|
||||
virtual const Vector &GetGLLMesh() const { return gsl_mesh; }
|
||||
};
|
||||
|
||||
/** \brief OversetFindPointsGSLIB enables use of findpts for arbitrary number of
|
||||
@@ -715,28 +437,25 @@ public:
|
||||
Note: not tested with periodic meshes (L2).
|
||||
Note: the input mesh \p m must have Nodes set.
|
||||
|
||||
@param[in] m Input mesh.
|
||||
@param[in] meshid A unique # for each overlapping mesh.
|
||||
This id is used to make sure that points
|
||||
being searched are not looked for in the
|
||||
mesh that they belong to.
|
||||
@param[in] gfmax (Optional) GridFunction in H1 that is used
|
||||
as a discriminator when one point is
|
||||
located in multiple meshes. The mesh that
|
||||
maximizes gfmax is chosen. For example,
|
||||
using the distance field based on the
|
||||
overlapping boundaries is helpful for
|
||||
convergence during Schwarz iterations.
|
||||
@param[in] bbox_rel_size_inc (Optional) Relative size increase applied
|
||||
when expanding each element bounding box.
|
||||
@param[in] newt_tol (Optional) Newton tolerance for the gslib
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for
|
||||
simultaneous iteration. This alters
|
||||
performance and memory footprint.*/
|
||||
void Setup(Mesh &m, const int meshid, GridFunction *gfmax = nullptr,
|
||||
const double bbox_rel_size_inc = 0.1,
|
||||
const double newt_tol = 1.0e-12,
|
||||
@param[in] m Input mesh.
|
||||
@param[in] meshid A unique # for each overlapping mesh. This id is
|
||||
used to make sure that points being searched are not
|
||||
looked for in the mesh that they belong to.
|
||||
@param[in] gfmax (Optional) GridFunction in H1 that is used as a
|
||||
discriminator when one point is located in multiple
|
||||
meshes. The mesh that maximizes gfmax is chosen.
|
||||
For example, using the distance field based on the
|
||||
overlapping boundaries is helpful for convergence
|
||||
during Schwarz iterations.
|
||||
@param[in] bb_t (Optional) Relative size of bounding box around
|
||||
each element.
|
||||
@param[in] newt_tol (Optional) Newton tolerance for the gslib
|
||||
search methods.
|
||||
@param[in] npt_max (Optional) Number of points for simultaneous
|
||||
iteration. This alters performance and
|
||||
memory footprint.*/
|
||||
void Setup(Mesh &m, const int meshid, GridFunction *gfmax = NULL,
|
||||
const double bb_t = 0.1, const double newt_tol = 1.0e-12,
|
||||
const int npt_max = 256);
|
||||
|
||||
/** Searches positions given in physical space by \p point_pos. All output
|
||||
@@ -792,7 +511,7 @@ class GSOPGSLIB
|
||||
protected:
|
||||
struct gslib::crystal *cr; // gslib's internal data
|
||||
struct gslib::comm *gsl_comm; // gslib's internal data
|
||||
struct gslib::gs_data *gsl_data = nullptr;
|
||||
struct gslib::gs_data *gsl_data = NULL;
|
||||
int num_ids;
|
||||
|
||||
public:
|
||||
@@ -817,116 +536,6 @@ public:
|
||||
void GS(Vector &senddata, GSOp op);
|
||||
};
|
||||
|
||||
#if defined(MFEM_USE_MPI)
|
||||
/** \brief Class to map a point in physical space to candidate ranks.
|
||||
*
|
||||
* This class builds a Cartesian-aligned tensor grid that covers the entire
|
||||
* domain and precomputes which ranks have elements intersecting each
|
||||
* grid cell. Given a point in physical space, the grid cell containing
|
||||
* the point is determined, and the list of candidate ranks whose
|
||||
* elements intersect that cell is returned. This yields a fast, conservative
|
||||
* point-to-rank candidate query. This is used internally by FindPointsGSLIB
|
||||
* to speed up point searches in parallel.
|
||||
*
|
||||
* See Mittal et al., "General Field Evaluation in High-Order Meshes on GPUs".
|
||||
* (2025). Computers & Fluids. for technical details.
|
||||
*
|
||||
*/
|
||||
class GlobalBBoxTensorGridMap
|
||||
{
|
||||
private:
|
||||
struct gslib::crystal *cr = nullptr; // gslib's internal data
|
||||
struct gslib::comm *gsl_comm = nullptr; // gslib's internal data
|
||||
int sdim, n_local_cells, num_procs;
|
||||
Array<int> gmap_n;
|
||||
Vector gmap_bnd_min, gmap_bnd_max;
|
||||
Vector gmap_fac;
|
||||
Array<int> ggrid_map;
|
||||
|
||||
void SetupCrystal(const MPI_Comm &comm);
|
||||
public:
|
||||
/// Constructor for a given mesh and number of tensor grid divisions
|
||||
GlobalBBoxTensorGridMap(ParMesh &pmesh, int nx);
|
||||
|
||||
/** @brief Constructor for given element bounds and spatial dimension.
|
||||
*
|
||||
* @details This constructor must be called collectively on \a comm.
|
||||
* Supports spatial dimensions 1, 2, and 3, and accepts nel == 0 on a rank.
|
||||
*
|
||||
* Assumes elmin, elmax Ordering::byNodes:
|
||||
* elmin -> [x_{0,min},x_{1,min},... ,y_{0,min},y_{1,min},..,z_{nel-1,min}]
|
||||
* elmax -> [x_{0,max},x_{1,max},... ,y_{0,max},y_{1,max},..,z_{nel-1,max}]
|
||||
* Note elmin, elmax can be obtained using GridFunction::GetElementBounds()
|
||||
*
|
||||
* When by_max_size=false, n gives the number of tensor-grid divisions in
|
||||
* each direction. When by_max_size=true, n is a per-rank size hint used to
|
||||
* derive a uniform global resolution. The communicator-wide sum of n is
|
||||
* converted to nx = ceil(pow(sum(n), 1./sdim)) in each direction, so n is
|
||||
* not a hard cap on ggrid_map.Size().
|
||||
*/
|
||||
GlobalBBoxTensorGridMap(const MPI_Comm &comm, Vector &elmin,
|
||||
Vector &elmax, int nel, int sdim, int n,
|
||||
bool by_max_size);
|
||||
|
||||
/** @brief Constructor for given element bounds, spatial dimension, and
|
||||
* tensor-grid divisions in each direction.
|
||||
*
|
||||
* @details This constructor must be called collectively on \a comm.
|
||||
* Supports spatial dimensions 1, 2, and 3, and accepts nel == 0 on a rank.
|
||||
* Requires nx.Size() == sdim and positive entries in nx.
|
||||
*
|
||||
* Assumes elmin, elmax Ordering::byNodes:
|
||||
* elmin -> [x_{0,min},x_{1,min},... ,y_{0,min},y_{1,min},..,z_{nel-1,min}]
|
||||
* elmax -> [x_{0,max},x_{1,max},... ,y_{0,max},y_{1,max},..,z_{nel-1,max}]
|
||||
* Note elmin, elmax can be obtained using GridFunction::GetElementBounds()
|
||||
*/
|
||||
GlobalBBoxTensorGridMap(const MPI_Comm &comm, Vector &elmin,
|
||||
Vector &elmax, int nel, int sdim, Array<int> &nx);
|
||||
|
||||
~GlobalBBoxTensorGridMap();
|
||||
|
||||
/** @brief Get list of procs corresponding to the list of points.
|
||||
*
|
||||
* @details This method must be called collectively on the communicator
|
||||
* used to construct the map. The input points can be ordered byNodes:
|
||||
* (XXX...,YYY...,ZZZ) or byVDIM: (XYZ,XYZ,...), as specified by
|
||||
* \a ordering.
|
||||
*
|
||||
* The output map contains one entry for each input point, keyed by the
|
||||
* point's local index in \a xyz. Points with no candidate ranks, including
|
||||
* points outside the global bounding box, have an empty list of candidate
|
||||
* ranks.
|
||||
*/
|
||||
void MapPointsToProcs(Vector &xyz, int ordering,
|
||||
std::map<int, std::vector<int>> &pt_to_procs) const;
|
||||
|
||||
// Some getters
|
||||
const Array<int> &GetGridMap() const { return ggrid_map; }
|
||||
const Vector &GetGridFac() const { return gmap_fac; }
|
||||
const Vector &GetGridMin() const { return gmap_bnd_min; }
|
||||
const Vector &GetGridMax() const { return gmap_bnd_max; }
|
||||
const Array<int> &GetGridN() const { return gmap_n; }
|
||||
|
||||
private:
|
||||
/// Setup the map given element bounds and number of tensor grid divisions.
|
||||
void Setup(const MPI_Comm &comm, Vector &elmin, Vector &elmax,
|
||||
int nel, Array<int> &nx);
|
||||
|
||||
/// Get global hash cell index for a given point.
|
||||
int GetGlobalGridCellFromPoint(Vector &xyz) const;
|
||||
|
||||
/** @brief Get owning proc and local index on that proc for given global
|
||||
* grid cell index. */
|
||||
void GlobalGridCellToProcAndLocalIndex(int i, int &proc, int &idx) const;
|
||||
|
||||
/// Map a point to proc and local index of the corresponding grid cell
|
||||
void GetProcAndLocalIndexFromPoint(Vector &xyz, int &proc, int &idx) const;
|
||||
|
||||
/// Given local cell index, return list of procs saved in the map
|
||||
Array<int> MapCellToProcs(int l_idx) const;
|
||||
};
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_GSLIB
|
||||
|
||||
+177
-73
@@ -11,7 +11,7 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -27,6 +27,8 @@
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
|
||||
#include <climits>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
#if GSLIB_RELEASE_VERSION >= 10009
|
||||
@@ -52,14 +54,127 @@ struct findptsElementGPT_t
|
||||
double x[DIM], jac[DIM * DIM], hes[4];
|
||||
};
|
||||
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<DIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<DIM>;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_first_der;
|
||||
using gslib::lag_eval_second_der;
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[DIM], A[DIM * DIM];
|
||||
dbl_range_t x[DIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[DIM];
|
||||
double fac[DIM];
|
||||
unsigned int *offset;
|
||||
int max;
|
||||
};
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first derivative at x.
|
||||
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2 * (x - z[j]);
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN+i] = 2.0 * lCoeff[i] * u1;
|
||||
}
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first and second derivative at x.
|
||||
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2 * (x - z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN+i] = 2.0 * lCoeff[i] * u1;
|
||||
p0[2*pN+i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
// Axis-aligned bounding box test.
|
||||
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
|
||||
const double x[2])
|
||||
{
|
||||
double test = 1;
|
||||
for (int d = 0; d < 2; ++d)
|
||||
{
|
||||
double b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
test = test < 0 ? test : b_d;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
|
||||
// Axis-aligned bounding box test followed by oriented bounding-box test.
|
||||
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
|
||||
const double x[2])
|
||||
{
|
||||
const double bxyz = AABB_test(b, x);
|
||||
if (bxyz < 0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else
|
||||
{
|
||||
double dxyz[2];
|
||||
for (int d = 0; d < 2; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
double test = 1;
|
||||
for (int d = 0; d < 2; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e = 0; e < 2; ++e)
|
||||
{
|
||||
rst += b->A[d * 2 + e] * dxyz[e];
|
||||
}
|
||||
double brst = (rst + 1) * (1 - rst);
|
||||
test = test < 0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
// Element index corresponding to hash mesh that the point is located in.
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[2])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d = 2 - 1; d >= 0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i < 0 ? 0 : (n - 1 < i ? n - 1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
/*Solve Ax=y. A is row-major */
|
||||
static MFEM_HOST_DEVICE inline void lin_solve_2(double x[2], const double A[4],
|
||||
@@ -70,6 +185,12 @@ static MFEM_HOST_DEVICE inline void lin_solve_2(double x[2], const double A[4],
|
||||
x[1] = idet*(A[0]*y[1] - A[2]*y[0]);
|
||||
}
|
||||
|
||||
/* L2 norm squared. */
|
||||
static MFEM_HOST_DEVICE inline double l2norm2(const double x[2])
|
||||
{
|
||||
return x[0] * x[0] + x[1] * x[1];
|
||||
}
|
||||
|
||||
/* the bit structure of flags is CSSRR
|
||||
the C bit --- 1<<4 --- is set when the point is converged
|
||||
RR is 0 = 00b if r is unconstrained,
|
||||
@@ -231,7 +352,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *res,
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2<2>(resid);
|
||||
const double dist2 = l2norm2(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
for (int d = 0; d < 2; ++d)
|
||||
@@ -441,7 +562,7 @@ newton_area_fin:
|
||||
int f = flags >> (2 * dd) & 3u;
|
||||
res->r[dd] = f == 0 ? r0[dd] + dr[dd] : (f == 1 ? -1 : 1);
|
||||
}
|
||||
res->flags = flags | ((p->flags & FLAG_MASK) << 5);
|
||||
res->flags = flags | (p->flags << 5);
|
||||
}
|
||||
|
||||
// Full Newton solve on the face. One of r/s/t is constrained.
|
||||
@@ -514,8 +635,7 @@ newton_edge_fin:
|
||||
res->r[de] = nr;
|
||||
res->r[dn]=p->r[dn];
|
||||
res->dist2p = -v;
|
||||
res->flags = flags | new_flags | ((p->flags & FLAG_MASK) << 5);
|
||||
#undef EVAL
|
||||
res->flags = flags | new_flags | (p->flags << 5);
|
||||
}
|
||||
|
||||
// Find closest mesh node to the sought point.
|
||||
@@ -574,26 +694,27 @@ static MFEM_HOST_DEVICE double tensor_ig2_j(double *g_partials,
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsLocal2DKernel(const int npt,
|
||||
const double tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
static void FindPointsLocal2D_Kernel(const int npt,
|
||||
const double tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
#define MAX_CONST(a, b) (((a) > (b)) ? (a) : (b))
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
const int p_NE = D1D*D1D;
|
||||
@@ -608,7 +729,7 @@ static void FindPointsLocal2DKernel(const int npt,
|
||||
// 3D1D for seed, 10D1D+6 for area, 3D1D+9 for edge
|
||||
constexpr int size1 = 10*MD1 + 6;
|
||||
constexpr int size2 = MD1*4; // edge constraints
|
||||
constexpr int size3 = MD1*MD1*DIM; // local element coordinates
|
||||
constexpr int size3 = MD1*MD1*MD1*DIM; // local element coordinates
|
||||
|
||||
MFEM_SHARED double r_workspace[size1];
|
||||
MFEM_SHARED findptsElementPoint_t el_pts[2];
|
||||
@@ -1041,9 +1162,9 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
|
||||
auto pgslm = gsl_mesh.Read();
|
||||
auto pwt = DEV.wtend.Read();
|
||||
auto pbb = DEV.bb.Read();
|
||||
auto plhm = DEV.lh_min.Read();
|
||||
auto plhf = DEV.lh_fac.Read();
|
||||
auto plho = DEV.lh_offset.ReadWrite();
|
||||
auto plhm = DEV.loc_hash_min.Read();
|
||||
auto plhf = DEV.loc_hash_fac.Read();
|
||||
auto plho = DEV.loc_hash_offset.ReadWrite();
|
||||
auto pcode = code.Write();
|
||||
auto pelem = elem.Write();
|
||||
auto pref = ref.Write();
|
||||
@@ -1054,49 +1175,32 @@ void FindPointsGSLIB::FindPointsLocal2(const Vector &point_pos,
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
FindPointsLocal2DKernel<2>(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
return FindPointsLocal2D_Kernel<2>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
case 3:
|
||||
FindPointsLocal2DKernel<3>(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
return FindPointsLocal2D_Kernel<3>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
case 4:
|
||||
FindPointsLocal2DKernel<4>(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
return FindPointsLocal2D_Kernel<4>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
case 5:
|
||||
FindPointsLocal2DKernel<5>(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
return FindPointsLocal2D_Kernel<5>(
|
||||
npt, DEV.newt_tol, pp, point_pos_ordering, pgslm, NE_split_total, pwt,
|
||||
pbb, DEV.h_nx, plhm, plhf, plho, pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
default:
|
||||
FindPointsLocal2DKernel(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc, DEV.dof1d);
|
||||
break;
|
||||
return FindPointsLocal2D_Kernel(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx,
|
||||
plhm, plhf, plho, pcode, pelem,
|
||||
pref, pdist, pgll1d, plc, DEV.dof1d);
|
||||
}
|
||||
}
|
||||
#undef DIM2
|
||||
#undef DIM
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
|
||||
+166
-38
@@ -11,7 +11,9 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
#include "../../linalg/kernels.hpp"
|
||||
|
||||
#include <climits>
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -57,15 +59,128 @@ struct findptsElemPt
|
||||
double x[DIM], jac[DIM * DIM], hes[18];
|
||||
};
|
||||
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<DIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<DIM>;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_first_der;
|
||||
using gslib::lag_eval_second_der;
|
||||
using gslib::lin_solve_sym_2;
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[DIM], A[DIM * DIM];
|
||||
dbl_range_t x[DIM];
|
||||
};
|
||||
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[DIM];
|
||||
double fac[DIM];
|
||||
unsigned int *offset;
|
||||
// int max;
|
||||
};
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first derivative at x.
|
||||
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2*(x-z[j]);
|
||||
u1 = d_j*u1+u0;
|
||||
u0 = d_j*u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i]*u0;
|
||||
p0[pN+i] = 2.0*lCoeff[i]*u1;
|
||||
}
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first and second derivative at x.
|
||||
// Note: lCoeff stores pre-computed coefficients for fast evaluation.
|
||||
static MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
double d_j = 2*(x-z[j]);
|
||||
u2 = d_j*u2+u1;
|
||||
u1 = d_j*u1+u0;
|
||||
u0 = d_j*u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i]*u0;
|
||||
p0[pN+i] = 2.0*lCoeff[i]*u1;
|
||||
p0[2*pN+i] = 8.0*lCoeff[i]*u2;
|
||||
}
|
||||
|
||||
// Axis-aligned bounding box test.
|
||||
static MFEM_HOST_DEVICE inline double AABB_test(const obbox_t *const b,
|
||||
const double x[3])
|
||||
{
|
||||
double b_d;
|
||||
for (int d = 0; d < 3; ++d)
|
||||
{
|
||||
b_d = (x[d]-b->x[d].min)*(b->x[d].max-x[d]);
|
||||
if (b_d < 0) { return b_d; }
|
||||
}
|
||||
return b_d;
|
||||
}
|
||||
|
||||
// Axis-aligned bounding box test followed by oriented bounding-box test.
|
||||
static MFEM_HOST_DEVICE inline double bbox_test(const obbox_t *const b,
|
||||
const double x[3])
|
||||
{
|
||||
const double bxyz = AABB_test(b, x);
|
||||
if (bxyz < 0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
else
|
||||
{
|
||||
double dxyz[3];
|
||||
for (int d = 0; d < 3; ++d)
|
||||
{
|
||||
dxyz[d] = x[d]-b->c0[d];
|
||||
}
|
||||
double test = 1;
|
||||
for (int d = 0; d < 3; ++d)
|
||||
{
|
||||
double rst = 0;
|
||||
for (int e = 0; e < 3; ++e)
|
||||
{
|
||||
rst += b->A[d*3+e]*dxyz[e];
|
||||
}
|
||||
double brst = (rst+1)*(1-rst);
|
||||
test = test < 0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
}
|
||||
|
||||
// Element index corresponding to hash mesh that the point is located in.
|
||||
static MFEM_HOST_DEVICE inline int hash_index(const findptsLocalHashData_t *p,
|
||||
const double x[3])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d = 3-1; d >= 0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
int i = (int)floor((x[d]-p->bnd[d].min)*p->fac[d]);
|
||||
sum += i < 0 ? 0 : (n-1 < i ? n-1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
// Solve Ax=y. A is row-major.
|
||||
static MFEM_HOST_DEVICE inline void lin_solve_3(double x[3], const double A[9],
|
||||
@@ -84,6 +199,22 @@ static MFEM_HOST_DEVICE inline void lin_solve_3(double x[3], const double A[9],
|
||||
x[2] = idet*(inv6*y[0]+inv7*y[1]+inv8*y[2]);
|
||||
}
|
||||
|
||||
// Solve Ax=y. A is a symmetric 2x2 matrix.
|
||||
static MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
|
||||
const double A[3],
|
||||
const double y[2])
|
||||
{
|
||||
const double idet = 1 / (A[0]*A[2]-A[1]*A[1]);
|
||||
x[0] = idet*(A[2]*y[0]-A[1]*y[1]);
|
||||
x[1] = idet*(A[0]*y[1]-A[1]*y[0]);
|
||||
}
|
||||
|
||||
// L2 norm.
|
||||
static MFEM_HOST_DEVICE inline double l2norm2(const double x[3])
|
||||
{
|
||||
return x[0]*x[0]+x[1]*x[1]+x[2]*x[2];
|
||||
}
|
||||
|
||||
/* the bit structure of flags is CTTSSRR
|
||||
the C bit --- 1<<6 --- is set when the point is converged
|
||||
RR is 0 = 00b if r is unconstrained,
|
||||
@@ -328,7 +459,7 @@ static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsPt *res,
|
||||
const findptsPt *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2<3>(resid);
|
||||
const double dist2 = l2norm2(resid);
|
||||
const double decr = p->dist2-dist2;
|
||||
const double pred = p->dist2p;
|
||||
for (int d = 0; d < 3; ++d)
|
||||
@@ -575,7 +706,7 @@ newton_vol_fin:
|
||||
int f = flags >> (2*dd) & 3u;
|
||||
res->r[dd] = f == 0 ? r0[dd]+dr[dd] : (f == 1 ? -1 : 1);
|
||||
}
|
||||
res->flags = flags | ((p->flags & FLAG_MASK) << 7);
|
||||
res->flags = flags | (p->flags << 7);
|
||||
}
|
||||
|
||||
// Full Newton solve on the face. One of r/s/t is constrained.
|
||||
@@ -758,7 +889,7 @@ newton_face_fin:
|
||||
res->r[dn] = p->r[dn];
|
||||
res->r[d1] = r[0];
|
||||
res->r[d2] = r[1];
|
||||
res->flags = new_flags | ((p->flags & FLAG_MASK) << 7);
|
||||
res->flags = new_flags | (p->flags << 7);
|
||||
}
|
||||
|
||||
// Full Newton solve on the edge. Two of r/s/t are constrained.
|
||||
@@ -842,8 +973,7 @@ newton_edge_fin:
|
||||
res->r[dn1] = p->r[dn1];
|
||||
res->r[dn2] = p->r[dn2];
|
||||
res->dist2p = -v;
|
||||
res->flags = flags | new_flags | ((p->flags & FLAG_MASK) << 7);
|
||||
#undef EVAL
|
||||
res->flags = flags | new_flags | (p->flags << 7);
|
||||
}
|
||||
|
||||
// Find closest mesh node to the sought point.
|
||||
@@ -1122,6 +1252,7 @@ static void FindPointsLocal3DKernel(const int npt,
|
||||
case 0: // findpt_vol
|
||||
{
|
||||
double *wtr = r_workspace_ptr;
|
||||
|
||||
double *resid = wtr+6*D1D;
|
||||
double *jac = resid+3;
|
||||
double *resid_temp = jac+9;
|
||||
@@ -1372,7 +1503,7 @@ static void FindPointsLocal3DKernel(const int npt,
|
||||
// Hes_T is transposed version (i.e. in col major)
|
||||
// n1*[2, 1, 1, 0, 0]
|
||||
// j==1 => wt_j = wt+n1
|
||||
double *wt_j = wt+D1D*(2 - (row+1)/2);
|
||||
double *wt_j = wt+D1D*(2-(row+1) / 2);
|
||||
const double *x = e_x[row+1][d];
|
||||
hes_T[j] = 0.0;
|
||||
for (int k = 0; k < D1D; ++k)
|
||||
@@ -1391,6 +1522,7 @@ static void FindPointsLocal3DKernel(const int npt,
|
||||
hes[j] += resid[d]*hes_T[j*3+d];
|
||||
}
|
||||
}
|
||||
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(l,x,1)
|
||||
@@ -1648,7 +1780,6 @@ static void FindPointsLocal3DKernel(const int npt,
|
||||
} //findpts_local
|
||||
} //elp
|
||||
});
|
||||
#undef MAXC
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
|
||||
@@ -1665,9 +1796,9 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
|
||||
auto pgslm = gsl_mesh.Read();
|
||||
auto pwt = DEV.wtend.Read();
|
||||
auto pbb = DEV.bb.Read();
|
||||
auto plhm = DEV.lh_min.Read();
|
||||
auto plhf = DEV.lh_fac.Read();
|
||||
auto plho = DEV.lh_offset.ReadWrite();
|
||||
auto plhm = DEV.loc_hash_min.Read();
|
||||
auto plhf = DEV.loc_hash_fac.Read();
|
||||
auto plho = DEV.loc_hash_offset.ReadWrite();
|
||||
auto pcode = code.Write();
|
||||
auto pelem = elem.Write();
|
||||
auto pref = ref.Write();
|
||||
@@ -1678,36 +1809,33 @@ void FindPointsGSLIB::FindPointsLocal3(const Vector &point_pos,
|
||||
{
|
||||
case 2:
|
||||
FindPointsLocal3DKernel<2>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
break;
|
||||
case 3:
|
||||
FindPointsLocal3DKernel<3>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
break;
|
||||
case 4:
|
||||
FindPointsLocal3DKernel<4>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
break;
|
||||
case 5:
|
||||
FindPointsLocal3DKernel<5>(npt, DEV.newt_tol, pp, point_pos_ordering,
|
||||
pgslm, NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc);
|
||||
pgslm, NE_split_total, pwt, pbb, DEV.h_nx, plhm,
|
||||
plhf, plho, pcode, pelem, pref, pdist, pgll1d,
|
||||
plc);
|
||||
break;
|
||||
default:
|
||||
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist, pgll1d, plc,
|
||||
FindPointsLocal3DKernel(npt, DEV.newt_tol, pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, DEV.h_nx, plhm, plhf,
|
||||
plho, pcode, pelem, pref, pdist, pgll1d, plc,
|
||||
DEV.dof1d);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef pMax
|
||||
|
||||
@@ -1,656 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-function"
|
||||
#endif
|
||||
#include "gslib.h"
|
||||
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
|
||||
#define GSLIB_RELEASE_VERSION 10007
|
||||
#endif
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
#if GSLIB_RELEASE_VERSION >= 10009
|
||||
#define CODE_INTERNAL 0
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
#define sDIM 2
|
||||
#define sDIM2 4
|
||||
#define rDIM 1
|
||||
|
||||
struct findptsElementPoint_t
|
||||
{
|
||||
double x[sDIM], r, oldr, dist2, dist2p, tr;
|
||||
int flags;
|
||||
};
|
||||
|
||||
struct findptsElementGEdge_t
|
||||
{
|
||||
double *x[sDIM];
|
||||
};
|
||||
|
||||
struct findptsElementGPT_t
|
||||
{
|
||||
double x[sDIM], jac[sDIM*rDIM], hes[sDIM*rDIM];
|
||||
};
|
||||
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<sDIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
|
||||
using gslib::AABB_test;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_second_der;
|
||||
|
||||
/* the bit structure of flags is CRR
|
||||
the C bit --- 1<<2 --- is set when the point is converged
|
||||
RR is 0 = 00b if r is unconstrained,
|
||||
1 = 01b if r is constrained at -1, i.e., rmin
|
||||
2 = 10b if r is constrained at +1, i.e., rmax
|
||||
*/
|
||||
|
||||
#define CONVERGED_FLAG (1u<<2)
|
||||
#define FLAG_MASK 0x07u // = 111b
|
||||
|
||||
/* returns 1 if r direction (the only free direction in 2D) is constrained.
|
||||
returns 1 if either 1st or 2nd bit of flags is set.
|
||||
*/
|
||||
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
|
||||
{
|
||||
return ((flags | flags>>1) & 1u);
|
||||
}
|
||||
|
||||
/* pi=0, r=-1; pi=1, r=+1 */
|
||||
static MFEM_HOST_DEVICE inline int point_index(const int x)
|
||||
{
|
||||
return ((x>>1) & 1u);
|
||||
}
|
||||
|
||||
/* check reduction in objective against prediction, and adjust
|
||||
trust region radius (p->tr) accordingly;
|
||||
may reject the prior step, returning 1; otherwise returns 0
|
||||
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
|
||||
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
|
||||
const double resid[2],
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2<2>(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
out_pt->x[0] = p->x[0];
|
||||
out_pt->x[1] = p->x[1];
|
||||
out_pt->oldr = p->r;
|
||||
out_pt->dist2 = dist2;
|
||||
if (decr >= 0.01*pred)
|
||||
{
|
||||
if (decr >= 0.9*pred) // very good iteration
|
||||
{
|
||||
out_pt->tr = p->tr*2;
|
||||
}
|
||||
else // somewhat good iteration
|
||||
{
|
||||
out_pt->tr = p->tr;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
else
|
||||
{
|
||||
/* reject step; note: the point will pass through this routine
|
||||
again, and we set things up here so it gets classed as a
|
||||
"very good iteration" --- this doubles the trust radius,
|
||||
which is why we divide by 4 below */
|
||||
double v0 = fabs(p->r - p->oldr);
|
||||
out_pt->tr = v0/4.0;
|
||||
out_pt->dist2 = p->dist2;
|
||||
out_pt->r = p->oldr;
|
||||
out_pt->flags = p->flags>>3;
|
||||
out_pt->dist2p = -HUGE_VAL;
|
||||
if (pred < dist2*tol)
|
||||
{
|
||||
out_pt->flags |= CONVERGED_FLAG;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline void newton_edge( findptsElementPoint_t *const
|
||||
out_pt,
|
||||
const double jac[2],
|
||||
const double rhess,
|
||||
const double resid[2],
|
||||
int flags,
|
||||
const findptsElementPoint_t *const p,
|
||||
const double tol )
|
||||
{
|
||||
const double tr = p->tr;
|
||||
const double A = jac[0] * jac[0] + jac[1] * jac[1] -
|
||||
rhess; // A = J^T J - resid_d H_d
|
||||
const double y = jac[0]*resid[0] + jac[1]*resid[1]; // y = J^T resid
|
||||
|
||||
const double oldr = p->r;
|
||||
double dr, newr, tdr, tnewr, v, tv;
|
||||
int new_flags=0, tnew_flags=0;
|
||||
|
||||
#define EVAL(dr) ( (dr*A - 2*y) * dr )
|
||||
if (A>0)
|
||||
{
|
||||
dr = y/A;
|
||||
if (fabs(dr)<tol)
|
||||
{
|
||||
dr=0.0;
|
||||
newr = oldr;
|
||||
}
|
||||
else
|
||||
{
|
||||
newr = oldr+dr;
|
||||
}
|
||||
|
||||
if (fabs(dr)<tr && fabs(newr)<1)
|
||||
{
|
||||
v = EVAL(dr);
|
||||
goto newton_edge_fin;
|
||||
}
|
||||
}
|
||||
|
||||
if ((newr=oldr-tr) > -1)
|
||||
{
|
||||
dr = -tr;
|
||||
}
|
||||
else
|
||||
{
|
||||
newr = -1, dr = -1-oldr, new_flags = flags|1u;
|
||||
}
|
||||
v = EVAL(dr);
|
||||
|
||||
if ((tnewr=oldr+tr) < 1)
|
||||
{
|
||||
tdr = tr;
|
||||
}
|
||||
else
|
||||
{
|
||||
tnewr = 1, tdr = 1-oldr, tnew_flags = flags|2u;
|
||||
}
|
||||
tv = EVAL(tdr);
|
||||
|
||||
if (tv<v)
|
||||
{
|
||||
newr = tnewr, dr = tdr, v = tv, new_flags = tnew_flags;
|
||||
}
|
||||
#undef EVAL
|
||||
|
||||
newton_edge_fin:
|
||||
// check convergence by testing if change in r is less than tol
|
||||
if (fabs(dr)<tol)
|
||||
{
|
||||
new_flags |= CONVERGED_FLAG;
|
||||
}
|
||||
out_pt->r = newr;
|
||||
out_pt->dist2p = -v;
|
||||
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE void seed_j( const double *elx[sDIM],
|
||||
const double x[sDIM],
|
||||
const double *z,
|
||||
double *dist2,
|
||||
double *r,
|
||||
const int ir,
|
||||
const int pN )
|
||||
{
|
||||
double dx[sDIM];
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dx[d] = x[d] - elx[d][ir];
|
||||
}
|
||||
dist2[ir] = HUGE_VAL;
|
||||
const double dist2_rs = l2norm2(dx);
|
||||
if (dist2[ir]>dist2_rs)
|
||||
{
|
||||
dist2[ir] = dist2_rs;
|
||||
r[ir] = z[ir];
|
||||
}
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsEdgeLocal2DKernel( const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const bool obb_check,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0 )
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
const int p_NEL = nel*D1D;
|
||||
MFEM_VERIFY(MD1<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(D1D!=0, "Polynomial order not specified.");
|
||||
const int nThreads = D1D*sDIM;
|
||||
|
||||
mfem::forall_2D(npt, nThreads, 1, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
// 2D1D for seed, 3D1D + 7 for edge
|
||||
constexpr int size1 = 3*MD1 + 7;
|
||||
// edge coordinates = D1D*2
|
||||
constexpr int size2 = 2*MD1;
|
||||
// local element coordinates in shared memory
|
||||
constexpr int size3 = MD1*sDIM;
|
||||
|
||||
MFEM_SHARED findptsElementPoint_t el_pts[2];
|
||||
MFEM_SHARED double r_workspace[size1];
|
||||
|
||||
MFEM_SHARED double constraint_workspace[size2];
|
||||
|
||||
MFEM_SHARED double elem_coords[MD1 <= 6 ? size3 : 1];
|
||||
|
||||
double *r_workspace_ptr = r_workspace;
|
||||
findptsElementPoint_t *fpt, *tmp;
|
||||
fpt = el_pts + 0;
|
||||
tmp = el_pts + 1;
|
||||
|
||||
// x and y coord index within point_pos for point i
|
||||
int id_x = point_pos_ordering == 0 ? i : i*sDIM;
|
||||
int id_y = point_pos_ordering == 0 ? i+npt : i*sDIM+1;
|
||||
double x_i[2] = {x[id_x], x[id_y]};
|
||||
|
||||
unsigned int *code_i = code_base + i;
|
||||
double *dist2_i = dist2_base + i;
|
||||
|
||||
//---------------- map_points_to_els --------------------
|
||||
findptsLocalHashData_t hash;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
hash.bnd[d].min = hashMin[d];
|
||||
hash.fac[d] = hashFac[d];
|
||||
}
|
||||
hash.hash_n = hash_n;
|
||||
hash.offset = hashOffset;
|
||||
|
||||
const int hi = hash_index(&hash, x_i);
|
||||
const unsigned int *elp = hash.offset + hash.offset[hi];
|
||||
const unsigned int *const ele = hash.offset + hash.offset[hi+1];
|
||||
*code_i = CODE_NOT_FOUND;
|
||||
*dist2_i = HUGE_VAL;
|
||||
|
||||
for (; elp!=ele; ++elp)
|
||||
{
|
||||
const unsigned int el = *elp;
|
||||
|
||||
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
|
||||
bool pass_bb = true;
|
||||
obbox_t box;
|
||||
if (obb_check)
|
||||
{
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
}
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
pass_bb = (bbox_test(&box, x_i) >= 0);
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int d = 0; d < sDIM; ++d)
|
||||
{
|
||||
box.x[d].min = boxinfo[n_box_ents*el + d];
|
||||
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
|
||||
}
|
||||
pass_bb = (AABB_test(&box, x_i) >= 0);
|
||||
}
|
||||
|
||||
if (pass_bb)
|
||||
{
|
||||
//------------ findpts_local ------------------
|
||||
{
|
||||
if (MD1 <= 6)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
|
||||
{
|
||||
const int qp = j % D1D;
|
||||
const int d = j / D1D;
|
||||
elem_coords[qp + d*D1D] =
|
||||
xElemCoord[qp + el*D1D + d*p_NEL];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
const double *elx[sDIM];
|
||||
for (int d=0; d<sDIM; d++)
|
||||
{
|
||||
elx[d] = MD1<= 6 ? &elem_coords[d*D1D] :
|
||||
xElemCoord + d*p_NEL + el*D1D;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
//// findpts_el ////
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
fpt->dist2 = HUGE_VAL;
|
||||
fpt->dist2p = 0;
|
||||
fpt->tr = 1;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
fpt->x[j] = x_i[j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
{
|
||||
double *dist2_temp = r_workspace_ptr;
|
||||
double *r_temp = dist2_temp + D1D;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
seed_j(elx, x_i, gll1D, dist2_temp, r_temp, j, D1D);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
for (int ir=0; ir<D1D; ++ir)
|
||||
{
|
||||
if (dist2_temp[ir]<fpt->dist2)
|
||||
{
|
||||
fpt->dist2 = dist2_temp[ir];
|
||||
fpt->r = r_temp[ir];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
} //seed done
|
||||
|
||||
// Initialize tmp struct with fpt values before starting Newton iterations
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
tmp->dist2 = HUGE_VAL;
|
||||
tmp->dist2p = 0;
|
||||
tmp->tr = 1;
|
||||
tmp->flags = 0;
|
||||
tmp->r = fpt->r;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
tmp->x[j] = fpt->x[j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
|
||||
for (int step=0; step<50; step++)
|
||||
{
|
||||
int nc = num_constrained(tmp->flags & FLAG_MASK);
|
||||
switch (nc)
|
||||
{
|
||||
case 0:
|
||||
{
|
||||
double *wt = r_workspace_ptr;
|
||||
double *resid = wt + 3*D1D;
|
||||
double *jac = resid + sDIM;
|
||||
double *hess = jac + sDIM*rDIM;
|
||||
|
||||
findptsElementGEdge_t edge;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d][j] = elx[d][j];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
// compute basis function info upto 2nd derivative
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
lag_eval_second_der(wt, tmp->r, j, gll1D,
|
||||
lagcoeff, D1D);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
resid[j] = tmp->x[j];
|
||||
jac[j] = 0.0;
|
||||
hess[j] = 0.0;
|
||||
for (int k=0; k<D1D; ++k)
|
||||
{
|
||||
resid[j] -= wt[ k]*edge.x[j][k];
|
||||
jac[j] += wt[D1D+k]*edge.x[j][k];
|
||||
hess[j] += wt[2*D1D+k]*edge.x[j][k];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
hess[2] = resid[0]*hess[0] + resid[1]*hess[1];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
if (!reject_prior_step_q(fpt, resid, tmp, tol))
|
||||
{
|
||||
newton_edge(fpt, jac, hess[2], resid,
|
||||
tmp->flags & FLAG_MASK, tmp, tol);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
break;
|
||||
}
|
||||
case 1: // r is constrained to either -1 or 1
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
const int pi = point_index(tmp->flags &
|
||||
FLAG_MASK);
|
||||
const double *wt = wtend + pi*3*D1D;
|
||||
findptsElementGPT_t gpt;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
gpt.x[d] = elx[d][pi*(D1D-1)];
|
||||
gpt.jac[d] = 0.0;
|
||||
gpt.hes[d] = 0.0;
|
||||
for (int k=0; k<D1D; ++k)
|
||||
{
|
||||
gpt.jac[d] += wt[D1D +k]*elx[d][k];
|
||||
gpt.hes[d] += wt[2*D1D+k]*elx[d][k];
|
||||
}
|
||||
}
|
||||
|
||||
const double *const pt_x = gpt.x;
|
||||
const double *const jac = gpt.jac;
|
||||
const double *const hes = gpt.hes;
|
||||
double resid[sDIM], steep, sr;
|
||||
resid[0] = fpt->x[0] - pt_x[0];
|
||||
resid[1] = fpt->x[1] - pt_x[1];
|
||||
steep = jac[0]*resid[0] + jac[1]*resid[1];
|
||||
sr = steep*tmp->r;
|
||||
if ( !reject_prior_step_q(fpt, resid, tmp, tol) )
|
||||
{
|
||||
if (sr<0)
|
||||
{
|
||||
const double rhess = resid[0]*hes[0] +
|
||||
resid[1]*hes[1];
|
||||
newton_edge(fpt, jac, rhess,
|
||||
resid, 0, tmp, tol);
|
||||
}
|
||||
else // sr==0
|
||||
{
|
||||
fpt->r = tmp->r;
|
||||
fpt->dist2p = 0;
|
||||
fpt->flags = tmp->flags | CONVERGED_FLAG;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
break;
|
||||
} // case 1
|
||||
} //switch
|
||||
if (fpt->flags & CONVERGED_FLAG)
|
||||
{
|
||||
break;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
*tmp = *fpt;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
} //for int step<50
|
||||
} //findpts_el
|
||||
|
||||
bool converged_internal =
|
||||
((fpt->flags&FLAG_MASK) == CONVERGED_FLAG) &&
|
||||
(fpt->dist2<dist2tol);
|
||||
|
||||
if (*code_i == CODE_NOT_FOUND || converged_internal ||
|
||||
fpt->dist2 < *dist2_i)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
*(el_base+i) = el;
|
||||
*code_i = converged_internal ? CODE_INTERNAL : CODE_BORDER;
|
||||
*dist2_i = fpt->dist2;
|
||||
*(r_base+i) = fpt->r;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
if (converged_internal)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
} //findpts_local
|
||||
} //obbox_test
|
||||
} //elp
|
||||
});
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &code,
|
||||
Array<unsigned int> &elem,
|
||||
Vector &ref,
|
||||
Vector &dist,
|
||||
int npt )
|
||||
{
|
||||
if (npt==0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
MFEM_VERIFY(dim==1 && spacedim==2,"Function for 2D edges only");
|
||||
bool use_dev = point_pos.UseDevice();
|
||||
auto pp = point_pos.Read(use_dev);
|
||||
auto pgslm = gsl_mesh.Read(use_dev);
|
||||
auto pwt = DEV.wtend.Read(use_dev);
|
||||
auto pbb = DEV.bb.Read(use_dev);
|
||||
auto plhm = DEV.lh_min.Read(use_dev);
|
||||
auto plhf = DEV.lh_fac.Read(use_dev);
|
||||
auto plho = DEV.lh_offset.ReadWrite(use_dev);
|
||||
auto pcode = code.Write(use_dev);
|
||||
auto pelem = elem.Write(use_dev);
|
||||
auto pref = ref.Write(use_dev);
|
||||
auto pdist = dist.Write(use_dev);
|
||||
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
|
||||
auto plc = DEV.lagcoeff.Read(use_dev);
|
||||
double dist2tol = DEV.surf_dist_tol;
|
||||
const bool obb_chk = obb_check;
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
FindPointsEdgeLocal2DKernel<2>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 3:
|
||||
FindPointsEdgeLocal2DKernel<3>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 4:
|
||||
FindPointsEdgeLocal2DKernel<4>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
default:
|
||||
FindPointsEdgeLocal2DKernel(npt, DEV.newt_tol, dist2tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc, DEV.dof1d);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef sDIM
|
||||
#undef rDIM
|
||||
#undef sDIM2
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
#else
|
||||
void FindPointsGSLIB::FindPointsEdgeLocal2( const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &code,
|
||||
Array<unsigned int> &elem,
|
||||
Vector &ref,
|
||||
Vector &dist,
|
||||
int npt ) {} ;
|
||||
#endif
|
||||
} // namespace mfem
|
||||
|
||||
#endif //ifdef MFEM_USE_GSLIB
|
||||
@@ -1,661 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-function"
|
||||
#endif
|
||||
#include "gslib.h"
|
||||
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
|
||||
#define GSLIB_RELEASE_VERSION 10007
|
||||
#endif
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
#if GSLIB_RELEASE_VERSION >= 10009
|
||||
#define CODE_INTERNAL 0
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
#define sDIM 3
|
||||
#define rDIM 1
|
||||
#define sDIM2 (sDIM*sDIM)
|
||||
#define rDIM2 (rDIM*rDIM)
|
||||
|
||||
struct findptsElementPoint_t
|
||||
{
|
||||
double x[sDIM], r, oldr, dist2, dist2p, tr;
|
||||
int flags;
|
||||
};
|
||||
|
||||
struct findptsElementGEdge_t
|
||||
{
|
||||
double *x[sDIM], *dxdn[sDIM], *d2xdn[sDIM];
|
||||
};
|
||||
|
||||
struct findptsElementGPT_t
|
||||
{
|
||||
double x[sDIM], jac[sDIM], hes[sDIM*(1+1)];
|
||||
};
|
||||
|
||||
using dbl_range_t = gslib::dbl_range_t;
|
||||
using obbox_t = gslib::obbox_t<sDIM>;
|
||||
using findptsLocalHashData_t = gslib::findptsLocalHashData_t<sDIM>;
|
||||
using gslib::AABB_test;
|
||||
using gslib::bbox_test;
|
||||
using gslib::hash_index;
|
||||
using gslib::l2norm2;
|
||||
using gslib::lag_eval_second_der;
|
||||
|
||||
/* the bit structure of flags is CRR
|
||||
the C bit --- 1<<2 --- is set when the point is converged
|
||||
RR is 0 = 00b if r is unconstrained,
|
||||
1 = 01b if r is constrained at -1, i.e., rmin
|
||||
2 = 10b if r is constrained at +1, i.e., rmax
|
||||
*/
|
||||
#define CONVERGED_FLAG (1u<<2)
|
||||
#define FLAG_MASK 0x07u
|
||||
|
||||
/* returns the number of constrained reference coordinates, max 1
|
||||
*/
|
||||
static MFEM_HOST_DEVICE inline int num_constrained(const int flags)
|
||||
{
|
||||
return ((flags | flags>>1) & 1u);
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline int point_index(const int x)
|
||||
{
|
||||
return ((x>>1)&1u);
|
||||
}
|
||||
|
||||
/* check reduction in objective against prediction, and adjust
|
||||
trust region radius (p->tr) accordingly;
|
||||
may reject the prior step, returning 1; otherwise returns 0
|
||||
sets out_pt->dist2, out_pt->index, out_pt->x, out_pt->oldr in any event,
|
||||
leaving out_pt->r, out_pt->dr, out_pt->flags to be set when returning 0 */
|
||||
static MFEM_HOST_DEVICE bool reject_prior_step_q(findptsElementPoint_t *out_pt,
|
||||
const double resid[3],
|
||||
const findptsElementPoint_t *p,
|
||||
const double tol)
|
||||
{
|
||||
const double dist2 = l2norm2<sDIM>(resid);
|
||||
const double decr = p->dist2 - dist2;
|
||||
const double pred = p->dist2p;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
out_pt->x[d] = p->x[d];
|
||||
}
|
||||
out_pt->oldr = p->r;
|
||||
out_pt->dist2 = dist2;
|
||||
if (decr>=0.01*pred)
|
||||
{
|
||||
if (decr>=0.9*pred) // very good iteration
|
||||
{
|
||||
out_pt->tr = 2*p->tr;
|
||||
}
|
||||
else // good iteration
|
||||
{
|
||||
out_pt->tr = p->tr;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
else // if the iteration in not good
|
||||
{
|
||||
/* reject step; note: the point will pass through this routine
|
||||
again, and we set things up here so it gets classed as a
|
||||
"very good iteration" --- this doubles the trust radius,
|
||||
which is why we divide by 4 below */
|
||||
double v0 = fabs(p->r - p->oldr);
|
||||
out_pt->tr = v0/4.0;
|
||||
out_pt->dist2 = p->dist2;
|
||||
out_pt->r = p->oldr;
|
||||
out_pt->flags = p->flags>>3;
|
||||
out_pt->dist2p = -HUGE_VAL;
|
||||
if (pred<dist2*tol)
|
||||
{
|
||||
out_pt->flags |= CONVERGED_FLAG;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE inline void newton_edge(findptsElementPoint_t *const
|
||||
out_pt,
|
||||
const double jac[sDIM*rDIM],
|
||||
const double rhes,
|
||||
const double resid[sDIM],
|
||||
int flags,
|
||||
const findptsElementPoint_t *const p,
|
||||
const double tol)
|
||||
{
|
||||
const double tr = p->tr;
|
||||
/* A = J^T J - resid_d H_d */
|
||||
const double A = jac[0]*jac[0]+ jac[1] * jac[1] + jac[2] * jac[2]
|
||||
- rhes;
|
||||
/* y = J^T r */
|
||||
const double y = jac[0]*resid[0] + jac[1]*resid[1] + jac[0+2]*resid[2];
|
||||
|
||||
const double oldr = p->r;
|
||||
double dr, nr, tdr, tnr;
|
||||
double v, tv;
|
||||
int new_flags = 0, tnew_flags = 0;
|
||||
|
||||
#define EVAL(dr) (dr*A - 2*y)*dr
|
||||
|
||||
/* if A is not SPD, quadratic model has no minimum */
|
||||
if (A>0)
|
||||
{
|
||||
dr = y/A;
|
||||
|
||||
if (fabs(dr)<tol)
|
||||
{
|
||||
dr=0.0;
|
||||
nr = oldr;
|
||||
}
|
||||
else
|
||||
{
|
||||
nr = oldr+dr;
|
||||
}
|
||||
if ( fabs(dr)<tr && fabs(nr)<1 )
|
||||
{
|
||||
v = EVAL(dr);
|
||||
goto newton_edge_fin;
|
||||
}
|
||||
}
|
||||
|
||||
if ( (nr=oldr-tr)>-1 )
|
||||
{
|
||||
dr = -tr;
|
||||
}
|
||||
else
|
||||
{
|
||||
nr = -1, dr = -1-oldr, new_flags = flags | 1u;
|
||||
}
|
||||
v = EVAL(dr);
|
||||
|
||||
if ( (tnr = oldr+tr)<1 )
|
||||
{
|
||||
tdr = tr;
|
||||
}
|
||||
else
|
||||
{
|
||||
tnr = 1, tdr = 1-oldr, tnew_flags = flags | 2u;
|
||||
}
|
||||
tv = EVAL(tdr);
|
||||
|
||||
if (tv<v)
|
||||
{
|
||||
nr = tnr, dr = tdr, v = tv, new_flags = tnew_flags;
|
||||
}
|
||||
|
||||
newton_edge_fin:
|
||||
/* check convergence */
|
||||
if ( fabs(dr)<tol )
|
||||
{
|
||||
new_flags |= CONVERGED_FLAG;
|
||||
}
|
||||
out_pt->r = nr;
|
||||
out_pt->dist2p = -v;
|
||||
out_pt->flags = flags | new_flags | ((p->flags & FLAG_MASK)<<3);
|
||||
#undef EVAL
|
||||
}
|
||||
|
||||
static MFEM_HOST_DEVICE void seed_j(const double *elx[sDIM],
|
||||
const double x[sDIM],
|
||||
const double *z,
|
||||
double *dist2,
|
||||
double *r,
|
||||
const int ir,
|
||||
const int pN)
|
||||
{
|
||||
if (ir>=pN)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
double dx[sDIM];
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
dx[d] = x[d] - elx[d][ir];
|
||||
}
|
||||
dist2[ir] = l2norm2(dx);
|
||||
r[ir] = z[ir];
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void FindPointsEdgeLocal3DKernel(const int npt,
|
||||
const double tol,
|
||||
const double dist2tol,
|
||||
const double *x,
|
||||
const int point_pos_ordering,
|
||||
const double *xElemCoord,
|
||||
const int nel,
|
||||
const double *wtend,
|
||||
const double *boxinfo,
|
||||
const bool obb_check,
|
||||
const int hash_n,
|
||||
const double *hashMin,
|
||||
const double *hashFac,
|
||||
unsigned int *hashOffset,
|
||||
unsigned int *const code_base,
|
||||
unsigned int *const el_base,
|
||||
double *const r_base,
|
||||
double *const dist2_base,
|
||||
const double *gll1D,
|
||||
const double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
const int p_NEL = nel*D1D;
|
||||
MFEM_VERIFY(MD1<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(D1D!=0, "Polynomial order not specified.");
|
||||
const int nThreads = D1D*sDIM;
|
||||
|
||||
mfem::forall_2D(npt, nThreads, 1, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
constexpr int size1 = 3*MD1 + 13;
|
||||
constexpr int size2 = 3*MD1;
|
||||
constexpr int size3 = MD1*sDIM;
|
||||
|
||||
MFEM_SHARED findptsElementPoint_t el_pts[2];
|
||||
MFEM_SHARED double r_workspace[size1];
|
||||
|
||||
MFEM_SHARED double constraint_workspace[size2];
|
||||
|
||||
MFEM_SHARED double elem_coords[MD1 <= 6 ? size3 : 1];
|
||||
|
||||
double *r_workspace_ptr = r_workspace;
|
||||
findptsElementPoint_t *fpt, *tmp;
|
||||
fpt = el_pts + 0;
|
||||
tmp = el_pts + 1;
|
||||
|
||||
int id_x = point_pos_ordering==0 ? i : i*sDIM;
|
||||
int id_y = point_pos_ordering==0 ? npt+i : 1+i*sDIM;
|
||||
int id_z = point_pos_ordering==0 ? 2*npt+i : 2+i*sDIM;
|
||||
double x_i[3] = {x[id_x], x[id_y], x[id_z]};
|
||||
|
||||
unsigned int *code_i = code_base + i;
|
||||
double *dist2_i = dist2_base + i;
|
||||
|
||||
//// map_points_to_els ////
|
||||
findptsLocalHashData_t hash;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
hash.bnd[d].min = hashMin[d];
|
||||
hash.fac[d] = hashFac[d];
|
||||
}
|
||||
hash.hash_n = hash_n;
|
||||
hash.offset = hashOffset;
|
||||
|
||||
const unsigned int hi = hash_index(&hash, x_i);
|
||||
const unsigned int *elp = hash.offset + hash.offset[hi];
|
||||
const unsigned int *const ele = hash.offset + hash.offset[hi+1];
|
||||
*code_i = CODE_NOT_FOUND;
|
||||
*dist2_i = HUGE_VAL;
|
||||
|
||||
for (; elp!=ele; ++elp)
|
||||
{
|
||||
const unsigned int el = *elp;
|
||||
|
||||
const int n_box_ents = obb_check ? (3*sDIM + sDIM2) : (2*sDIM);
|
||||
bool pass_bb = true;
|
||||
obbox_t box;
|
||||
if (obb_check)
|
||||
{
|
||||
for (int idx = 0; idx < sDIM; ++idx)
|
||||
{
|
||||
box.c0[idx] = boxinfo[n_box_ents*el + idx];
|
||||
box.x[idx].min = boxinfo[n_box_ents*el + sDIM + idx];
|
||||
box.x[idx].max = boxinfo[n_box_ents*el + 2*sDIM + idx];
|
||||
}
|
||||
for (int idx = 0; idx < sDIM2; ++idx)
|
||||
{
|
||||
box.A[idx] = boxinfo[n_box_ents*el + 3*sDIM + idx];
|
||||
}
|
||||
pass_bb = (bbox_test(&box, x_i) >= 0);
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int d = 0; d < sDIM; ++d)
|
||||
{
|
||||
box.x[d].min = boxinfo[n_box_ents*el + d];
|
||||
box.x[d].max = boxinfo[n_box_ents*el + sDIM + d];
|
||||
}
|
||||
pass_bb = (AABB_test(&box, x_i) >= 0);
|
||||
}
|
||||
|
||||
if (pass_bb)
|
||||
{
|
||||
//// findpts_local ////
|
||||
{
|
||||
if (MD1 <= 6)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,D1D*sDIM)
|
||||
{
|
||||
const int qp = j % D1D;
|
||||
const int d = j / D1D;
|
||||
elem_coords[qp + d*D1D] =
|
||||
xElemCoord[qp + el*D1D + d*p_NEL];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
|
||||
const double *elx[sDIM];
|
||||
for (int d=0; d<sDIM; d++)
|
||||
{
|
||||
elx[d] = MD1<= 6 ? &elem_coords[d*D1D] :
|
||||
xElemCoord + d*p_NEL + el*D1D;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
//// findpts_el ////
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
fpt->dist2 = HUGE_VAL;
|
||||
fpt->dist2p = 0;
|
||||
fpt->tr = 1.0;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
fpt->x[j] = x_i[j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
//// seed ////
|
||||
{
|
||||
double *dist2_temp = r_workspace_ptr;
|
||||
double *r_temp = dist2_temp + D1D;
|
||||
MFEM_FOREACH_THREAD(j,x,nThreads)
|
||||
{
|
||||
seed_j(elx, x_i, gll1D, dist2_temp, r_temp, j, D1D);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
fpt->dist2 = HUGE_VAL;
|
||||
for (int ir=0; ir<D1D; ++ir)
|
||||
{
|
||||
if (dist2_temp[ir] < fpt->dist2)
|
||||
{
|
||||
fpt->dist2 = dist2_temp[ir];
|
||||
fpt->r = r_temp[ir];
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
} //seed done
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
tmp->dist2 = HUGE_VAL;
|
||||
tmp->dist2p = 0;
|
||||
tmp->tr = 1;
|
||||
tmp->flags = 0;
|
||||
tmp->r = fpt->r;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
tmp->x[j] = fpt->x[j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int step=0; step<50; step++)
|
||||
{
|
||||
switch (num_constrained(tmp->flags & FLAG_MASK))
|
||||
{
|
||||
case 0:
|
||||
{
|
||||
double *wt = r_workspace_ptr;
|
||||
double *resid = wt + 3*D1D;
|
||||
double *jac = resid + sDIM;
|
||||
double *hess = jac + sDIM*rDIM;
|
||||
|
||||
findptsElementGEdge_t edge;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d] = constraint_workspace + d*D1D;
|
||||
}
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
edge.x[d][j] = elx[d][j];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
lag_eval_second_der(wt, tmp->r, j, gll1D,
|
||||
lagcoeff, D1D);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,sDIM)
|
||||
{
|
||||
resid[j] = tmp->x[j];
|
||||
jac[j] = 0.0;
|
||||
hess[j] = 0.0;
|
||||
for (int k=0; k<D1D; ++k)
|
||||
{
|
||||
resid[j] -= wt[ k]*edge.x[j][k];
|
||||
jac[j] += wt[D1D+k]*edge.x[j][k];
|
||||
hess[j] += wt[2*D1D+k]*edge.x[j][k];
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
hess[3] = resid[0]*hess[0] + resid[1]*hess[1] +
|
||||
resid[2]*hess[2];
|
||||
}
|
||||
|
||||
MFEM_FOREACH_THREAD(l,x,1)
|
||||
{
|
||||
if (!reject_prior_step_q(fpt,resid,tmp,tol))
|
||||
{
|
||||
newton_edge(fpt,jac,hess[3],resid,
|
||||
tmp->flags&FLAG_MASK,tmp,tol);
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
break;
|
||||
}
|
||||
case 1:
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
const int pi = point_index(tmp->flags &
|
||||
FLAG_MASK);
|
||||
const double *wt = wtend + pi*3*D1D;
|
||||
findptsElementGPT_t gpt;
|
||||
for (int d=0; d<sDIM; ++d)
|
||||
{
|
||||
gpt.x[d] = elx[d][pi*(D1D-1)];
|
||||
gpt.jac[d] = 0.0;
|
||||
gpt.hes[d] = 0.0;
|
||||
for (int k=0; k<D1D; ++k)
|
||||
{
|
||||
gpt.jac[d] += wt[D1D +k]*elx[d][k];
|
||||
gpt.hes[d] += wt[2*D1D+k]*elx[d][k];
|
||||
}
|
||||
}
|
||||
|
||||
const double *const pt_x = gpt.x;
|
||||
const double *const jac = gpt.jac;
|
||||
const double *const hes = gpt.hes;
|
||||
double resid[sDIM], steep, sr;
|
||||
resid[0] = fpt->x[0] - pt_x[0];
|
||||
resid[1] = fpt->x[1] - pt_x[1];
|
||||
resid[2] = fpt->x[2] - pt_x[2];
|
||||
steep = jac[0]*resid[0] + jac[1]*resid[1] +
|
||||
jac[2]*resid[2];
|
||||
sr = steep*tmp->r;
|
||||
if (!reject_prior_step_q(fpt, resid, tmp, tol))
|
||||
{
|
||||
if (sr<0)
|
||||
{
|
||||
const double rhess = resid[0]*hes[0] +
|
||||
resid[1]*hes[1] +
|
||||
resid[2]*hes[2];
|
||||
newton_edge(fpt, jac, rhess,
|
||||
resid, 0, tmp, tol);
|
||||
}
|
||||
else // sr==0
|
||||
{
|
||||
fpt->r = tmp->r;
|
||||
fpt->dist2p = 0;
|
||||
fpt->flags = tmp->flags | CONVERGED_FLAG;
|
||||
}
|
||||
}
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
break;
|
||||
} // case 1
|
||||
} //switch
|
||||
if (fpt->flags & CONVERGED_FLAG)
|
||||
{
|
||||
break;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
*tmp = *fpt;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
} // for step<50
|
||||
} // findpts_el
|
||||
|
||||
bool converged_internal =
|
||||
((fpt->flags&FLAG_MASK) == CONVERGED_FLAG) &&
|
||||
(fpt->dist2<dist2tol);
|
||||
if (*code_i==CODE_NOT_FOUND || converged_internal ||
|
||||
fpt->dist2<*dist2_i)
|
||||
{
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
*(el_base+i) = el;
|
||||
*code_i = converged_internal?CODE_INTERNAL:CODE_BORDER;
|
||||
*dist2_i = fpt->dist2;
|
||||
*(r_base+i) = fpt->r;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
if (converged_internal)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
} // findpts_local
|
||||
} // obbox_test
|
||||
} // elp
|
||||
});
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::FindPointsEdgeLocal3(const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &code,
|
||||
Array<unsigned int> &elem,
|
||||
Vector &ref,
|
||||
Vector &dist,
|
||||
int npt)
|
||||
{
|
||||
if (npt == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
MFEM_VERIFY(spacedim==3 && dim == 1,"Function for 3D edges only");
|
||||
bool use_dev = point_pos.UseDevice();
|
||||
auto pp = point_pos.Read(use_dev);
|
||||
auto pgslm = gsl_mesh.Read(use_dev);
|
||||
auto pwt = DEV.wtend.Read(use_dev);
|
||||
auto pbb = DEV.bb.Read(use_dev);
|
||||
auto plhm = DEV.lh_min.Read(use_dev);
|
||||
auto plhf = DEV.lh_fac.Read(use_dev);
|
||||
auto plho = DEV.lh_offset.ReadWrite(use_dev);
|
||||
auto pcode = code.Write(use_dev);
|
||||
auto pelem = elem.Write(use_dev);
|
||||
auto pref = ref.Write(use_dev);
|
||||
auto pdist = dist.Write(use_dev);
|
||||
auto pgll1d = DEV.gll1d.ReadWrite(use_dev);
|
||||
auto plc = DEV.lagcoeff.Read(use_dev);
|
||||
double dist2tol = DEV.surf_dist_tol;
|
||||
const bool obb_chk = obb_check;
|
||||
switch (DEV.dof1d)
|
||||
{
|
||||
case 2:
|
||||
FindPointsEdgeLocal3DKernel<2>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 3:
|
||||
FindPointsEdgeLocal3DKernel<3>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
case 4:
|
||||
FindPointsEdgeLocal3DKernel<4>(npt, DEV.newt_tol, dist2tol,
|
||||
pp, point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc);
|
||||
break;
|
||||
default:
|
||||
FindPointsEdgeLocal3DKernel(npt, DEV.newt_tol, dist2tol, pp,
|
||||
point_pos_ordering, pgslm,
|
||||
NE_split_total, pwt, pbb, obb_chk,
|
||||
DEV.lh_nx, plhm, plhf, plho,
|
||||
pcode, pelem, pref, pdist,
|
||||
pgll1d, plc, DEV.dof1d);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef rDIM2
|
||||
#undef sDIM2
|
||||
#undef rDIM
|
||||
#undef sDIM
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
#else
|
||||
void FindPointsGSLIB::FindPointsEdgeLocal3( const Vector &point_pos,
|
||||
int point_pos_ordering,
|
||||
Array<unsigned int> &code,
|
||||
Array<unsigned int> &elem,
|
||||
Vector &ref,
|
||||
Vector &dist,
|
||||
int npt ) {} ;
|
||||
#endif
|
||||
} // namespace mfem
|
||||
|
||||
#endif //ifdef MFEM_USE_GSLIB
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,190 +0,0 @@
|
||||
#ifndef MFEM_GSLIB_KERNEL_HELPERS_HPP
|
||||
#define MFEM_GSLIB_KERNEL_HELPERS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
|
||||
#include <cmath>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace gslib
|
||||
{
|
||||
|
||||
struct dbl_range_t
|
||||
{
|
||||
double min, max;
|
||||
};
|
||||
|
||||
template <int SDIM>
|
||||
struct obbox_t
|
||||
{
|
||||
double c0[SDIM], A[SDIM * SDIM];
|
||||
dbl_range_t x[SDIM];
|
||||
};
|
||||
|
||||
template <int SDIM>
|
||||
struct findptsLocalHashData_t
|
||||
{
|
||||
int hash_n;
|
||||
dbl_range_t bnd[SDIM];
|
||||
double fac[SDIM];
|
||||
unsigned int *offset;
|
||||
};
|
||||
|
||||
// Eval the ith Lagrange interpolant at x.
|
||||
MFEM_HOST_DEVICE inline void lagrange_eval(double *p0, double x,
|
||||
int i, int p_Nq,
|
||||
double *z, double *lagrangeCoeff)
|
||||
{
|
||||
double p_i = (1 << (p_Nq - 1));
|
||||
for (int j = 0; j < p_Nq; ++j)
|
||||
{
|
||||
const double d_j = x - z[j];
|
||||
p_i *= j == i ? 1 : d_j;
|
||||
}
|
||||
p0[i] = lagrangeCoeff[i] * p_i;
|
||||
}
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first derivative at x.
|
||||
MFEM_HOST_DEVICE inline void lag_eval_first_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
const double d_j = 2 * (x - z[j]);
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN + i] = 2.0 * lCoeff[i] * u1;
|
||||
}
|
||||
|
||||
// Eval the ith Lagrange interpolant and its first and second derivative at x.
|
||||
MFEM_HOST_DEVICE inline void lag_eval_second_der(double *p0, double x,
|
||||
int i, const double *z,
|
||||
const double *lCoeff,
|
||||
int pN)
|
||||
{
|
||||
double u0 = 1, u1 = 0, u2 = 0;
|
||||
for (int j = 0; j < pN; ++j)
|
||||
{
|
||||
if (i != j)
|
||||
{
|
||||
const double d_j = 2 * (x - z[j]);
|
||||
u2 = d_j * u2 + u1;
|
||||
u1 = d_j * u1 + u0;
|
||||
u0 = d_j * u0;
|
||||
}
|
||||
}
|
||||
p0[i] = lCoeff[i] * u0;
|
||||
p0[pN + i] = 2.0 * lCoeff[i] * u1;
|
||||
p0[2 * pN + i] = 8.0 * lCoeff[i] * u2;
|
||||
}
|
||||
|
||||
// Solve Ax=y where A is a symmetric 2x2 matrix packed as {a00, a01, a11}.
|
||||
MFEM_HOST_DEVICE inline void lin_solve_sym_2(double x[2],
|
||||
const double A[3],
|
||||
const double y[2])
|
||||
{
|
||||
const double idet = 1 / (A[0] * A[2] - A[1] * A[1]);
|
||||
x[0] = idet * (A[2] * y[0] - A[1] * y[1]);
|
||||
x[1] = idet * (A[0] * y[1] - A[1] * y[0]);
|
||||
}
|
||||
|
||||
// Positive when the point is inside the axis-aligned bounding box.
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline double AABB_test(const obbox_t<SDIM> *const b,
|
||||
const double (&x)[SDIM])
|
||||
{
|
||||
double test = 1.0;
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
const double b_d = (x[d] - b->x[d].min) * (b->x[d].max - x[d]);
|
||||
test = test < 0.0 ? test : b_d;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
|
||||
// Positive when the point is inside the oriented bounding box.
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline double bbox_test(const obbox_t<SDIM> *const b,
|
||||
const double (&x)[SDIM])
|
||||
{
|
||||
const double bxyz = AABB_test(b, x);
|
||||
if (bxyz < 0.0)
|
||||
{
|
||||
return bxyz;
|
||||
}
|
||||
|
||||
double dxyz[SDIM];
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
dxyz[d] = x[d] - b->c0[d];
|
||||
}
|
||||
|
||||
double test = 1.0;
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
double rst = 0.0;
|
||||
for (int e = 0; e < SDIM; ++e)
|
||||
{
|
||||
rst += b->A[d * SDIM + e] * dxyz[e];
|
||||
}
|
||||
const double brst = (rst + 1.0) * (1.0 - rst);
|
||||
test = test < 0.0 ? test : brst;
|
||||
}
|
||||
return test;
|
||||
}
|
||||
|
||||
// Hash index in the hash table for the point x.
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline int hash_index(
|
||||
const findptsLocalHashData_t<SDIM> *const p,
|
||||
const double (&x)[SDIM])
|
||||
{
|
||||
const int n = p->hash_n;
|
||||
int sum = 0;
|
||||
for (int d = SDIM - 1; d >= 0; --d)
|
||||
{
|
||||
sum *= n;
|
||||
const int i = (int)floor((x[d] - p->bnd[d].min) * p->fac[d]);
|
||||
sum += i < 0 ? 0 : (n - 1 < i ? n - 1 : i);
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
// Squared Euclidean norm.
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline double l2norm2(const double (&x)[SDIM])
|
||||
{
|
||||
double sum = 0.0;
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
sum += x[d] * x[d];
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
template <int SDIM>
|
||||
MFEM_HOST_DEVICE inline double l2norm2(const double *x)
|
||||
{
|
||||
double sum = 0.0;
|
||||
for (int d = 0; d < SDIM; ++d)
|
||||
{
|
||||
sum += x[d] * x[d];
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
} // namespace gslib
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif
|
||||
@@ -1,152 +0,0 @@
|
||||
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-function"
|
||||
#endif
|
||||
#include "gslib.h"
|
||||
#ifndef GSLIB_RELEASE_VERSION //gslib v1.0.7
|
||||
#define GSLIB_RELEASE_VERSION 10007
|
||||
#endif
|
||||
#ifdef MFEM_HAVE_GCC_PRAGMA_DIAGNOSTIC
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
namespace mfem
|
||||
{
|
||||
#if GSLIB_RELEASE_VERSION >= 10009
|
||||
#define CODE_INTERNAL 0
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
|
||||
using gslib::lagrange_eval;
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void InterpolateLocal1DKernel(const double *const gf_in,
|
||||
int *const el,
|
||||
double *const r,
|
||||
double *const int_out,
|
||||
const int npt,
|
||||
const int nfields,
|
||||
double *gll1D,
|
||||
double *lagcoeff,
|
||||
const int pN = 0)
|
||||
{
|
||||
const int MD1 = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
|
||||
const int D1D = T_D1D ? T_D1D : pN;
|
||||
const int p_Nq = D1D;
|
||||
MFEM_VERIFY(MD1 <= DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(D1D != 0, "Polynomial order not specified.");
|
||||
// for each point of the npt points, create a thread block of size dof1Dsol
|
||||
mfem::forall_2D(npt, D1D, 1, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
MFEM_SHARED double wtr[MD1];
|
||||
MFEM_SHARED double sums[MD1];
|
||||
|
||||
// Evaluate basis functions at the reference space coordinates
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
lagrange_eval(wtr, r[i], j, p_Nq, gll1D, lagcoeff);
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
for (int fld=0; fld<nfields; ++fld)
|
||||
{
|
||||
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
|
||||
// offset would be `el[i] * p_Nq + fld * gf_offset`.
|
||||
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
|
||||
const int elemOffset = el[i]*nfields*p_Nq + fld*p_Nq;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
sums[j] = wtr[j] * gf_in[elemOffset + j];
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
|
||||
MFEM_FOREACH_THREAD(j,x,1)
|
||||
{
|
||||
double sumv = 0.0;
|
||||
// sum the contributions of each lagrange polynomial
|
||||
for (int jj=0; jj<D1D; ++jj)
|
||||
{
|
||||
sumv += sums[jj];
|
||||
}
|
||||
int_out[fld*npt + i] = sumv;
|
||||
}
|
||||
MFEM_SYNC_THREAD;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void FindPointsGSLIB::InterpolateLocal1( const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt,
|
||||
int ncomp,
|
||||
int dof1Dsol )
|
||||
{
|
||||
MFEM_VERIFY(dim == 1, "Kernel for edges only.");
|
||||
if (npt == 0) { return; }
|
||||
bool use_dev = field_in.UseDevice();
|
||||
auto pfin = field_in.Read(use_dev);
|
||||
auto pgsl = gsl_elem_dev_l.ReadWrite(use_dev);
|
||||
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
|
||||
auto pfout = field_out.Write(use_dev);
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2:
|
||||
InterpolateLocal1DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 3:
|
||||
InterpolateLocal1DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 4:
|
||||
InterpolateLocal1DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 5:
|
||||
InterpolateLocal1DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
default:
|
||||
InterpolateLocal1DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf, dof1Dsol);
|
||||
break;
|
||||
}
|
||||
}
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
#else
|
||||
void FindPointsGSLIB::InterpolateLocal1(const Vector &field_in,
|
||||
Array<int> &gsl_elem_dev_l,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int dof1Dsol) {};
|
||||
#endif
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif //ifdef MFEM_USE_GSLIB
|
||||
@@ -11,7 +11,6 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -33,7 +32,18 @@ namespace mfem
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
|
||||
using gslib::lagrange_eval;
|
||||
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
|
||||
int i, int p_Nq,
|
||||
double *z, double *lagrangeCoeff)
|
||||
{
|
||||
double p_i = (1 << (p_Nq - 1));
|
||||
for (int j = 0; j < p_Nq; ++j)
|
||||
{
|
||||
double d_j = x - z[j];
|
||||
p_i *= j == i ? 1 : d_j;
|
||||
}
|
||||
p0[i] = lagrangeCoeff[i] * p_i;
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void InterpolateLocal2DKernel(const double *const gf_in,
|
||||
@@ -42,6 +52,8 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
|
||||
double *const int_out,
|
||||
const int npt,
|
||||
const int ncomp,
|
||||
const int nel,
|
||||
const int gf_offset,
|
||||
double *gll1D,
|
||||
double *lagcoeff,
|
||||
const int pN = 0)
|
||||
@@ -52,8 +64,6 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
|
||||
const int p_Np = D1D*D1D;
|
||||
MFEM_VERIFY(MD1 <= DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(pN<=DofQuadLimits::MAX_D1D,
|
||||
"Increase Max allowable polynomial order.");
|
||||
MFEM_VERIFY(D1D != 0, "Polynomial order not specified.");
|
||||
mfem::forall_2D(npt, D1D, D1D, [=] MFEM_HOST_DEVICE (int i)
|
||||
{
|
||||
@@ -72,9 +82,9 @@ static void InterpolateLocal2DKernel(const double *const gf_in,
|
||||
|
||||
for (int fld = 0; fld < Nfields; ++fld)
|
||||
{
|
||||
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
|
||||
// offset would be `el[i] * p_Np + fld * gf_offset`.
|
||||
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
|
||||
// If using GetNodalValues, ordering is NDOFSxNELxVDIM
|
||||
// const int elemOffset = el[i] * p_Np + fld * gf_offset;
|
||||
//if using R->Mult for L -> E-Vec use below: NDOFSxVDIMxNEL
|
||||
const int elemOffset = el[i] * p_Np * Nfields + fld * p_Np;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
@@ -110,38 +120,33 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int dof1Dsol)
|
||||
int nel, int dof1Dsol)
|
||||
{
|
||||
if (npt == 0) { return; }
|
||||
bool use_dev = field_in.UseDevice();
|
||||
auto pfin = field_in.Read(use_dev);
|
||||
auto pgsl = gsl_elem_dev_l.ReadWrite(use_dev);
|
||||
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
|
||||
auto pfout = field_out.Write(use_dev);
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
const int gf_offset = field_in.Size()/ncomp;
|
||||
auto pfin = field_in.Read();
|
||||
auto pgsl = gsl_elem_dev_l.ReadWrite();
|
||||
auto pgslr = gsl_ref_l.ReadWrite();
|
||||
auto pfout = field_out.Write();
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite();
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite();
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2:
|
||||
InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 3:
|
||||
InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 4:
|
||||
InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 5:
|
||||
InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
default:
|
||||
InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf, dof1Dsol);
|
||||
break;
|
||||
case 2: return InterpolateLocal2DKernel<2>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf);
|
||||
case 3: return InterpolateLocal2DKernel<3>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf);
|
||||
case 4: return InterpolateLocal2DKernel<4>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf);
|
||||
case 5: return InterpolateLocal2DKernel<5>(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf);
|
||||
default: return InterpolateLocal2DKernel(pfin, pgsl, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf, dof1Dsol);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -155,7 +160,7 @@ void FindPointsGSLIB::InterpolateLocal2(const Vector &field_in,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int dof1Dsol) {};
|
||||
int nel, int dof1Dsol) {};
|
||||
#endif
|
||||
} // namespace mfem
|
||||
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
|
||||
#include "../gslib.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
#include "gslib_kernel_helpers.hpp"
|
||||
|
||||
#ifdef MFEM_USE_GSLIB
|
||||
|
||||
@@ -33,7 +32,18 @@ namespace mfem
|
||||
#define CODE_BORDER 1
|
||||
#define CODE_NOT_FOUND 2
|
||||
|
||||
using gslib::lagrange_eval;
|
||||
static MFEM_HOST_DEVICE void lagrange_eval(double *p0, double x,
|
||||
int i, int p_Nq,
|
||||
double *z, double *lagrangeCoeff)
|
||||
{
|
||||
double p_i = (1 << (p_Nq - 1));
|
||||
for (int j = 0; j < p_Nq; ++j)
|
||||
{
|
||||
double d_j = x - z[j];
|
||||
p_i *= j == i ? 1 : d_j;
|
||||
}
|
||||
p0[i] = lagrangeCoeff[i] * p_i;
|
||||
}
|
||||
|
||||
template<int T_D1D = 0>
|
||||
static void InterpolateLocal3DKernel(const double *const gf_in,
|
||||
@@ -42,6 +52,8 @@ static void InterpolateLocal3DKernel(const double *const gf_in,
|
||||
double *const int_out,
|
||||
const int npt,
|
||||
const int ncomp,
|
||||
const int nel,
|
||||
const int gf_offset,
|
||||
double *gll1D,
|
||||
double *lagcoeff,
|
||||
const int pN = 0)
|
||||
@@ -72,9 +84,9 @@ static void InterpolateLocal3DKernel(const double *const gf_in,
|
||||
|
||||
for (int fld = 0; fld < Nfields; ++fld)
|
||||
{
|
||||
// If using GetNodalValues, ordering is NDOFS x NEL x VDIM and the
|
||||
// offset would be `el[i] * p_Np + fld * gf_offset`.
|
||||
// R->Mult produces element vectors in NDOFS x VDIM x NEL layout.
|
||||
// If using GetNodalValues, ordering is NDOFSxNELxVDIM
|
||||
// const int elemOffset = el[i] * p_Np + fld * gf_offset;
|
||||
//if using R->Mult for L -> E-Vec use below.
|
||||
const int elemOffset = el[i] * p_Np * Nfields + fld * p_Np;
|
||||
MFEM_FOREACH_THREAD(j,x,D1D)
|
||||
{
|
||||
@@ -113,43 +125,37 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int dof1Dsol)
|
||||
int nel, int dof1Dsol)
|
||||
{
|
||||
if (npt == 0) { return; }
|
||||
bool use_dev = field_in.UseDevice();
|
||||
auto pfin = field_in.Read(use_dev);
|
||||
auto pgsle = gsl_elem_dev_l.ReadWrite(use_dev);
|
||||
auto pgslr = gsl_ref_l.ReadWrite(use_dev);
|
||||
auto pfout = field_out.Write(use_dev);
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite(use_dev);
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite(use_dev);
|
||||
const int gf_offset = field_in.Size()/ncomp;
|
||||
auto pfin = field_in.Read();
|
||||
auto pgsle = gsl_elem_dev_l.ReadWrite();
|
||||
auto pgslr = gsl_ref_l.ReadWrite();
|
||||
auto pfout = field_out.Write();
|
||||
auto pgll = DEV.gll1d_sol.ReadWrite();
|
||||
auto plcf = DEV.lagcoeff_sol.ReadWrite();
|
||||
switch (dof1Dsol)
|
||||
{
|
||||
case 2:
|
||||
InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 3:
|
||||
InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 4:
|
||||
InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
case 5:
|
||||
InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf);
|
||||
break;
|
||||
default:
|
||||
InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, pgll, plcf, dof1Dsol);
|
||||
break;
|
||||
case 2: return InterpolateLocal3DKernel<2>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf);
|
||||
case 3: return InterpolateLocal3DKernel<3>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf);
|
||||
case 4: return InterpolateLocal3DKernel<4>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf);
|
||||
case 5: return InterpolateLocal3DKernel<5>(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf);
|
||||
default: return InterpolateLocal3DKernel(pfin, pgsle, pgslr, pfout,
|
||||
npt, ncomp, nel, gf_offset,
|
||||
pgll, plcf, dof1Dsol);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
#undef MAXC
|
||||
#undef CODE_INTERNAL
|
||||
#undef CODE_BORDER
|
||||
#undef CODE_NOT_FOUND
|
||||
@@ -159,7 +165,7 @@ void FindPointsGSLIB::InterpolateLocal3(const Vector &field_in,
|
||||
Vector &gsl_ref_l,
|
||||
Vector &field_out,
|
||||
int npt, int ncomp,
|
||||
int dof1Dsol) {};
|
||||
int nel, int dof1Dsol) {};
|
||||
#endif
|
||||
} // namespace mfem
|
||||
|
||||
|
||||
@@ -178,8 +178,6 @@ void ConvectionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
// Assumes tensor-product elements
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetTypicalFE();
|
||||
MFEM_VERIFY(el.GetMapType() == FiniteElement::VALUE,
|
||||
"Only value map type currently supported");
|
||||
ElementTransformation &Trans = *mesh->GetTypicalElementTransformation();
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, Trans);
|
||||
if (DeviceCanUseCeed())
|
||||
|
||||
@@ -10,7 +10,6 @@
|
||||
// CONTRIBUTING.md for details.
|
||||
|
||||
#include "bilininteg_diffusion_kernels.hpp"
|
||||
#include "bilininteg_diffusion_pa_simplices.hpp" // IWYU pragma: keep
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -20,13 +19,6 @@ namespace mfem
|
||||
DiffusionIntegrator::Kernels::Kernels()
|
||||
{
|
||||
// 2D
|
||||
// Q = P, only for simplex
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,2,1>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,3,2>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,4,3>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,5,4>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,6,5>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,7,6>();
|
||||
// Q = P+1
|
||||
DiffusionIntegrator::AddSpecialization<2,1,1>();
|
||||
DiffusionIntegrator::AddSpecialization<2,2,2>();
|
||||
@@ -48,18 +40,7 @@ DiffusionIntegrator::Kernels::Kernels()
|
||||
DiffusionIntegrator::AddSpecialization<2,8,9>();
|
||||
DiffusionIntegrator::AddSpecialization<2,9,10>();
|
||||
// others
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,2,5>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<2,3,6>();
|
||||
|
||||
// 3D
|
||||
// Q = P, only for simplex
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,2,1>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,3,2>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,4,3>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,5,4>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,6,5>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,7,6>();
|
||||
DiffusionIntegrator::AddSimplexSpecialization<3,8,7>();
|
||||
// Q = P+1
|
||||
DiffusionIntegrator::AddSpecialization<3,1,1>();
|
||||
DiffusionIntegrator::AddSpecialization<3,2,2>();
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#ifndef MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
|
||||
#define MFEM_BILININTEG_DIFFUSION_KERNELS_HPP
|
||||
|
||||
#include "../kernel_dispatch.hpp"
|
||||
#include "../../config/config.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
#include "../../general/forall.hpp"
|
||||
@@ -19,8 +20,6 @@
|
||||
#include "../../linalg/vector.hpp"
|
||||
#include "../bilininteg.hpp"
|
||||
|
||||
#include "bilininteg_diffusion_pa_simplices.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
@@ -638,8 +637,8 @@ inline void SmemPADiffusionApply2D(const int NE,
|
||||
const bool symmetric,
|
||||
const Array<real_t> &b_,
|
||||
const Array<real_t> &g_,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &,
|
||||
const Array<real_t> &bt_,
|
||||
const Array<real_t> >_,
|
||||
const Vector &d_,
|
||||
const Vector &x_,
|
||||
Vector &y_,
|
||||
@@ -1219,47 +1218,43 @@ inline void SmemPADiffusionApply3D(const int NE,
|
||||
namespace
|
||||
{
|
||||
using ApplyKernelType = DiffusionIntegrator::ApplyKernelType;
|
||||
using ApplySimplexKernelType = DiffusionIntegrator::ApplySimplexKernelType;
|
||||
using DiagonalKernelType = DiffusionIntegrator::DiagonalKernelType;
|
||||
}
|
||||
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
template<int DIM, int T_D1D, int T_Q1D>
|
||||
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<D1D, Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<D1D, Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionApply2D<T_D1D,T_Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionApply3D<T_D1D, T_Q1D>; }
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
inline
|
||||
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int, int)
|
||||
ApplyKernelType DiffusionIntegrator::ApplyPAKernels::Fallback(int DIM, int, int)
|
||||
{
|
||||
if (dim == 2) { return internal::PADiffusionApply2D; }
|
||||
else if (dim == 3) { return internal::PADiffusionApply3D; }
|
||||
if (DIM == 2) { return internal::PADiffusionApply2D; }
|
||||
else if (DIM == 3) { return internal::PADiffusionApply3D; }
|
||||
else { MFEM_ABORT(""); }
|
||||
}
|
||||
|
||||
template<int DIM, int D1D, int Q1D>
|
||||
DiagonalKernelType DiffusionIntegrator::DiagonalPAKernels::Kernel()
|
||||
{
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D, Q1D>; }
|
||||
if constexpr (DIM == 2) { return internal::SmemPADiffusionDiagonal2D<D1D,Q1D>; }
|
||||
else if constexpr (DIM == 3) { return internal::SmemPADiffusionDiagonal3D<D1D, Q1D>; }
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
MFEM_ABORT("");
|
||||
}
|
||||
|
||||
inline DiagonalKernelType
|
||||
DiffusionIntegrator::DiagonalPAKernels::Fallback(int dim, int, int)
|
||||
DiffusionIntegrator::DiagonalPAKernels::Fallback(int DIM, int, int)
|
||||
{
|
||||
if (dim == 2) { return internal::PADiffusionDiagonal2D; }
|
||||
else if (dim == 3) { return internal::PADiffusionDiagonal3D; }
|
||||
if (DIM == 2) { return internal::PADiffusionDiagonal2D; }
|
||||
else if (DIM == 3) { return internal::PADiffusionDiagonal3D; }
|
||||
else { MFEM_ABORT(""); }
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
/// \endcond DO_NOT_DOCUMENT
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -15,7 +15,6 @@
|
||||
#include "../../mesh/nurbs.hpp"
|
||||
#include "../ceed/integrators/diffusion/diffusion.hpp"
|
||||
#include "bilininteg_diffusion_kernels.hpp"
|
||||
#include "bilininteg_diffusion_pa_simplices.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
@@ -69,24 +68,6 @@ void DiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
|
||||
}
|
||||
#endif // MFEM_USE_OCCA
|
||||
|
||||
if (fespace->UsesRaggedTensorBasis())
|
||||
{
|
||||
const auto *rmaps = static_cast<const RaggedDofToQuad*>(maps);
|
||||
return ApplySimplexPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric,
|
||||
rmaps->lex_map,
|
||||
rmaps->forward_map2d_diff,
|
||||
rmaps->inverse_map2d_diff,
|
||||
rmaps->forward_map3d_diff,
|
||||
rmaps->inverse_map3d_diff,
|
||||
rmaps->Ga1,
|
||||
rmaps->Ga2,
|
||||
rmaps->Ga3,
|
||||
rmaps->Ga1t,
|
||||
rmaps->Ga2t,
|
||||
rmaps->Ga3t,
|
||||
Dv, x, y, dofs1D, quad1D);
|
||||
}
|
||||
|
||||
ApplyPAKernels::Run(dim, dofs1D, quad1D, ne, symmetric, B, G, Bt,
|
||||
Gt, Dv, x, y, dofs1D, quad1D);
|
||||
}
|
||||
@@ -113,8 +94,7 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
fespace = &fes;
|
||||
Mesh *mesh = fes.GetMesh();
|
||||
const FiniteElement &el = *fes.GetTypicalFE();
|
||||
const bool stroud = fes.UsesRaggedTensorBasis();
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el, stroud);
|
||||
const IntegrationRule *ir = IntRule ? IntRule : &GetRule(el, el);
|
||||
if (DeviceCanUseCeed())
|
||||
{
|
||||
delete ceedOp;
|
||||
@@ -139,22 +119,13 @@ void DiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
|
||||
dim = mesh->Dimension();
|
||||
ne = fes.GetNE();
|
||||
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
|
||||
if (stroud)
|
||||
{
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::RAGGED_TENSOR);
|
||||
}
|
||||
else
|
||||
{
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
}
|
||||
const int sdim = mesh->SpaceDimension();
|
||||
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
|
||||
dofs1D = maps->ndof;
|
||||
quad1D = maps->nqpt;
|
||||
|
||||
QuadratureSpace qs(*mesh, *ir);
|
||||
CoefficientVector coeff(qs, CoefficientStorage::COMPRESSED);
|
||||
// QuadratureSpace expects ir defined in reference simplex for Bernstein
|
||||
// elements with partial assembly
|
||||
|
||||
if (MQ) { coeff.ProjectTranspose(*MQ); }
|
||||
else if (VQ) { coeff.Project(*VQ); }
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -91,15 +91,15 @@ void ElasticityAddMultPA(const int dim, const int nDofs,
|
||||
void ElasticityAssembleDiagonalPA(const int dim, const int nDofs,
|
||||
const CoefficientVector &lambda,
|
||||
const CoefficientVector &mu, const GeometricFactors &geom,
|
||||
const DofToQuad &maps, const IntegrationRule &ir, Vector &diag)
|
||||
const DofToQuad &maps, QuadratureFunction &QVec, Vector &diag)
|
||||
{
|
||||
switch (dim)
|
||||
{
|
||||
case 2:
|
||||
ElasticityAssembleDiagonalPA_<2>(nDofs, lambda, mu, geom, maps, ir, diag);
|
||||
ElasticityAssembleDiagonalPA_<2>(nDofs, lambda, mu, geom, maps, QVec, diag);
|
||||
break;
|
||||
case 3:
|
||||
ElasticityAssembleDiagonalPA_<3>(nDofs, lambda, mu, geom, maps, ir, diag);
|
||||
ElasticityAssembleDiagonalPA_<3>(nDofs, lambda, mu, geom, maps, QVec, diag);
|
||||
break;
|
||||
default:
|
||||
MFEM_ABORT("Only dimensions 2 and 3 supported.");
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user