Compare commits
42
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
33ccd307bd | ||
|
|
958e0f27c4 | ||
|
|
07d043ec77 | ||
|
|
a61e836c4a | ||
|
|
aa58b549ab | ||
|
|
87b9412e80 | ||
|
|
36a9a3ac92 | ||
|
|
5bd5e169a3 | ||
|
|
a2ad9af08f | ||
|
|
c967429b2d | ||
|
|
c33e2edb72 | ||
|
|
7057bde885 | ||
|
|
fa1f7666c4 | ||
|
|
066c37520a | ||
|
|
61cd1aa8cd | ||
|
|
f87dbdc2ad | ||
|
|
6c0777c0e1 | ||
|
|
3051b7ed11 | ||
|
|
e77ee6a3a3 | ||
|
|
d0c90c8505 | ||
|
|
8ee2e444be | ||
|
|
7336d8ea84 | ||
|
|
cf053cdc59 | ||
|
|
e845d83cce | ||
|
|
9764593415 | ||
|
|
5beb85d4ce | ||
|
|
7f1d0689ef | ||
|
|
bdc8e0c16a | ||
|
|
3a6ef2cd85 | ||
|
|
d49258aaaa | ||
|
|
abfa3bc631 | ||
|
|
4f11a7194d | ||
|
|
bfaf7a8da9 | ||
|
|
65e1de0f2b | ||
|
|
8980df563e | ||
|
|
b79c9fc31c | ||
|
|
65e1fe7365 | ||
|
|
60834fc386 | ||
|
|
7a1b66b203 | ||
|
|
7d0d3bb6e1 | ||
|
|
788bcea676 | ||
|
|
2c3ce1b4db |
+14
-16
@@ -15,10 +15,8 @@ install:
|
||||
- msmpisdk.msi /passive
|
||||
- set PATH=C:\Program Files\Microsoft MPI\Bin;%PATH%
|
||||
|
||||
# Install METIS, use a mirror because the original source server is not always
|
||||
# up. Original url:
|
||||
# http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz
|
||||
- ps: Start-FileDownload 'https://mfem.github.io/tpls/metis-5.1.0.tar.gz'
|
||||
# Install METIS
|
||||
- ps: Start-FileDownload 'http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/metis-5.1.0.tar.gz'
|
||||
- 7z x metis-5.1.0.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd metis-5.1.0
|
||||
- ps: ( get-content "GKlib\gk_arch.h") | % { If ($_.ReadCount -ge 52) {$_ -replace "#ifdef __MSC__","#ifdef DISABLE_THIS_ANCIENT_MSC_CHECK"} Else {$_} } | set-content "GKlib\gk_arch.h"
|
||||
@@ -28,24 +26,24 @@ install:
|
||||
- cd ..
|
||||
|
||||
# Install hypre
|
||||
- ps: Start-FileDownload 'https://github.com/hypre-space/hypre/archive/v2.19.0.tar.gz'
|
||||
- 7z x v2.19.0.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd hypre-2.19.0/src
|
||||
- cmake -H. -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
- ps: Start-FileDownload 'https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz'
|
||||
- 7z x hypre-2.10.0b.tar.gz -so | 7z x -si -ttar > nul
|
||||
- cd hypre-2.10.0b
|
||||
- cmake -Hsrc -Bbuild -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
# - cmake -Hsrc -Bbuild -DCMAKE_BUILD_TYPE=Release -DMPI_C_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DMPI_C_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include"
|
||||
- cmake --build build
|
||||
- cmake --build build --target install
|
||||
- cd ../..
|
||||
- cd ..
|
||||
|
||||
# MFEM
|
||||
before_build:
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_DIR=%cd%\hypre-2.19.0\src\hypre -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_parallel -DMFEM_USE_MPI=TRUE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2.10.0b\src\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2.10.0b\src\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
- cmake -H. -DCMAKE_INSTALL_PREFIX=install -Bbuild_serial -DMFEM_USE_MPI=FALSE -DMFEM_USE_METIS_5=TRUE -DMPI_CXX_LIBRARIES="C:\Program Files (x86)\Microsoft SDKs\MPI\Lib\x86\msmpi.lib" -DMPI_CXX_INCLUDE_PATH="C:\Program Files (x86)\Microsoft SDKs\MPI\Include" -DHYPRE_LIBRARIES=%cd%\hypre-2.10.0b\src\hypre\lib\HYPRE.lib -DHYPRE_INCLUDE_DIRS=%cd%\hypre-2.10.0b\src\hypre\include -DHYPRE_VERSION=21000 -DMETIS_LIBRARIES=%cd%\metis-5.1.0\build\libmetis\Debug\metis.lib -DMETIS_INCLUDE_DIRS=%cd%\metis-5.1.0\include
|
||||
|
||||
build_script:
|
||||
- cmake --build build_parallel --config Release -j 4
|
||||
- cmake --build build_serial --config Release -j 4
|
||||
- cmake --build build_serial --target exec --config Release -j 4
|
||||
- cmake --build build_parallel
|
||||
- cmake --build build_serial
|
||||
|
||||
after_build:
|
||||
- cd build_serial
|
||||
- ctest -C Release --output-on-failure
|
||||
# - cmake --build build_parallel --target check
|
||||
- cmake --build build_serial --target check
|
||||
|
||||
@@ -1,32 +0,0 @@
|
||||
codecov:
|
||||
require_ci_to_pass: yes
|
||||
|
||||
coverage:
|
||||
precision: 2
|
||||
round: nearest
|
||||
range: "0...100"
|
||||
status:
|
||||
patch:
|
||||
default:
|
||||
target: auto
|
||||
threshold: 0%
|
||||
base: auto
|
||||
branches:
|
||||
- master
|
||||
if_ci_failed: error
|
||||
informational: true
|
||||
only_pulls: true
|
||||
project:
|
||||
default:
|
||||
target: auto # compares coverage to the previous base commit
|
||||
threshold: 1% # allows variations around the target
|
||||
base: auto
|
||||
branches:
|
||||
- master
|
||||
if_ci_failed: error
|
||||
only_pulls: true
|
||||
|
||||
github_checks:
|
||||
annotations: false
|
||||
|
||||
comment: false
|
||||
@@ -1,61 +0,0 @@
|
||||
# Configuration for probot-stale - https://github.com/probot/stale
|
||||
|
||||
# Number of days of inactivity before an Issue or Pull Request becomes stale
|
||||
daysUntilStale: 30
|
||||
|
||||
# Number of days of inactivity before an Issue or Pull Request with the stale
|
||||
# label is closed. Set to false to disable. If disabled, issues still need to
|
||||
# be closed manually, but will remain marked as stale.
|
||||
daysUntilClose: 7
|
||||
|
||||
# Only issues or pull requests with all of these labels are check if stale.
|
||||
# Defaults to `[]` (disabled)
|
||||
onlyLabels: []
|
||||
|
||||
# Issues or Pull Requests with these labels will never be considered stale. Set
|
||||
# to `[]` to disable
|
||||
exemptLabels:
|
||||
- bug
|
||||
- WIP
|
||||
- ready-for-review
|
||||
- in-review
|
||||
- in-next
|
||||
|
||||
# Set to true to ignore issues in a project (defaults to false)
|
||||
exemptProjects: false
|
||||
|
||||
# Set to true to ignore issues in a milestone (defaults to false)
|
||||
exemptMilestones: false
|
||||
|
||||
# Set to true to ignore issues with an assignee (defaults to false)
|
||||
exemptAssignees: false
|
||||
|
||||
# Label to use when marking an issue as stale
|
||||
staleLabel: stale
|
||||
|
||||
# Comment to post when marking an issue as stale. Set to `false` to disable
|
||||
markComment: >
|
||||
:warning: This issue or PR has been automatically marked as stale because it has not
|
||||
had any activity in the last month. *If no activity occurs in the next week, it will
|
||||
be automatically closed.* Thank you for your contributions.
|
||||
|
||||
# Comment to post when closing a stale issue. Set to `false` to disable
|
||||
closeComment: false
|
||||
|
||||
# Limit the number of actions per hour, from 1-30. Default is 30
|
||||
limitPerRun: 30
|
||||
|
||||
# Limit to only `issues` or `pulls`
|
||||
# only: issues
|
||||
|
||||
# Optionally, specify configuration settings that are specific to just 'issues' or 'pulls':
|
||||
# pulls:
|
||||
# daysUntilStale: 30
|
||||
# markComment: >
|
||||
# This pull request has been automatically marked as stale because it has not had
|
||||
# recent activity. It will be closed if no further activity occurs. Thank you
|
||||
# for your contributions.
|
||||
|
||||
# issues:
|
||||
# exemptLabels:
|
||||
# - confirmed
|
||||
@@ -1,208 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# In this CI section, we build different variants of mfem and run test on them.
|
||||
name: builds-and-tests
|
||||
|
||||
# Github actions can use the default "GITHUB_TOKEN". By default, this token
|
||||
# is set to have permissive access. However, this is not a good practice
|
||||
# security-wise. Here we use an external action, so we restrict the
|
||||
# permission to the minimum required.
|
||||
# When the 'permissions' is set, all the scopes not mentioned are set to the
|
||||
# most restrictive setting. So the following is enough.
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
- next
|
||||
pull_request:
|
||||
|
||||
env:
|
||||
HYPRE_ARCHIVE: v2.19.0.tar.gz
|
||||
HYPRE_TOP_DIR: hypre-2.19.0
|
||||
METIS_ARCHIVE: metis-4.0.3.tar.gz
|
||||
METIS_TOP_DIR: metis-4.0.3
|
||||
MFEM_TOP_DIR: mfem
|
||||
|
||||
# Note for future improvements:
|
||||
#
|
||||
# We cannot reuse cached dependencies and have to build them for each target
|
||||
# although they could be shared sometimes. That's because Github cache Action
|
||||
# has no read-only mode. But there is a PR ready for this
|
||||
# (https://github.com/actions/cache/pull/489)
|
||||
|
||||
jobs:
|
||||
builds-and-tests:
|
||||
strategy:
|
||||
matrix:
|
||||
os: [ubuntu-18.04, macos-10.15]
|
||||
target: [dbg, opt]
|
||||
mpi: [seq, par]
|
||||
build-system: [make]
|
||||
hypre-target: [int32]
|
||||
# 'include' allows us to:
|
||||
# - Add a variable to all jobs without creating a new matrix dimension.
|
||||
# Codecov is defined that way.
|
||||
# - Add a new combination.
|
||||
# 'build-system: cmake' and 'hypre-target: int64'
|
||||
#
|
||||
# note: we will gather coverage info for any non-debug run except the
|
||||
# CMake build.
|
||||
include:
|
||||
- target: dbg
|
||||
codecov: NO
|
||||
- target: opt
|
||||
codecov: YES
|
||||
- os: ubuntu-18.04
|
||||
target: opt
|
||||
codecov: NO
|
||||
mpi: par
|
||||
build-system: cmake
|
||||
hypre-target: int32
|
||||
- os: ubuntu-18.04
|
||||
target: opt
|
||||
codecov: NO
|
||||
mpi: par
|
||||
build-system: make
|
||||
hypre-target: int64
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
steps:
|
||||
# This external action allows to interrupt a workflow already running on
|
||||
# the same branch to save resource
|
||||
- name: Cancel Previous Runs
|
||||
uses: styfle/cancel-workflow-action@0.9.0
|
||||
with:
|
||||
access_token: ${{ github.token }}
|
||||
|
||||
# Checkout MFEM in "mfem" subdirectory. Final path:
|
||||
# /home/runner/work/mfem/mfem/mfem
|
||||
# Note: Done now to access "install-hypre" and "install-metis" actions.
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v2
|
||||
with:
|
||||
path: ${{ env.MFEM_TOP_DIR }}
|
||||
# Fetch the complete history for codecov to access commits ID
|
||||
fetch-depth: 0
|
||||
|
||||
# Only get MPI if defined for the job.
|
||||
# TODO: It would be nice to have only one step, e.g. with a dedicated
|
||||
# action, but I (@adrienbernede) don't see how at the moment.
|
||||
- name: get MPI (Linux)
|
||||
if: matrix.mpi == 'par' && matrix.os == 'ubuntu-18.04'
|
||||
run: |
|
||||
sudo apt-get install mpich libmpich-dev
|
||||
export MAKE_CXX_FLAG="MPICXX=mpic++"
|
||||
|
||||
- name: get lcov (Linux)
|
||||
if: matrix.codecov == 'YES' && matrix.os == 'ubuntu-18.04'
|
||||
run: |
|
||||
sudo apt-get install lcov
|
||||
|
||||
- name: Set up Homebrew
|
||||
if: ( matrix.mpi == 'par' || matrix.codecov == 'YES' ) && matrix.os == 'macos-10.15'
|
||||
uses: Homebrew/actions/setup-homebrew@c4aafe8c4620bf08883dd4679c374f11e73329d3
|
||||
|
||||
- name: get MPI (MacOS)
|
||||
if: matrix.mpi == 'par' && matrix.os == 'macos-10.15'
|
||||
run: |
|
||||
export HOMEBREW_NO_INSTALL_CLEANUP=1
|
||||
brew install openmpi
|
||||
export MAKE_CXX_FLAG="MPICXX=mpic++"
|
||||
|
||||
- name: get MPI (MacOS)
|
||||
if: matrix.codecov == 'YES' && matrix.os == 'macos-10.15'
|
||||
run: |
|
||||
export HOMEBREW_NO_INSTALL_CLEANUP=1
|
||||
brew install lcov
|
||||
|
||||
# Get Hypre through cache, or build it.
|
||||
# Install will only run on cache miss.
|
||||
- name: cache hypre
|
||||
id: hypre-cache
|
||||
if: matrix.mpi == 'par'
|
||||
uses: actions/cache@v2
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.HYPRE_TOP_DIR }}-${{ matrix.hypre-target }}-v2.0
|
||||
|
||||
- name: get hypre
|
||||
if: matrix.mpi == 'par' && steps.hypre-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-hypre@v2.0
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
target: ${{ matrix.hypre-target }}
|
||||
|
||||
# Get Metis through cache, or build it.
|
||||
# Install will only run on cache miss.
|
||||
- name: cache metis
|
||||
id: metis-cache
|
||||
if: matrix.mpi == 'par'
|
||||
uses: actions/cache@v2
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.0
|
||||
|
||||
- name: install metis
|
||||
if: matrix.mpi == 'par' && steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.0
|
||||
with:
|
||||
archive: ${{ env.METIS_ARCHIVE }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
|
||||
# MFEM build and test
|
||||
- name: build
|
||||
uses: mfem/github-actions/build-mfem@v2.0
|
||||
with:
|
||||
os: ${{ matrix.os }}
|
||||
target: ${{ matrix.target }}
|
||||
codecov: ${{ matrix.codecov }}
|
||||
mpi: ${{ matrix.mpi }}
|
||||
build-system: ${{ matrix.build-system }}
|
||||
hypre-dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
metis-dir: ${{ env.METIS_TOP_DIR }}
|
||||
mfem-dir: ${{ env.MFEM_TOP_DIR }}
|
||||
|
||||
# Run checks (and only checks) on debug targets
|
||||
- name: checks
|
||||
if: matrix.build-system == 'make' && matrix.target == 'dbg'
|
||||
run: |
|
||||
cd ${{ env.MFEM_TOP_DIR }} && make check
|
||||
|
||||
- name: unit tests
|
||||
if: matrix.build-system == 'make' && matrix.target == 'opt'
|
||||
run: |
|
||||
cd ${{ env.MFEM_TOP_DIR }} && make unittest
|
||||
|
||||
- name: tests
|
||||
if: matrix.build-system == 'make' && matrix.target == 'opt'
|
||||
run: |
|
||||
cd ${{ env.MFEM_TOP_DIR }} && make test
|
||||
|
||||
- name: cmake unit tests
|
||||
if: matrix.build-system == 'cmake'
|
||||
run: |
|
||||
cd ${{ env.MFEM_TOP_DIR }}/build/tests/unit && ctest --output-on-failure
|
||||
|
||||
# Code coverage (process and upload reports)
|
||||
- name: codecov
|
||||
if: matrix.codecov == 'YES'
|
||||
uses: mfem/github-actions/upload-coverage@v2.0
|
||||
with:
|
||||
name: ${{ matrix.os }}-${{ matrix.build-system }}-${{ matrix.target }}-${{ matrix.mpi }}-${{ matrix.hypre-target }}
|
||||
project_dir: ${{ env.MFEM_TOP_DIR }}
|
||||
directories: "fem general linalg mesh"
|
||||
@@ -1,100 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
name: build-analysis
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
- next
|
||||
pull_request:
|
||||
|
||||
env:
|
||||
HYPRE_ARCHIVE: v2.19.0.tar.gz
|
||||
HYPRE_TOP_DIR: hypre-2.19.0
|
||||
METIS_ARCHIVE: metis-4.0.3.tar.gz
|
||||
METIS_TOP_DIR: metis-4.0.3
|
||||
COVERAGE_ENV: mfem-coverage
|
||||
|
||||
jobs:
|
||||
gitignore:
|
||||
runs-on: ubuntu-18.04
|
||||
|
||||
steps:
|
||||
- name: Cancel Previous Runs
|
||||
uses: styfle/cancel-workflow-action@0.9.0
|
||||
with:
|
||||
access_token: ${{ github.token }}
|
||||
|
||||
- name: checkout MFEM
|
||||
uses: actions/checkout@v2
|
||||
with:
|
||||
path: mfem
|
||||
|
||||
- name: Get MPI (Linux)
|
||||
run: |
|
||||
sudo apt-get install mpich libmpich-dev
|
||||
export MAKE_CXX_FLAG="MPICXX=mpic++"
|
||||
|
||||
- name: Cache Hypre Install
|
||||
id: hypre-cache
|
||||
uses: actions/cache@v2
|
||||
with:
|
||||
path: ${{ env.HYPRE_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.HYPRE_TOP_DIR }}-v2.0
|
||||
|
||||
- name: Get Hypre
|
||||
if: steps.hypre-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-hypre@v2.0
|
||||
with:
|
||||
archive: ${{ env.HYPRE_ARCHIVE }}
|
||||
dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
target: int32
|
||||
|
||||
- name: Cache Metis Install
|
||||
id: metis-cache
|
||||
uses: actions/cache@v2
|
||||
with:
|
||||
path: ${{ env.METIS_TOP_DIR }}
|
||||
key: ${{ runner.os }}-build-${{ env.METIS_TOP_DIR }}-v2.0
|
||||
|
||||
- name: Install Metis
|
||||
if: steps.metis-cache.outputs.cache-hit != 'true'
|
||||
uses: mfem/github-actions/build-metis@v2.0
|
||||
with:
|
||||
archive: ${{ env.METIS_ARCHIVE }}
|
||||
dir: ${{ env.METIS_TOP_DIR }}
|
||||
|
||||
# MFEM build and test
|
||||
- name: build-mfem
|
||||
uses: mfem/github-actions/build-mfem@v2.0
|
||||
with:
|
||||
os: ${{ runner.os }}
|
||||
target: optim
|
||||
codecov: NO
|
||||
mpi: parallel
|
||||
build-system: make
|
||||
hypre-dir: ${{ env.HYPRE_TOP_DIR }}
|
||||
metis-dir: ${{ env.METIS_TOP_DIR }}
|
||||
mfem-dir: mfem
|
||||
|
||||
- name: test (no clean)
|
||||
run: |
|
||||
cd mfem && make test-noclean
|
||||
|
||||
- name: gitignore
|
||||
run: |
|
||||
cd mfem/tests/scripts
|
||||
./runtest gitignore
|
||||
@@ -1,110 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
name: repo-check
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
on:
|
||||
push:
|
||||
|
||||
jobs:
|
||||
file-headers-check:
|
||||
runs-on: ubuntu-18.04
|
||||
|
||||
steps:
|
||||
- name: Cancel Previous Runs
|
||||
uses: styfle/cancel-workflow-action@0.9.0
|
||||
with:
|
||||
access_token: ${{ github.token }}
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: copyright check
|
||||
id: copyright
|
||||
run: |
|
||||
./config/githooks/pre-push --copyright
|
||||
|
||||
continue-on-error: true
|
||||
|
||||
- name: license check
|
||||
id: license
|
||||
run: |
|
||||
./config/githooks/pre-push --license
|
||||
continue-on-error: true
|
||||
|
||||
- name: release check
|
||||
id: release
|
||||
run: |
|
||||
./config/githooks/pre-push --release
|
||||
continue-on-error: true
|
||||
|
||||
- name: wrap-up
|
||||
if: steps.copyright.outcome != 'success' || steps.license.outcome != 'success' || steps.release.outcome != 'success'
|
||||
run: |
|
||||
if [[ "${{ steps.copyright.outcome }}" != "success" ]]; then
|
||||
echo "copyright check failed, unroll log for details"
|
||||
fi
|
||||
if [[ "${{ steps.license.outcome }}" != "success" ]]; then
|
||||
echo "license check failed, unroll log for details"
|
||||
fi
|
||||
if [[ "${{ steps.release.outcome }}" != "success" ]]; then
|
||||
echo "release check failed, unroll log for details"
|
||||
fi
|
||||
exit 1
|
||||
|
||||
code-style:
|
||||
runs-on: ubuntu-16.04 # needed for astyle 2.05.1
|
||||
|
||||
steps:
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: get astyle
|
||||
run: |
|
||||
sudo apt-get install astyle=2.05.1-0ubuntu1
|
||||
|
||||
- name: style check
|
||||
run: |
|
||||
./config/githooks/pre-push --style
|
||||
|
||||
documentation:
|
||||
runs-on: ubuntu-18.04
|
||||
|
||||
steps:
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: get doxygen and graphviz
|
||||
run: |
|
||||
sudo apt-get install doxygen graphviz
|
||||
|
||||
- name: build documentation
|
||||
run: |
|
||||
cd tests/scripts
|
||||
./runtest documentation
|
||||
|
||||
branch-history:
|
||||
if: github.ref != 'refs/heads/next' && github.ref != 'refs/heads/master'
|
||||
runs-on: ubuntu-18.04
|
||||
|
||||
steps:
|
||||
- name: checkout mfem
|
||||
uses: actions/checkout@v2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: branch-history
|
||||
run: |
|
||||
git fetch origin master:master
|
||||
git checkout -b gh-actions-branch-history
|
||||
./config/githooks/pre-push --history
|
||||
+40
-188
@@ -9,7 +9,6 @@
|
||||
# Object and library files
|
||||
*.o
|
||||
/libmfem.*
|
||||
/miniapps/common/libmfem-common.*
|
||||
|
||||
# CMake generated files
|
||||
CMakeCache.txt
|
||||
@@ -26,12 +25,9 @@ CMakeFiles/
|
||||
config/_config.hpp
|
||||
config/config.mk
|
||||
config/sample-runs-build.log
|
||||
config/user.mk
|
||||
doc/CodeDocumentation.conf
|
||||
doc/CodeDocumentation.html
|
||||
doc/CodeDocumentation
|
||||
doc/undoc.log
|
||||
doc/warnings.log
|
||||
|
||||
# Temporary files created by the tests.
|
||||
*.stderr
|
||||
@@ -45,19 +41,16 @@ doc/warnings.log
|
||||
|
||||
# Example and miniapp binaries and outputs
|
||||
|
||||
examples/ex[0-9]
|
||||
examples/ex[0-9]p
|
||||
examples/ex[1-9]
|
||||
examples/ex[1-9]p
|
||||
examples/ex1[04-9]
|
||||
examples/ex1[0-9]p
|
||||
examples/ex2[0-9]
|
||||
examples/ex2[0-9]p
|
||||
|
||||
examples/refined.mesh
|
||||
examples/displaced.mesh
|
||||
examples/mesh.*
|
||||
examples/ex5.mesh
|
||||
examples/Example5*
|
||||
examples/ParaView
|
||||
examples/Example9*
|
||||
examples/Example15*
|
||||
examples/Example16*
|
||||
@@ -65,9 +58,6 @@ examples/sphere_refined.*
|
||||
examples/sol.*
|
||||
examples/sol_u.*
|
||||
examples/sol_p.*
|
||||
examples/sol_r.*
|
||||
examples/sol_i.*
|
||||
examples/ex6p-checkpoint.*
|
||||
examples/ex9.mesh
|
||||
examples/ex9-mesh.*
|
||||
examples/ex9-init.*
|
||||
@@ -76,10 +66,6 @@ examples/deformed.*
|
||||
examples/velocity.*
|
||||
examples/elastic_energy.*
|
||||
examples/mode_*
|
||||
examples/ex5-p-*.bp
|
||||
examples/ex9-p-*.bp
|
||||
examples/ex12-p-*.bp
|
||||
examples/ex16-p-*.bp
|
||||
examples/ex16.mesh
|
||||
examples/ex16-mesh.*
|
||||
examples/ex16-init.*
|
||||
@@ -90,73 +76,12 @@ examples/vortex-?-init.*
|
||||
examples/vortex-?-final.*
|
||||
examples/deformation.*
|
||||
examples/pressure.*
|
||||
examples/ex20.dat
|
||||
examples/ex20p_?????.dat
|
||||
examples/gnuplot_ex20.inp
|
||||
examples/gnuplot_ex20p.inp
|
||||
examples/ex21*.mesh
|
||||
examples/ex21*.sol
|
||||
examples/ex21p_*.*
|
||||
examples/ex23-*.gf
|
||||
examples/ex23*.mesh
|
||||
examples/Example23*
|
||||
examples/ex25.mesh
|
||||
examples/ex25-*.gf
|
||||
examples/ex25p-*.*
|
||||
examples/ex28_*
|
||||
examples/ex28p_*
|
||||
examples/flux.*
|
||||
|
||||
examples/amgx/ex1
|
||||
examples/amgx/ex1p
|
||||
examples/amgx/.logamgx
|
||||
examples/amgx/refined.mesh
|
||||
examples/amgx/sol.gf
|
||||
examples/amgx/mesh.*
|
||||
examples/amgx/sol.*
|
||||
|
||||
examples/gingko/ex1
|
||||
examples/gingko/refined.mesh
|
||||
examples/gingko/sol.gf
|
||||
examples/gingko/mesh.*
|
||||
examples/gingko/sol.*
|
||||
|
||||
examples/hiop/ex9
|
||||
examples/hiop/ex9p
|
||||
examples/hiop/ex9.mesh
|
||||
examples/hiop/ex9-mesh.*
|
||||
examples/hiop/ex9-init.*
|
||||
examples/hiop/ex9-final.*
|
||||
|
||||
examples/petsc/ex[1-69]p
|
||||
examples/petsc/ex1[0-1]p
|
||||
examples/petsc/mesh.*
|
||||
examples/petsc/sol.*
|
||||
examples/petsc/sol_p.*
|
||||
examples/petsc/sol_u.*
|
||||
examples/petsc/Example5*
|
||||
examples/petsc/ex9.mesh
|
||||
examples/petsc/ex9-mesh.*
|
||||
examples/petsc/ex9-init.*
|
||||
examples/petsc/ex9-final.*
|
||||
examples/petsc/Example9*
|
||||
examples/petsc/deformed.*
|
||||
examples/petsc/velocity.*
|
||||
examples/petsc/elastic_energy.*
|
||||
examples/petsc/mode_*
|
||||
|
||||
examples/pumi/ex1
|
||||
examples/pumi/ex[126]p
|
||||
examples/pumi/refined.mesh
|
||||
examples/pumi/sol.gf
|
||||
examples/pumi/mesh.*
|
||||
examples/pumi/sol.*
|
||||
examples/pumi/displaced.mesh
|
||||
|
||||
examples/sundials/ex9
|
||||
examples/sundials/ex1[06]
|
||||
examples/sundials/ex9p
|
||||
examples/sundials/ex1[06]p
|
||||
|
||||
examples/sundials/ex9.mesh
|
||||
examples/sundials/ex9-mesh.*
|
||||
examples/sundials/ex9-init.*
|
||||
@@ -171,145 +96,72 @@ examples/sundials/ex16-init.*
|
||||
examples/sundials/ex16-final.*
|
||||
examples/sundials/Example16*
|
||||
|
||||
examples/superlu/ex1p
|
||||
examples/superlu/mesh.*
|
||||
examples/superlu/sol.*
|
||||
examples/petsc/ex[1-69]p
|
||||
examples/petsc/ex10p
|
||||
|
||||
miniapps/adjoint/cvsRoberts_ASAi_dns
|
||||
miniapps/adjoint/adjoint_advection_diffusion
|
||||
examples/petsc/mesh.*
|
||||
examples/petsc/sol.*
|
||||
examples/petsc/sol_p.*
|
||||
examples/petsc/sol_u.*
|
||||
examples/petsc/Example5*
|
||||
examples/petsc/ex9-mesh.*
|
||||
examples/petsc/ex9-init.*
|
||||
examples/petsc/ex9-final.*
|
||||
examples/petsc/Example9*
|
||||
examples/petsc/deformed.*
|
||||
examples/petsc/velocity.*
|
||||
examples/petsc/elastic_energy.*
|
||||
|
||||
examples/pumi/ex1
|
||||
examples/pumi/ex[126]p
|
||||
|
||||
examples/pumi/refined.mesh
|
||||
examples/pumi/sol.gf
|
||||
examples/pumi/mesh.*
|
||||
examples/pumi/sol.*
|
||||
examples/pumi/displaced.mesh
|
||||
|
||||
miniapps/electromagnetics/volta
|
||||
miniapps/electromagnetics/tesla
|
||||
miniapps/electromagnetics/maxwell
|
||||
miniapps/electromagnetics/joule
|
||||
|
||||
miniapps/electromagnetics/Volta-AMR*
|
||||
miniapps/electromagnetics/Tesla-AMR*
|
||||
miniapps/electromagnetics/Maxwell-Parallel*
|
||||
miniapps/electromagnetics/Joule_*
|
||||
|
||||
miniapps/gslib/field-diff
|
||||
miniapps/gslib/field-interp
|
||||
miniapps/gslib/findpts
|
||||
miniapps/gslib/pfindpts
|
||||
|
||||
miniapps/meshing/mobius-strip
|
||||
miniapps/meshing/klein-bottle
|
||||
miniapps/meshing/toroid
|
||||
miniapps/meshing/twist
|
||||
miniapps/meshing/mesh-explorer
|
||||
miniapps/meshing/shaper
|
||||
miniapps/meshing/extruder
|
||||
miniapps/meshing/trimmer
|
||||
miniapps/meshing/mesh-optimizer
|
||||
miniapps/meshing/pmesh-optimizer
|
||||
miniapps/meshing/minimal-surface
|
||||
miniapps/meshing/pminimal-surface
|
||||
miniapps/meshing/polar-nc
|
||||
|
||||
miniapps/meshing/mobius-strip.mesh
|
||||
miniapps/meshing/klein-bottle.mesh
|
||||
miniapps/meshing/toroid-*.mesh
|
||||
miniapps/meshing/twist-*.mesh
|
||||
miniapps/meshing/mesh-explorer.mesh
|
||||
miniapps/meshing/partitioning.txt
|
||||
miniapps/meshing/shaper.mesh
|
||||
miniapps/meshing/extruder.mesh
|
||||
miniapps/meshing/trimmer.mesh
|
||||
miniapps/meshing/optimized*
|
||||
miniapps/meshing/perturbed*
|
||||
miniapps/meshing/polar-nc.mesh
|
||||
|
||||
miniapps/mtop/parheat
|
||||
miniapps/mtop/ParHeat*
|
||||
miniapps/mtop/seqheat
|
||||
miniapps/mtop/SeqHeat*
|
||||
miniapps/performance/ex1
|
||||
miniapps/performance/ex1p
|
||||
|
||||
miniapps/navier/navier_mms
|
||||
miniapps/navier/navier_kovasznay
|
||||
miniapps/navier/navier_kovasznay_vs
|
||||
miniapps/navier/navier_tgv
|
||||
miniapps/navier/navier_shear
|
||||
miniapps/navier/navier_3dfoc
|
||||
miniapps/navier/tgv_out*.txt
|
||||
miniapps/navier/*_output
|
||||
miniapps/performance/refined.mesh
|
||||
miniapps/performance/mesh.*
|
||||
miniapps/performance/sol.*
|
||||
|
||||
miniapps/nurbs/nurbs_ex1
|
||||
miniapps/nurbs/nurbs_ex1p
|
||||
miniapps/nurbs/nurbs_ex11p
|
||||
miniapps/tools/display-basis
|
||||
miniapps/tools/load-dc
|
||||
miniapps/tools/convert-dc
|
||||
|
||||
miniapps/nurbs/ex1
|
||||
miniapps/nurbs/ex1p
|
||||
miniapps/nurbs/ex11p
|
||||
miniapps/nurbs/refined.mesh
|
||||
miniapps/nurbs/mesh.*
|
||||
miniapps/nurbs/sol.*
|
||||
miniapps/nurbs/mode_*
|
||||
miniapps/nurbs/Example1*
|
||||
|
||||
miniapps/performance/ex1
|
||||
miniapps/performance/ex1p
|
||||
miniapps/performance/refined.mesh
|
||||
miniapps/performance/mesh.*
|
||||
miniapps/performance/sol.*
|
||||
|
||||
miniapps/shifted/distance
|
||||
miniapps/shifted/ParaViewDistance
|
||||
miniapps/shifted/diffusion
|
||||
miniapps/shifted/diffusion.mesh
|
||||
miniapps/shifted/diffusion.gf
|
||||
|
||||
miniapps/tools/display-basis
|
||||
miniapps/tools/load-dc
|
||||
miniapps/tools/convert-dc
|
||||
miniapps/tools/lor-transfer
|
||||
miniapps/tools/get-values
|
||||
|
||||
miniapps/toys/automata
|
||||
miniapps/toys/life
|
||||
miniapps/toys/mandel
|
||||
miniapps/toys/rubik
|
||||
miniapps/toys/snake
|
||||
miniapps/toys/lissajous
|
||||
miniapps/toys/mondrian
|
||||
miniapps/toys/snake-init.mesh
|
||||
miniapps/toys/snake-user.mesh
|
||||
miniapps/toys/snake-joined.mesh
|
||||
miniapps/toys/snake-c*.mesh
|
||||
miniapps/toys/automata.gf
|
||||
miniapps/toys/automata.mesh
|
||||
miniapps/toys/rubik-init.mesh
|
||||
miniapps/toys/mandel.mesh
|
||||
miniapps/toys/life.gf
|
||||
miniapps/toys/life.mesh
|
||||
miniapps/toys/lissajous.mesh
|
||||
miniapps/toys/lissajous.gf
|
||||
miniapps/toys/mondrian.mesh
|
||||
|
||||
miniapps/solvers/block-solvers
|
||||
miniapps/solvers/lor_solvers
|
||||
miniapps/solvers/plor_solvers
|
||||
miniapps/solvers/ParaView
|
||||
miniapps/solvers/mesh.*
|
||||
miniapps/solvers/sol.*
|
||||
|
||||
# Unit test binary and outputs
|
||||
tests/unit/output_meshes
|
||||
tests/unit/unit_tests
|
||||
tests/unit/punit_tests
|
||||
tests/unit/sedov_tests_*
|
||||
tests/unit/psedov_tests_*
|
||||
tests/unit/tmop_pa_tests_*
|
||||
tests/unit/ptmop_pa_tests_*
|
||||
tests/unit/ceed_tests
|
||||
|
||||
# Test script output
|
||||
tests/scripts/*.err
|
||||
tests/scripts/*.out
|
||||
tests/scripts/*.msg
|
||||
|
||||
# Other tests
|
||||
tests/convergence/rates
|
||||
tests/convergence/prates
|
||||
tests/par-mesh-format/ex1p
|
||||
|
||||
# VPATH builds
|
||||
build-*/*
|
||||
|
||||
# PETSc automated build
|
||||
petsc-build/*
|
||||
pkg.gitcommit
|
||||
|
||||
-254
@@ -1,254 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# General GitLab pipelines configurations for supercomputers and Linux clusters
|
||||
# at Lawrence Livermore National Laboratory (LLNL). This entire pipeline is
|
||||
# LLNL-specific!
|
||||
|
||||
# We define the following GitLab pipeline variables:
|
||||
#
|
||||
# BUILD_ROOT:
|
||||
# The path to the shared resources between all jobs. For example, external
|
||||
# repositories like 'tests' and 'tpls' are cloned here. Also, 'tpls' is built
|
||||
# once for all targets, so that build happen here. The BUILD_ROOT is unique to
|
||||
# the pipeline, preventing any form of concurrency with other pipelines. This
|
||||
# also means that the BUILD_ROOT directory will never be cleaned.
|
||||
# TODO: add a clean-up mechanism
|
||||
#
|
||||
# REBASELINE:
|
||||
# Defines the default choice for updating the saved baseline results. By default
|
||||
# the baseline can only be updated from the master branch. This variable offers
|
||||
# the option to manually ask for rebaselining from another branch if necessary.
|
||||
#
|
||||
# MFEM_ALLOC_NAME:
|
||||
# On LLNL's quartz, there is only one allocation shared among jobs in order to
|
||||
# save time and resources. This allocation has to be uniquely named so that we
|
||||
# are sure to retrieve it.
|
||||
#
|
||||
# TPLS_REPO & TESTS_REPO:
|
||||
# Git repositories used in the pipeline
|
||||
#
|
||||
# ARTIFACTS_DIR:
|
||||
# Directory used to place artifacts.
|
||||
|
||||
variables:
|
||||
BUILD_ROOT: ${CI_BUILDS_DIR}/MFEM/${CI_PROJECT_NAME}_${CI_COMMIT_REF_SLUG}_${CI_PIPELINE_ID}
|
||||
AUTOTEST_ROOT: ${CI_BUILDS_DIR}/MFEM
|
||||
REBASELINE: "NO"
|
||||
AUTOTEST: "NO"
|
||||
ALLOC_NAME: ${CI_PROJECT_NAME}_ci_${CI_PIPELINE_ID}
|
||||
TPLS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tpls.git
|
||||
TESTS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tests.git
|
||||
AUTOTEST_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/autotest.git
|
||||
MFEM_DATA_REPO: https://github.com/mfem/data.git
|
||||
ARTIFACTS_DIR: artifacts
|
||||
|
||||
# The pipeline is divided into stages. Usually, these are also synchronization
|
||||
# points, however, we use "needs" keyword to express the DAG of jobs for more
|
||||
# efficiency.
|
||||
# - We use setup and setup_baseline phases to download content outside of mfem
|
||||
# directory.
|
||||
# - Allocate/Release is where quartz resources are allocated/released once for all.
|
||||
# - Build and Test is where we build and MFEM for multiple toolchains.
|
||||
# - Baseline_checks gathers baseline-type test suites execution
|
||||
# - Baseline_publish, only available on master, allows to update baseline
|
||||
# results
|
||||
stages:
|
||||
- setup
|
||||
- q_allocate_resources
|
||||
- q_build_and_test
|
||||
- q_release_resources
|
||||
- l_build_and_test
|
||||
- c_build_and_test
|
||||
- setup_baseline
|
||||
- baseline_check
|
||||
- baseline_to_autotest
|
||||
- baseline_publish
|
||||
|
||||
# setup clones the mfem/data repo in ${BUILD_ROOT}. The build_and_test script
|
||||
# then symlinks the repo to the parent directory of the MFEM source directory.
|
||||
# Unit tests that depend on the mfem/data repo will then detect that this
|
||||
# directory is present and be enabled.
|
||||
setup:
|
||||
tags:
|
||||
- shell
|
||||
- quartz
|
||||
stage: setup
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
script:
|
||||
- mkdir -p ${BUILD_ROOT} && cd ${BUILD_ROOT}
|
||||
- if [ ! -d data ]; then git clone ${MFEM_DATA_REPO}; fi
|
||||
needs: []
|
||||
|
||||
# The setup_baseline job in setup stage_baseline doesn't rely on MFEM git repo.
|
||||
# It prepares a pipeline-wide working directory downloading/updating external
|
||||
# repos. TODO: updating tests and tpls is not necessary anymore since pipelines
|
||||
# are now using unique directories so repo are never shared with another
|
||||
# pipeline. This is not memory efficient (we keep a lot of data), hence this
|
||||
# reminder.
|
||||
setup_baseline:
|
||||
tags:
|
||||
- shell
|
||||
- quartz
|
||||
stage: setup_baseline
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
script:
|
||||
- mkdir -p ${BUILD_ROOT} && cd ${BUILD_ROOT}
|
||||
- if [ ! -d "tpls" ]; then git clone ${TPLS_REPO}; fi
|
||||
- if [ ! -d "tests" ]; then git clone ${TESTS_REPO}; fi
|
||||
- cd tpls && git pull && cd ..
|
||||
- cd tests && git pull && cd ..
|
||||
- cd ${AUTOTEST_ROOT}
|
||||
- if [ ! -d "autotest" ]; then git clone ${AUTOTEST_REPO}; fi
|
||||
- cd autotest && git pull && cd ..
|
||||
needs: []
|
||||
|
||||
.build_toss_3_x86_64_ib_script:
|
||||
script:
|
||||
- export THREADS=12
|
||||
- echo ${ALLOC_NAME}
|
||||
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
|
||||
- echo ${JOBID}
|
||||
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) -t 30 -N 1 tests/gitlab/build_and_test
|
||||
|
||||
.build_toss_3_x86_64_ib_corona_script:
|
||||
script:
|
||||
- srun -p mi60 -t 15 -N 1 tests/gitlab/build_and_test
|
||||
|
||||
# Lassen uses a different job scheduler (spectrum lsf) that does not
|
||||
# allow pre-allocation the same way slurm does.
|
||||
# We use pdebug queue on lassen to speed-up the allocation.
|
||||
# However this would not be scalable to multiple builds.
|
||||
.build_blueos_3_ppc64le_ib_script:
|
||||
script:
|
||||
- lalloc 1 -W 30 -q pdebug tests/gitlab/build_and_test
|
||||
|
||||
# Shared script for baseline and sample-run-baseline, the value of BASELINE_TEST
|
||||
# differentiates between the two tests.
|
||||
.baseline_script: &baseline_script |
|
||||
# locals
|
||||
_glob_err=${BASELINE_TEST}.err
|
||||
_base_diff=${BASELINE_TEST}-${SYS_TYPE}.diff
|
||||
_base_patch=${BASELINE_TEST}-${SYS_TYPE}.patch
|
||||
_base_out=${BASELINE_TEST}-${SYS_TYPE}.out
|
||||
# prepare
|
||||
cd ${BUILD_ROOT}
|
||||
ln -snf ${CI_PROJECT_DIR} mfem
|
||||
cd tests
|
||||
mkdir _${BASELINE_TEST} && cd _${BASELINE_TEST}
|
||||
# run
|
||||
srun --nodes=1 -p pdebug ../runtest ../../mfem "${BASELINE_TEST} ${ADDITIONAL_DIR}"
|
||||
# post
|
||||
mkdir ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}
|
||||
if [[ -s ${_glob_err} ]]
|
||||
then
|
||||
echo "ERROR during ${BASELINE_TEST} execution";
|
||||
echo "Here is the ${_glob_err} file content";
|
||||
cat ${_glob_err}
|
||||
cp ${_glob_err} ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/${_glob_err}
|
||||
exit 1;
|
||||
elif [[ ! -f ${_base_patch} && ! -f ${_base_out} ]]
|
||||
then
|
||||
echo "Something went WRONG in ${BASELINE_TEST}:";
|
||||
echo "Either ${_base_patch} or ${_base_out} should exists";
|
||||
exit 1;
|
||||
elif [[ -f ${_base_patch} ]]
|
||||
then
|
||||
echo "${BASELINE_TEST}: Differences found, patch generated"
|
||||
cp ${_base_patch} ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/${_base_patch}
|
||||
elif [[ -f ${_base_out} ]]
|
||||
then
|
||||
echo "${BASELINE_TEST}: Differences found, replacement file generated"
|
||||
cp ${_base_out} ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/${_base_out}
|
||||
fi
|
||||
# _base_diff won't even exist if there is no difference.
|
||||
if [[ -f ${_base_diff} ]]
|
||||
then
|
||||
echo "${BASELINE_TEST}: Relevant differences (filtered diff) ..."
|
||||
cat ${_base_diff}
|
||||
cp ${_base_diff} ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/${_base_diff}
|
||||
# We create a .err file, because that's how we signal that there was a diff.
|
||||
cp ${_base_diff} ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/gitlab-${BASELINE_TEST}-${SYS_TYPE}.err
|
||||
fi
|
||||
if [[ ! -s ${_base_diff} ]]
|
||||
then
|
||||
echo "${BASELINE_TEST}: PASSED"
|
||||
true
|
||||
else
|
||||
echo "${BASELINE_TEST}: FAILED"
|
||||
false
|
||||
fi
|
||||
|
||||
# Actual templates for baseline checks
|
||||
.baselinecheck_mfem:
|
||||
stage: baseline_check
|
||||
variables:
|
||||
BASELINE_TEST: baseline
|
||||
ADDITIONAL_DIR: ${BUILD_ROOT}/tpls
|
||||
script:
|
||||
- *baseline_script
|
||||
artifacts:
|
||||
when: always
|
||||
paths:
|
||||
- ${ARTIFACTS_DIR}
|
||||
allow_failure: true
|
||||
|
||||
.samplebaselinecheck_mfem:
|
||||
stage: baseline_check
|
||||
variables:
|
||||
BASELINE_TEST: sample-runs-baseline
|
||||
ADDITIONAL_DIR: ""
|
||||
script:
|
||||
- *baseline_script
|
||||
timeout: 4h
|
||||
artifacts:
|
||||
when: always
|
||||
paths:
|
||||
- ${ARTIFACTS_DIR}
|
||||
allow_failure: true
|
||||
|
||||
# This job can only be manually triggered on a pipeline for master branch, or if
|
||||
# the pipeline was triggered with REBASELINE="YES"
|
||||
.rebaseline_mfem:
|
||||
stage: baseline_publish
|
||||
rules:
|
||||
- if: '$CI_COMMIT_BRANCH == "master" || $REBASELINE == "YES"'
|
||||
when: manual
|
||||
script:
|
||||
- export PATCH_FILE=${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/baseline-${SYS_TYPE}.patch
|
||||
- export FULL_FILE=${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/baseline-${SYS_TYPE}.out
|
||||
- export DIFF_FILE=${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/baseline-${SYS_TYPE}.diff
|
||||
- cd ${BUILD_ROOT}/tests
|
||||
- |
|
||||
if [[ ! -f "${DIFF_FILE}" ]]
|
||||
then
|
||||
echo "Nothing to be done: no relevant change in baseline"
|
||||
exit 0
|
||||
elif [[ -f "${PATCH_FILE}" ]]
|
||||
then
|
||||
patch "./baseline-${SYS_TYPE}.saved" < "${PATCH_FILE}"
|
||||
elif [[ -f "${FULL_FILE}t" ]]
|
||||
then
|
||||
cp "${FULL_FILE}" "./baseline-${SYS_TYPE}.saved"
|
||||
else
|
||||
echo "File missing: expected ${PATCH_FILE} or ${FULL_FILE}"
|
||||
exit 1
|
||||
fi
|
||||
- git add baseline-${SYS_TYPE}.saved
|
||||
- git commit -m "${SYS_TYPE} rebaselined in GitLab pipeline ${CI_PIPELINE_ID}"
|
||||
- git push origin master
|
||||
|
||||
# The list on jobs is defined in machine-specific files.
|
||||
include:
|
||||
- local: .gitlab/quartz.yml
|
||||
- local: .gitlab/lassen.yml
|
||||
@@ -1,33 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# GitLab pipelines configurations for the Lassen machine at LLNL
|
||||
|
||||
.on_lassen:
|
||||
tags:
|
||||
- shell
|
||||
- lassen
|
||||
rules:
|
||||
- if: '$CI_COMMIT_BRANCH =~ /_lnone/ || $ON_LASSEN == "OFF"' #run except if ...
|
||||
when: never
|
||||
- when: on_success
|
||||
|
||||
# Spack helped builds
|
||||
# Generic lassen build job, extending build script
|
||||
.build_and_test_on_lassen:
|
||||
extends: [.build_blueos_3_ppc64le_ib_script, .on_lassen]
|
||||
stage: l_build_and_test
|
||||
needs: [setup]
|
||||
|
||||
opt_mpi_cuda_xl_16_1_1_8:
|
||||
variables:
|
||||
SPEC: "%xl@16.1.1.8 +mpi +cuda cuda_arch=sm_70"
|
||||
extends: .build_and_test_on_lassen
|
||||
@@ -1,166 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# GitLab pipelines configurations for the Quartz machine at LLNL
|
||||
|
||||
.on_quartz:
|
||||
tags:
|
||||
- shell
|
||||
- quartz
|
||||
rules:
|
||||
# Don’t run quartz jobs if...
|
||||
- if: '$CI_COMMIT_BRANCH =~ /_qnone/ || $ON_QUARTZ == "OFF"'
|
||||
when: never
|
||||
# Don’t run autotest update if...
|
||||
- if: '$CI_JOB_NAME =~ /update_autotest/ && $AUTOTEST != "YES"'
|
||||
when: never
|
||||
# Don’t run autotest update if...
|
||||
- if: '$CI_JOB_NAME =~ /q_report/ && $AUTOTEST != "YES"'
|
||||
when: never
|
||||
# Report success on success status
|
||||
- if: '$CI_JOB_NAME =~ /q_report_success/ && $AUTOTEST == "YES"'
|
||||
when: on_success
|
||||
# Report failure on failure status
|
||||
- if: '$CI_JOB_NAME =~ /q_report_failure/ && $AUTOTEST == "YES"'
|
||||
when: on_failure
|
||||
# Always release resources
|
||||
- if: '$CI_JOB_NAME =~ /release_resources/'
|
||||
when: always
|
||||
# Default is to run if previous stage succeeded
|
||||
- when: on_success
|
||||
|
||||
# Allocate
|
||||
q_allocate_resources:
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
extends: .on_quartz
|
||||
stage: q_allocate_resources
|
||||
script:
|
||||
- salloc --exclusive --nodes=1 --partition=pdebug --time=30 --no-shell --job-name=${ALLOC_NAME}
|
||||
timeout: 6h
|
||||
|
||||
# Release
|
||||
q_release_resources:
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
extends: .on_quartz
|
||||
stage: q_release_resources
|
||||
script:
|
||||
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
|
||||
- ([[ -n "${JOBID}" ]] && scancel ${JOBID})
|
||||
|
||||
# Release
|
||||
q_report_success:
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
extends: .on_quartz
|
||||
stage: q_release_resources
|
||||
script:
|
||||
- echo "Can only run if all the quartz jobs passed"
|
||||
- rundir="gitlab/$(date +%Y-%m-%d)-github-${CI_COMMIT_REF_SLUG}"
|
||||
- cd ${AUTOTEST_ROOT}/autotest && git pull
|
||||
- mkdir -p ${rundir}
|
||||
- echo "The Quartz jobs were successful" > ${rundir}/gitlab.out
|
||||
- git add ${rundir}
|
||||
- git commit -am "Gitlab CI log for baseline on quartz with intel ($(date +%Y-%m-%d))"
|
||||
- git push origin master
|
||||
|
||||
q_report_failure:
|
||||
variables:
|
||||
GIT_STRATEGY: none
|
||||
extends: .on_quartz
|
||||
stage: q_release_resources
|
||||
script:
|
||||
- echo "Runs if there was at least one failure on quartz"
|
||||
- rundir="gitlab/$(date +%Y-%m-%d)-github-${CI_COMMIT_REF_SLUG}"
|
||||
- cd ${AUTOTEST_ROOT}/autotest && git pull
|
||||
- mkdir -p ${rundir}
|
||||
- echo "There was an error while running CI on Quartz" > ${rundir}/gitlab.err
|
||||
- cp ${rundir}/gitlab.err ${rundir}/autotest-email.html
|
||||
- git add ${rundir}
|
||||
- git commit -am "Gitlab CI log for baseline on quartz with intel ($(date +%Y-%m-%d))"
|
||||
- git push origin master
|
||||
|
||||
# Spack helped builds
|
||||
# Generic quartz build job, extending build script
|
||||
.build_and_test_on_quartz:
|
||||
extends: [.build_toss_3_x86_64_ib_script, .on_quartz]
|
||||
stage: q_build_and_test
|
||||
needs: [setup]
|
||||
|
||||
# Build MFEM
|
||||
debug_ser_gcc_4_9_3:
|
||||
variables:
|
||||
SPEC: "%gcc@4.9.3 +debug~mpi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
debug_ser_gcc_6_1_0:
|
||||
variables:
|
||||
SPEC: "%gcc@6.1.0 +debug~mpi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
debug_par_gcc_6_1_0:
|
||||
variables:
|
||||
SPEC: "%gcc@6.1.0 +debug+mpi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_ser_gcc_6_1_0:
|
||||
variables:
|
||||
SPEC: "%gcc@6.1.0 ~mpi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_6_1_0:
|
||||
variables:
|
||||
SPEC: "%gcc@6.1.0"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_6_1_0_sundials:
|
||||
variables:
|
||||
SPEC: "%gcc@6.1.0 +sundials"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_6_1_0_petsc:
|
||||
variables:
|
||||
SPEC: "%gcc@6.1.0 +petsc ^petsc+mumps"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
opt_par_gcc_6_1_0_pumi:
|
||||
variables:
|
||||
SPEC: "%gcc@6.1.0 +pumi"
|
||||
extends: .build_and_test_on_quartz
|
||||
|
||||
# Baseline
|
||||
baselinecheck_mfem_intel_quartz:
|
||||
extends: [.baselinecheck_mfem, .on_quartz]
|
||||
needs: [setup_baseline]
|
||||
|
||||
update_autotest:
|
||||
extends: [.on_quartz]
|
||||
needs: [baselinecheck_mfem_intel_quartz]
|
||||
stage: baseline_to_autotest
|
||||
script:
|
||||
- rundir="quartz/$(date +%Y-%m-%d)-github-${CI_COMMIT_REF_SLUG}"
|
||||
- cd ${AUTOTEST_ROOT}/autotest && git pull
|
||||
- mkdir -p ${rundir}
|
||||
- cp ${CI_PROJECT_DIR}/${ARTIFACTS_DIR}/* ${rundir}
|
||||
# We create an autotest-email.html file, because that's how we signal that there was a diff (temporary).
|
||||
- |
|
||||
if [[ -f ${rundir}/*.err ]]
|
||||
then
|
||||
cp ${rundir}/*.err ${rundir}/autotest-email.html
|
||||
fi
|
||||
- git add ${rundir}
|
||||
- git commit -am "Gitlab CI log for baseline on quartz with intel ($(date +%Y-%m-%d))"
|
||||
- git push origin master
|
||||
|
||||
baselinepublish_mfem_quartz:
|
||||
extends: [.on_quartz, .rebaseline_mfem]
|
||||
needs: [baselinecheck_mfem_intel_quartz]
|
||||
+263
@@ -0,0 +1,263 @@
|
||||
sudo: false
|
||||
|
||||
language: cpp
|
||||
|
||||
matrix:
|
||||
include:
|
||||
#
|
||||
# Linux
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
env: DEBUG=YES
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
env: DEBUG=NO
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
addons:
|
||||
apt:
|
||||
# sources:
|
||||
# - ubuntu-toolchain-r-test
|
||||
packages:
|
||||
# GCC 4.9
|
||||
# - g++-4.9
|
||||
# MPICH
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
# OpenMPI
|
||||
# - openmpi-bin
|
||||
# - libopenmpi-dev
|
||||
env: DEBUG=YES
|
||||
MPI=YES
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
NPROCS=2
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
- os: linux
|
||||
compiler: gcc
|
||||
addons:
|
||||
apt:
|
||||
# sources:
|
||||
# - ubuntu-toolchain-r-test
|
||||
packages:
|
||||
# GCC 4.9
|
||||
# - g++-4.9
|
||||
# MPICH
|
||||
- mpich
|
||||
- libmpich-dev
|
||||
# OpenMPI
|
||||
# - openmpi-bin
|
||||
# - libopenmpi-dev
|
||||
env: DEBUG=NO
|
||||
MPI=YES
|
||||
CODECOV=YES
|
||||
MFEM_TEST_TARGET=test
|
||||
NPROCS=2
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
# Mac OS X
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=YES
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=NO
|
||||
MPI=NO
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=test
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=YES
|
||||
MPI=YES
|
||||
CODECOV=NO
|
||||
MFEM_TEST_TARGET=check
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
#
|
||||
- os: osx
|
||||
# osx_image: xcode7.3
|
||||
compiler: clang
|
||||
env: DEBUG=NO
|
||||
MPI=YES
|
||||
CODECOV=YES
|
||||
MFEM_TEST_TARGET=test
|
||||
NPROCS=4
|
||||
TMPDIR=/tmp
|
||||
cache:
|
||||
directories:
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/lib
|
||||
- $TRAVIS_BUILD_DIR/../hypre-2.10.0b/src/hypre/include
|
||||
- $TRAVIS_BUILD_DIR/../metis-4.0
|
||||
- $HOME/local-cached
|
||||
before_cache:
|
||||
- cd $TRAVIS_BUILD_DIR/../metis-4.0;
|
||||
mv libmetis.a ..; rm -rf *; mv ../libmetis.a .
|
||||
|
||||
|
||||
before_install:
|
||||
# No addon for brew yet, have to install OSX packages this way.
|
||||
# - if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
|
||||
# brew install open-mpi;
|
||||
# fi
|
||||
|
||||
# On Mac OS X, build and cache OpenMPI 2.1.1:
|
||||
- if [ $TRAVIS_OS_NAME == "osx" ] && [ $MPI == "YES" ]; then
|
||||
if [ ! -e $HOME/local-cached/bin/mpicc ]; then
|
||||
mkdir -p $HOME/builds && cd $HOME/builds &&
|
||||
wget https://www.open-mpi.org/software/ompi/v2.1/downloads/openmpi-2.1.1.tar.bz2 &&
|
||||
tar jxf openmpi-2.1.1.tar.bz2 &&
|
||||
mkdir openmpi-build && cd openmpi-build &&
|
||||
../openmpi-2.1.1/configure --prefix=$HOME/local-cached &&
|
||||
make -j3 all && make install;
|
||||
fi;
|
||||
PATH=$HOME/local-cached/bin:$PATH;
|
||||
cd $TRAVIS_BUILD_DIR;
|
||||
fi
|
||||
|
||||
# Update environment to find g++ 4.9 installation first.
|
||||
# - if [ $TRAVIS_OS_NAME == "linux" ]; then
|
||||
# mkdir -p latest-gcc-symlinks;
|
||||
# ln -s /usr/bin/g++-4.9 latest-gcc-symlinks/g++;
|
||||
# ln -s /usr/bin/gcc-4.9 latest-gcc-symlinks/gcc;
|
||||
# ln -s /usr/bin/gcov-4.9 latest-gcc-symlinks/gcov;
|
||||
# export PATH=$PWD/latest-gcc-symlinks:$PATH;
|
||||
# fi
|
||||
|
||||
# Install tool to upload code coverage reports to coveralls.io
|
||||
- if [ "$CODECOV" == "YES" ]; then
|
||||
export PYTHONUSERBASE=$HOME/local;
|
||||
pip install --user cpp-coveralls;
|
||||
pip install --user pyyaml;
|
||||
PATH=$HOME/local/bin:$PATH;
|
||||
fi
|
||||
|
||||
install:
|
||||
# Set MPI compilers, print compiler version
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ "$TRAVIS_OS_NAME" == "linux" ]; then
|
||||
export MPICH_CC="$CC";
|
||||
export MPICH_CXX="$CXX";
|
||||
else
|
||||
export OMPI_CC="$CC";
|
||||
export OMPI_CXX="$CXX";
|
||||
mpic++ --showme:version;
|
||||
fi;
|
||||
mpic++ -v;
|
||||
else
|
||||
$CXX -v;
|
||||
fi
|
||||
|
||||
# Back out of the mfem directory to install the libraries
|
||||
- cd ..
|
||||
|
||||
# hypre
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e hypre-2.10.0b/src/hypre/lib/libHYPRE.a ]; then
|
||||
wget https://computation.llnl.gov/project/linear_solvers/download/hypre-2.10.0b.tar.gz --no-check-certificate;
|
||||
rm -rf hypre-2.10.0b;
|
||||
tar xvzf hypre-2.10.0b.tar.gz;
|
||||
cd hypre-2.10.0b/src;
|
||||
./configure --disable-fortran --without-fei CC=mpicc CXX=mpic++;
|
||||
make -j3;
|
||||
cd ../..;
|
||||
else
|
||||
echo "Reusing cached hypre-2.10.0b/";
|
||||
fi;
|
||||
else
|
||||
echo "Serial build, not using hypre";
|
||||
fi
|
||||
|
||||
# METIS
|
||||
- if [ $MPI == "YES" ]; then
|
||||
if [ ! -e metis-4.0/libmetis.a ]; then
|
||||
wget http://glaros.dtc.umn.edu/gkhome/fetch/sw/metis/OLD/metis-4.0.3.tar.gz;
|
||||
tar xvzf metis-4.0.3.tar.gz;
|
||||
make -j3 -C metis-4.0.3/Lib CC="$CC" OPTFLAGS="-O2";
|
||||
rm -rf metis-4.0;
|
||||
mv metis-4.0.3 metis-4.0;
|
||||
else
|
||||
echo "Reusing cached metis-4.0/";
|
||||
fi;
|
||||
fi
|
||||
|
||||
script:
|
||||
# Compiler
|
||||
- if [ $MPI == "YES" ]; then
|
||||
export MYCXX=mpic++;
|
||||
else
|
||||
export MYCXX="$CXX";
|
||||
fi
|
||||
|
||||
# Print the compiler version
|
||||
- $MYCXX -v
|
||||
|
||||
# Set some variables
|
||||
- cd $TRAVIS_BUILD_DIR;
|
||||
CPPFLAGS="";
|
||||
SKIP_TEST_DIRS="";
|
||||
if [ "$CODECOV" == "YES" ]; then
|
||||
CPPFLAGS="--coverage -g";
|
||||
fi;
|
||||
if [ "$CXX" == "clang++" ]; then
|
||||
export MFEM_PERF_SW=clang;
|
||||
fi
|
||||
|
||||
# Configure the library
|
||||
- make config MFEM_USE_MPI=$MPI MFEM_DEBUG=$DEBUG MFEM_CXX="$MYCXX"
|
||||
MFEM_MPI_NP=$NPROCS CPPFLAGS="$CPPFLAGS"
|
||||
# Show the configuration
|
||||
- make info
|
||||
# Build the library
|
||||
- make -j3
|
||||
# Build the examples and the miniapps
|
||||
- make -j3 all
|
||||
# Run tests
|
||||
- make $MFEM_TEST_TARGET SKIP_TEST_DIRS="$SKIP_TEST_DIRS"
|
||||
|
||||
after_success:
|
||||
- if [ "$CODECOV" == "YES" ]; then
|
||||
coveralls --include fem --include general --include linalg --include
|
||||
mesh --exclude /usr --gcov-options '\-lp' --root $TRAVIS_BUILD_DIR;
|
||||
fi
|
||||
@@ -5,924 +5,14 @@
|
||||
| | | | | || _|| __/| | | | | |
|
||||
|_| |_| |_||_| \___||_| |_| |_|
|
||||
|
||||
https://mfem.org
|
||||
http://mfem.org
|
||||
|
||||
|
||||
Version 4.2.1 (development)
|
||||
Version 3.4.1 (development)
|
||||
===========================
|
||||
- Added initial support for GPU-accelerated versions of PETSc that works with
|
||||
MFEM_USE_CUDA if PETSc has been configured with CUDA support. Examples 1 and 9
|
||||
in the examples/petsc directory have been modified to work with --device cuda.
|
||||
Examples with GAMG (ex1p) and SLEPc (ex11p) are also provided.
|
||||
|
||||
- Memory management:
|
||||
* Added method Device::SetMemoryTypes that can be used to change the default
|
||||
host and device MemoryTypes before Device setup.
|
||||
* In class MemoryManager, added methods GetDualMemoryType and
|
||||
SetDualMemoryType; dual MemoryTypes are used to determine the second
|
||||
MemoryType (host or device) when only one MemoryType is specified in methods
|
||||
of class Memory.
|
||||
* Added Memory constructor for setting both the host and device MemoryTypes.
|
||||
* Switched the default behavior of device memory allocations so that they
|
||||
are deferred until the device pointer is needed.
|
||||
* Added a second Umpire device MemoryType, DEVICE_UMPIRE_2, with
|
||||
corresponding allocator that can be set with the method
|
||||
MemoryManager::SetUmpireDevice2AllocatorName.
|
||||
* Added HOST_PINNED MemoryType and a pinned host allocator for CUDA and HIP.
|
||||
|
||||
- Added support for Caliper: a library to integrate performance profiling
|
||||
capabilities into applications. See examples/caliper for more details.
|
||||
|
||||
- Added support for explicit vectorization in the high-performance templated
|
||||
code for Fujitsu's A64FX ARM microprocessor architecture.
|
||||
|
||||
- Added AlgebraicCeedSolver that does matrix-free algebraic p-multigrid for
|
||||
diffusion problems with the Ceed backend.
|
||||
|
||||
- Introduced new options for the mesh-explorer miniapp to visualize the actual
|
||||
element attributes in parallel meshes while retaining the visualization of
|
||||
the domain decomposition.
|
||||
|
||||
- Introduced solver interface for linear problems with constraints, a few
|
||||
concrete solvers that implement the interface, and a demonstration of their
|
||||
use in Example 28(p), which solves an elasticity problem with zero normal
|
||||
displacement (but allowed tangential displacement) on two boundaries.
|
||||
|
||||
- Added high-order matrix-free auxiliary Maxwell solver for H(curl) problems,
|
||||
as described in Barker and Kolev 2020 (https://doi.org/10.1002/nla.2348). See
|
||||
Example 3p and linalg/auxiliary.?pp.
|
||||
|
||||
- Added a new miniapp block-solvers that compares the performance of various
|
||||
solvers for mixed finite element discretization of the second order scalar
|
||||
elliptic equations. Currently available solvers in the miniapp include a
|
||||
block-diagonal preconditioner that is based on approximate Schur complement
|
||||
(implemented in ex5p), and a newly implemented solver DivFreeSolver, which
|
||||
exploits a multilevel decomposition of the Raviart-Thomas space and its
|
||||
divergence-free subspace. See the miniapps/solvers directory for more details.
|
||||
|
||||
- Added a new miniapp for computing (signed) distance functions to a point
|
||||
source or zero level set. See miniapps/shifted/distance.cpp.
|
||||
|
||||
- Added matrix-free GPU-enabled implementations of GradientInterpolator and
|
||||
IdentityInterpolator.
|
||||
|
||||
- Added interface to MUMPS direct solver. Its usage is demonstrated in ex25p.
|
||||
See http://mumps.enseeiht.fr/ for more details. Supported versions >= 5.1.1.
|
||||
|
||||
- Added three ESDIRK time integrators: implicit trapezoid rule, L-stable
|
||||
ESDIRK-32, and A-stable ESDIRK-33.
|
||||
|
||||
- Introduced a new non-conforming mesh format that fixes known inconsistencies
|
||||
of legacy "MFEM mesh v1.1" NC format and works consistently in both serial and
|
||||
parallel. ParMesh::ParPrint can now print non-conforming AMR meshes that can
|
||||
be used to restart a parallel AMR computation. Example 6p has been extended to
|
||||
demonstrate restarting from a previously saved checkpoint. Note that parallel
|
||||
NC data files are compatible with serial code, e.g., can be viewed with serial
|
||||
GLVis. Loading of legacy NC mesh files is still supported.
|
||||
|
||||
- Added support for 1D non-conforming meshes (which can be useful for parallel
|
||||
load balancing and derefinement).
|
||||
|
||||
- Added a "scaled Jacobian" visualization option in the Mesh Explorer miniapp to
|
||||
help identify elements with poor mesh quality.
|
||||
|
||||
- Added support for the "BR2" discontinuous Galerkin discretization for
|
||||
diffusion via DGDiffusionBR2Integrator (see Example 14/14p).
|
||||
|
||||
- Generalized the Multigrid class to support non-geometric multigrid. The
|
||||
previous functionality, based on FiniteElementSpaceHierarchy, is now available
|
||||
in the derived class GeometricMultigrid.
|
||||
|
||||
- Upgraded the Catch unit test framework from version 2.13.0 to version 2.13.2.
|
||||
|
||||
- The TMOP mesh optimization algorithms were extended to GPU:
|
||||
- QualityMetric #1, #2, #7 and #77 are available in 2D, #302, #303, #315
|
||||
and #321 in 3D
|
||||
- Both AnalyticAdaptTC and DiscreteAdaptTC TargetConstructor are available
|
||||
- Kernels for normalization and limiting have been added
|
||||
- The AdvectorCG now also supports AssemblyLevel::PARTIAL
|
||||
|
||||
- Added a new command line boolean option (`--all`) to the unit tests to launch
|
||||
*all* non-regression tests.
|
||||
|
||||
- Added support for different modes of QuadratureInterpolator on GPU.
|
||||
The layout (QVectorLayout::byNODES|byVDIM) and the tensor products modes can
|
||||
be enabled before calling the Mult, Values, Derivatives, PhysDerivatives and
|
||||
Determinants methods.
|
||||
|
||||
- Implemented a filter method for the Navier miniapp to stabilize highly
|
||||
turbulent flows in direct numerical simulation.
|
||||
|
||||
- Added HIP support to the CMake build system.
|
||||
|
||||
- Added support for reading high-order Lagrange meshes in VTK format. Arbitrary-
|
||||
orders and all element types are supported. See the VTK blog for more info:
|
||||
https://blog.kitware.com/wp-content/uploads/2018/09/Source_Issue_43.pdf.
|
||||
|
||||
- Added support for reading VTK meshes in XML format.
|
||||
|
||||
- Added partial assembly and device support to Example 25/25p, with diagonal
|
||||
preconditioning.
|
||||
|
||||
- Implemented a variable step-size IMEX (VSSIMEX) method for the Navier miniapp.
|
||||
|
||||
- Added new mesh quality metrics and improved the untangling capabilities of the
|
||||
TMOP-based mesh optimization algorithms.
|
||||
|
||||
- Added convective and skew-symmetric integrators for the nonlinear term in the
|
||||
Navier-Stokes equations.
|
||||
|
||||
- Added new miniapp directory mtop/ with optimization-oriented block parametric
|
||||
non-linear form and abstract integrators. Two new miniapps, ParHeat and
|
||||
SeqHeat, demonstrate parallel and sequential implementation of gradients
|
||||
evaluation for linear diffusion with discrete density.
|
||||
|
||||
- Changed the interface for the error estimator.
|
||||
|
||||
- Implemented the Kelly error indicator for scalar-valued problems, supported
|
||||
in serial and parallel builds.
|
||||
|
||||
- Added new classes DenseSymmetricMatrix and SymmetricMatrixCoefficient for
|
||||
efficient evaluation of symmetric matrix coefficients. This replaces the now
|
||||
deprecated EvalSymmetric in MatrixCoefficient. Added DiagonalMatrixCoefficient
|
||||
for clarity, which is a typedef of VectorCoefficient.
|
||||
|
||||
- Added support for AMG preconditioners for non-symmetric systems (e.g.
|
||||
advection-dominated problems) using hypre's approximate ideal restriction
|
||||
(AIR) AMG. Requires hypre version 2.14.0 or newer. Usage is illustrated in
|
||||
example 9/9p.
|
||||
|
||||
- Implemented an adaptive linear solver tolerance option for NewtonSolver based
|
||||
on the algorithm of Eisenstat and Walker.
|
||||
|
||||
- Added support for nonscalar coefficient with VectorDiffusionIntegrator.
|
||||
|
||||
- Extending support for L2 basis functions using MapTypes VALUE and INTEGRAL in
|
||||
linear interpolators and GridFunction "GetValue" methods.
|
||||
|
||||
- Variable order spaces, p- and hp-refinement. This is the initial (serial)
|
||||
support for variable-order FiniteElementCollection and FiniteElementSpace.
|
||||
The new method FiniteElementSpace::SetElementOrder can be called to set an
|
||||
arbitrary order for each mesh element. The conforming interpolation matrix
|
||||
will now automatically constrain p- and hp- interfaces, enabling general
|
||||
hp-refinement in both 2D and 3D, on uniform or mixed NC meshes. Support for
|
||||
parallel variable-order spaces will follow shortly.
|
||||
|
||||
- Added support for creating refined meshes for all element types (e.g. by
|
||||
splitting high-order elements into low-order refined elements), including
|
||||
mixed meshes. The LOR Transfer miniapp (miniapps/tools/lor-transfer.cpp) now
|
||||
supports meshes with any element geometry.
|
||||
|
||||
- Testing improvements:
|
||||
* Transitioned from Travis to GitHub Action for testing/CI on GitHub.
|
||||
* Effectively remove Travis from CI.
|
||||
* Use Spack (and Uberenv) to automate TPL building in LLNL GitLab tests.
|
||||
* Added a set of suggested git hooks for developers in config/githooks.
|
||||
|
||||
- Added new miniapps demonstrating: 1) the use of GSLIB for overlapping grids,
|
||||
see gslib/schwarz_ex1, and 2) coupling different physics in different domains,
|
||||
see navier/cht. Note that gslib v1.0.7 is require (see INSTALL for details).
|
||||
|
||||
- Added a new, very simple example (ex0 and parallel version ex0p). This
|
||||
example solves a simple Poisson problem using H1 elements (the same problem as
|
||||
ex1), but is intended to be extremely simple and approachable for new users.
|
||||
|
||||
- Meshes consisting of any type of elements (including mixed meshes) can be
|
||||
converted to all-simplex meshes using Mesh::MakeSimplicial.
|
||||
|
||||
- Several of the mesh constructors (creating Cartesian meshes, refined (LOR)
|
||||
meshes, simplex meshes, etc.) are now available as "named constructors", e.g.
|
||||
Mesh::MakeCartesian2D or Mesh::MakeRefined. The legacy constructors are marked
|
||||
as deprecated.
|
||||
|
||||
- Added support for creating periodic meshes with Mesh::MakePeriodic. The
|
||||
requisite periodic vertex mappings can be created with
|
||||
Mesh::CreatePeriodicVertexMapping.
|
||||
|
||||
- Added support for transferring dual fields between high-order and low-order
|
||||
refined finite element spaces using the transposed versions of the
|
||||
L2ProjectionGridTransfer operators. This functionality is illustrated in the
|
||||
lor-transfer miniapp.
|
||||
|
||||
- Improved interface for using the Ginkgo library, including: support for matrix-
|
||||
free operators in Ginkgo solvers, new wrappers for Ginkgo preconditioners, HIP
|
||||
support, and reduction of unnecessary data copies.
|
||||
|
||||
- Added initial support for hypre's mixed integer (mixedint) capability, which
|
||||
uses different data types for local and global indices in order to save memory
|
||||
in large problems. This capability requires that hypre was configured with the
|
||||
--enable-mixedint option. Note that this option is currently tested only in
|
||||
ex1p and may not work in more general settings.
|
||||
|
||||
- Added support for transferring fields (primary and dual) between high-order
|
||||
and low-order refined H1 finite element spaces using the
|
||||
L2ProjectionH1GridTransfer operators. This functionality is demonstrated
|
||||
through the lor-transfer miniapp when run with the -h1 option.
|
||||
|
||||
- Added new functionality for constructing low-order refined discretizations and
|
||||
solvers, see the LORDiscretization and LORSolver classes. A new basis type for
|
||||
H(curl) and H(div) spaces is introduced to give spectral equivalence. This
|
||||
functionality is illustrated in the LOR solvers miniapp in miniapps/solvers.
|
||||
|
||||
- Added sample meshes in the `data` subdirectory showing the reference elements
|
||||
of the six currently supported element types; ref-segment.mesh,
|
||||
ref-triangle.mesh, ref-square.mesh, ref-tetrahedron.mesh, ref-cube.mesh, and
|
||||
ref-prism.mesh.
|
||||
|
||||
- Added a high-order extension of the shifted boundary method to solve PDEs on
|
||||
non body-fitted meshes. This is illustrated in the new Shifted Diffusion
|
||||
miniapp, see miniapps/shifted/diffusion.cpp.
|
||||
|
||||
- Added makefile rule to generate TAGS table for vi or Emacs users.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- Added an abstract interface `mfem::FaceRestriction` for `H1FaceRestriction`
|
||||
and `L2FaceRestriction`.
|
||||
In order to conform with the semantic of `MultTranspose` in `mfem::Operator`,
|
||||
`mfem::FaceRestriction::MultTranspose` now sets instead of adding values, and
|
||||
`mfem::FaceRestriction::AddMultTranspose` should replace previous calls to
|
||||
`mfem::FaceRestriction::MultTranspose`.
|
||||
|
||||
libCEED integration improvements
|
||||
--------------------------------
|
||||
- Refactor the libCEED integration
|
||||
|
||||
- Add support for VectorCoefficient with libCEED backends.
|
||||
|
||||
- Add support for ConvectionIntegrator, and VectorConvectionNLFIntegrator with
|
||||
libCEED backends.
|
||||
|
||||
|
||||
Version 4.2, released on October 30, 2020
|
||||
=========================================
|
||||
|
||||
High-Performance Computing
|
||||
--------------------------
|
||||
- Added support for explicit vectorization in the high-performance templated
|
||||
code, which can now take advantage of specific classes on the following
|
||||
architectures:
|
||||
* x86 (SSE/AVX/AVX2/AVX512),
|
||||
* Power8 & Power9 (VSX),
|
||||
* BG/Q (QPX).
|
||||
These are disabled by default, but can be enabled with MFEM_USE_SIMD=YES.
|
||||
See the new file linalg/simd.hpp and the new directory linalg/simd.
|
||||
|
||||
- Added an Element Assembly mode compatible with GPU device execution for H1 and
|
||||
L2 spaces in the mass, convection, diffusion, transpose, and the face DG trace
|
||||
integrators. See option '-ea' in Example 9. When enabled, this assembly level
|
||||
stores independent dense matrices for the elements, and independent dense
|
||||
matrices for the faces in the DG case.
|
||||
|
||||
- Added a Full Assembly mode compatible with GPU device execution. This assembly
|
||||
level builds on top of the Element Assembly kernels to compute a global sparse
|
||||
matrix. All integrators supported by element assembly are also supported by
|
||||
full assembly. See the '-fa' option in Example 9.
|
||||
|
||||
- Optimized the AMD/HIP kernel support and enabled HIP support in the libCEED
|
||||
integration. This is now available via the "ceed-hip" device backend.
|
||||
|
||||
- Improved the libCEED integration to support:
|
||||
* AssemblyLevel::NONE for Mass, Diffusion, VectorMass, and VectorDiffusion
|
||||
Integrators. This level computes the full operator evaluation "on the fly".
|
||||
* VectorMassIntegrator and VectorDiffusionIntegrator.
|
||||
* All types of (scalar) Coefficients.
|
||||
|
||||
- Added partial assembly / device support for:
|
||||
* H(div) bilinear forms and VectorFEDivergenceIntegrator.
|
||||
* BlockOperator, see the updated Example 5.
|
||||
* Complex operators, see the updated Example 22.
|
||||
* Chebyshev accelerated polynomial smoother.
|
||||
* Convergent diagonal preconditioning on non-conforming adaptively refined
|
||||
meshes, see Example 6/6p.
|
||||
|
||||
- Added CUDA support for:
|
||||
* Sparse matrix-vector multiplication with cuSPARSE,
|
||||
* SUNDIALS ODE integrators, see updated SUNDIALS modification of Example 9/9p.
|
||||
|
||||
Linear and nonlinear solvers
|
||||
----------------------------
|
||||
- Added a new solver class for simple integration with NVIDIA's multigrid
|
||||
library, AmgX. The AmgX class is designed to work as a standalone solver or
|
||||
preconditioner for existing MFEM solvers. It uses MFEM's sparse matrix format
|
||||
for serial runs and the HypreParMatrix format for parallel runs. The new
|
||||
solver may be configured to run with one GPU per MPI rank or with more MPI
|
||||
ranks than GPUs. In the latter case, matrices and vectors are consolidated to
|
||||
ranks communicating with the GPUs and the solution is then broadcasted.
|
||||
|
||||
Although CUDA is required to build, the AmgX support is compatible with the
|
||||
MFEM CPU device configuration. The examples/amgx folder illustrates how to
|
||||
integrate AmgX in existing MFEM applications. The AmgX solver class is
|
||||
partially based on: "AmgXWrapper: An interface between PETSc and the NVIDIA
|
||||
AmgX library", by Pi-Yueh Chuang and Lorena A. Barba, doi:10.21105/joss.00280.
|
||||
|
||||
- Added initial support for geometric h- and p-multigrid preconditioners for
|
||||
matrix-based and matrix-free discretizations with basic GPU capability, see
|
||||
Example 26/26p.
|
||||
|
||||
- Added support for the CVODES package in SUNDIALS which provides ODE solvers
|
||||
with sensitivity analysis capabilities. See the CVODESSolver class and the new
|
||||
adjoint miniapps in the miniapps/adjoint directory.
|
||||
|
||||
- Added an interface to the MKL CPardiso solver, an MPI-parallel sparse direct
|
||||
solver developed by Intel. See Example 11p for an illustration of its usage.
|
||||
|
||||
- Added support for the SLEPc eigensolver package, https://slepc.upv.es.
|
||||
|
||||
- Upgraded SuperLU interface to use SuperLU_DIST 6.3.1. Added a simple SuperLU
|
||||
example in the new directory examples/superlu.
|
||||
|
||||
- Extended the KINSOL (SUNDIALS) nonlinear solver interface to support the
|
||||
Jacobian-free Newton-Krylov method. A usage example is shown in Example 10p.
|
||||
|
||||
- Block arrays of parallel matrices can now be merged into a single parallel
|
||||
matrix with the function HypreParMatrixFromBlocks. This could be useful for
|
||||
solving block systems with parallel direct solvers such as STRUMPACK.
|
||||
|
||||
- Added CUDA support for SUNDIALS ODE integrators. See the updated SUNDIALS
|
||||
modification of Example 9/9p.
|
||||
|
||||
- Added wrappers for hypre's flexible GMRES solver and the new parallel ILU
|
||||
preconditioner. The latter requires hypre version 2.19.0 or later.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Extended GSLIB-FindPoints integration to support simplices and interpolation
|
||||
of functions from L2, H(div) and H(curl) spaces.
|
||||
|
||||
- Added support for computing asymptotic error estimates and convergence rates
|
||||
for the whole de Rham sequence based on the new class ConvergenceStudy and new
|
||||
member methods in GridFunction and ParGridFunction. See the rates.cpp file in
|
||||
the tests/convergence directory for sample usage.
|
||||
|
||||
- Extended the GetValue and GetVectorValue methods of GridFunction to support
|
||||
evaluation on boundary elements and, in the continuous field case, arbitrary
|
||||
mesh edges and faces. This requires passing an ElementTransformation argument.
|
||||
|
||||
- Added support for matrix-free interpolation and restriction operators between
|
||||
continuous H1 finite element spaces of different order on the same mesh or
|
||||
with the same order on uniformly refined meshes.
|
||||
|
||||
- The Coefficient classes based on a C-function pointer (FunctionCoefficient,
|
||||
VectorFunctionCoefficient and MatrixFunctionCoefficient) now use the more
|
||||
general std::function class template. This allows the classes to be backward
|
||||
compatible (i.e. they can still work with C-functions) and, in addition,
|
||||
support any "callable", e.g. lambda functions.
|
||||
|
||||
- Non-conforming meshes are now supported with block nonlinear forms. See the
|
||||
updated Example 19/19p.
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
- The graph linear ordering library Gecko, previously an external dependency, is
|
||||
now included directly in MFEM. As a result, Mesh::GetGeckoElementOrdering is
|
||||
always available. The interface has also been improved, see for example the
|
||||
Mesh Explorer miniapp.
|
||||
|
||||
- Added support for finite difference-based gradient and Hessian approximation
|
||||
in the TMOP mesh optimization algorithms. This improves the accuracy of the
|
||||
Hessian for r-adaptivity using discrete fields, and allows use of skewness
|
||||
and orientation based metrics.
|
||||
|
||||
- Improved Gmsh reader (version 2.2), which now supports both high-order and
|
||||
periodic meshes. Segments, triangles, quadrilaterals, and tetrahedra are
|
||||
supported up to order 10. Wedges and hexahedra are supported up to order 9.
|
||||
For sample periodic meshes, see the periodic*.msh files in the data directory.
|
||||
|
||||
- Added support for construction of (serial) non-conforming meshes. Hanging
|
||||
nodes can be marked with Mesh::AddVertexParents when building the mesh with
|
||||
the "init" constructor. The usage is demonstrated in a new meshing miniapp,
|
||||
Polar NC, which generates meshes that are non-conforming from the start.
|
||||
|
||||
- Added support for r-adaptivity with more than one discrete field. This allows
|
||||
the user to specify different discrete functions for controlling the size,
|
||||
aspect-ratio, orientation, and skew of elements in the mesh.
|
||||
|
||||
- Additional TMOP improvements:
|
||||
* Capability for approximate tangential mesh relaxation.
|
||||
* Support and examples for using TMOP on mixed meshes.
|
||||
* Complete integrator action accounting for spatial derivatives of discrete
|
||||
and analytic targets.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added a new miniapp, Navier, that solves the time-dependent Navier-Stokes
|
||||
equations of incompressible fluid dynamics. See the miniapps/navier directory
|
||||
for more details.
|
||||
|
||||
- Added 10 new example codes:
|
||||
* Example 25/25p demonstrates the use of a Perfectly Matched Layer (PML) for
|
||||
electromagnetic wave propagation (indefinite Maxwell).
|
||||
* Example 26/26p shows how to construct matrix-free geometric and p-multigrid
|
||||
preconditioner for the Laplace problem.
|
||||
* Example 27/27p demonstrates the enforcement of Dirichlet, Neumann, Robin,
|
||||
and periodic boundary conditions with either H1 or DG Laplace problems.
|
||||
* Versions of Example 1/1p in examples/amgx demonstrating the use of AmgX,
|
||||
to solve the Laplace problem with AMG preconditioning on GPUs.
|
||||
* A version of Example 11p in examples/petsc demonstrating the use of SLEPc,
|
||||
to solve the Laplace eigenproblem with shift-and-invert transformation.
|
||||
* A version of Example 1 in examples/superlu demonstrating the use of SuperLU
|
||||
to solve the Laplace problem.
|
||||
|
||||
- Added a new Field Interpolation miniapp in miniapps/gslib that demonstrates
|
||||
transfer of grid functions between different meshes using GSLIB-FindPoints.
|
||||
|
||||
- Added 2 miniapps in the new miniapps/adjoint directory demonstrating how to
|
||||
solve adjoint problems in MFEM using the CVODES package in SUNDIALS. Both of
|
||||
these miniapps require the MFEM_USE_SUNDIALS configuration option.
|
||||
* The cvsRoberts_ASAi_dns miniapp solves a backward adjoint problem for a
|
||||
system of ODEs, evaluating both forward and adjoint quadratures in serial.
|
||||
* The adjoint_advection_diffusion miniapp solves a backward adjoint problem
|
||||
for an advection diffusion PDE, evaluating adjoint quadratures in parallel.
|
||||
|
||||
- Added 4 additional meshing miniapps:
|
||||
* The Minimal Surface miniapp solves Plateau's problem: the Dirichlet problem
|
||||
for the minimal surface equation.
|
||||
* The Twist miniapp demonstrates how to stitch together opposite surfaces of a
|
||||
mesh to create a topologically periodic mesh.
|
||||
* The Trimmer miniapp trims away parts of a mesh based on element attributes.
|
||||
* Polar NC shows the construction of polar non-conforming meshes.
|
||||
|
||||
- Several examples and miniapps were updated to include:
|
||||
* Full and element assembly support in Example 9/9p.
|
||||
* Partial assembly with diagonal preconditioning in Examples 4/4p/5/5p/22/22p.
|
||||
* Diagonal preconditioner in Example 6/6p for partial assembly with AMR.
|
||||
* The option to plot a function in Mesh Explorer.
|
||||
* A new test problem showing a mixed bilinear form for H1, H(curl), H(div) and
|
||||
L2, with partial assembly support in Example 24/24p.
|
||||
* Weak Dirichlet boundary conditions (Nitsche) to the NURBS miniapp.
|
||||
|
||||
Data management and Visualization
|
||||
---------------------------------
|
||||
- Added support for ADIOS2 for parallel I/O with ParaView visualization. See
|
||||
Examples 5, 9, 12, 16. The classes adios2stream and ADIOS2DataCollection
|
||||
provide the interface to generate ADIOS2 Binary Pack (BP4) directory datasets.
|
||||
|
||||
- Added VTU output of boundary elements and attributes and parallel VTU (PVTU)
|
||||
output of parallel meshes for visualization using ParaView.
|
||||
|
||||
Improved testing
|
||||
----------------
|
||||
- Added a GitLab pipeline that automates PR testing on supercomputing systems
|
||||
and Linux clusters at Lawrence Livermore National Lab (LLNL). This can be
|
||||
triggered only by LLNL developers, see .gitlab-ci.yml, the .gitlab directory
|
||||
and the updated CONTRIBUTING.md file.
|
||||
|
||||
- Added additional testing for convergence, the parallel mesh I/O, and for the
|
||||
libCEED integration in MFEM in the tests/ directory.
|
||||
|
||||
- Upgraded the Catch unit test framework from version 1.6.1 to version 2.13.0.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Renamed "Backend::DEBUG" to "Backend::DEBUG_DEVICE" to avoid conflicts,
|
||||
as DEBUG is sometimes used as a macro.
|
||||
|
||||
- Added a new IterativeSolverMonitor class that allows to monitor the residual
|
||||
and solution with an IterativeSolver after every iteration.
|
||||
|
||||
- Added power method to iteratively estimate the largest eigenvalue and the
|
||||
corresponding eigenvector of an operator.
|
||||
|
||||
- Added support for face integrals on the boundaries of NURBS meshes.
|
||||
|
||||
- In SLISolver, changed the residual inner product from (Br,r) to (Br,Br) so the
|
||||
solver can work with non-SPD preconditioner B.
|
||||
|
||||
- Added new coefficient and vector coefficient classes for QuadratureFunctions,
|
||||
with new LinearForm integrators which use them.
|
||||
|
||||
- The integration order used in the ComputeLpError and ComputeElementLpError
|
||||
methods of class GridFunction has been increased.
|
||||
|
||||
- Change the IntegrationRule inside VectorDiffusionIntegrator to use the same
|
||||
quadrature as DiffusionIntegrator.
|
||||
|
||||
- The README.html files previously included in several source directories have
|
||||
been removed. Use the corresponding pages at mfem.org instead.
|
||||
|
||||
- Various other simplifications, extensions, and bugfixes in the code.
|
||||
|
||||
|
||||
Version 4.1, released on March 10, 2020
|
||||
=======================================
|
||||
|
||||
Starting with this version, the MFEM open source license is changed to BSD-3.
|
||||
|
||||
Improved GPU capabilities
|
||||
-------------------------
|
||||
- Added initial support for AMD GPUs based on HIP: a C++ runtime API and kernel
|
||||
language that can run on both AMD and NVIDIA hardware.
|
||||
|
||||
- Added support for Umpire, a resource management library that allows the
|
||||
discovery, provision, and management of memory on machines with multiple
|
||||
memory devices like NUMA and GPUs, see https://github.com/LLNL/Umpire.
|
||||
|
||||
- GPU acceleration is now available in 3 additional examples: 3, 9 and 24.
|
||||
|
||||
- Improved RAJA backend and multi-GPU MPI communications.
|
||||
|
||||
- Added a "debug" device designed specifically to aid in debugging GPU code by
|
||||
following the "device" code path (using separate host/device memory spaces and
|
||||
host <-> device transfers) without any GPU hardware.
|
||||
|
||||
- Added support for matrix-free diagonal smoothers on GPUs.
|
||||
|
||||
- The current list of available device backends is: "ceed-cuda", "occa-cuda",
|
||||
"raja-cuda", "cuda", "hip", "debug", "occa-omp", "raja-omp", "omp",
|
||||
"ceed-cpu", "occa-cpu", "raja-cpu", and "cpu".
|
||||
|
||||
- The MFEM memory manager now supports different memory types, associated with
|
||||
the following memory backends:
|
||||
* Default host memory, using standard C++ new and delete,
|
||||
* CUDA pointers, using cudaMalloc and HIP pointers, using hipMalloc,
|
||||
* Managed CUDA/HIP memory (UVM), using cudaMallocManaged/hipMallocManaged,
|
||||
* Umpire-managed memory, including memory pools,
|
||||
* 32- or 64-byte aligned memory, using posix_memalign (WIN32 also supported),
|
||||
* Debug memory with mmap/mprotect protection used by the new "debug" device.
|
||||
|
||||
libCEED support
|
||||
---------------
|
||||
- Added support for libCEED, the portable library for high-order operator
|
||||
evaluation developed by the Center for Efficient Exascale Discretizations in
|
||||
the Exascale Computing Project, https://github.com/CEED/libCEED.
|
||||
|
||||
- This initial integration includes Mass and Diffusion integrators. libCEED GPU
|
||||
backends can be used without specific MFEM configuration, however it is highly
|
||||
recommended to use the "cuda" build option to minimize memory transfers.
|
||||
|
||||
- Both CPU and GPU modes are available as MFEM device backends (ceed-cpu and
|
||||
ceed-cuda), using some of the best performing CPU and GPU backends from
|
||||
libCEED, see the sample runs in examples 1 and 6.
|
||||
|
||||
- NOTE: The current default libCEED GPU backend (ceed-cuda) uses atomics and
|
||||
therefore is non-deterministic.
|
||||
|
||||
Partial assembly and matrix-free discretizations
|
||||
------------------------------------------------
|
||||
- The support for matrix-free methods on both CPU and GPU devices based on a
|
||||
partially assembled operator decomposition was extended to include:
|
||||
* DG integrators, (for now only in the Gauss-Lobatto basis), see Example 9,
|
||||
* H(curl) bilinear forms, see Example 3,
|
||||
* vector mass and vector diffusion bilinear integrators,
|
||||
* convection integrator with improved performance,
|
||||
* gradient and vector divergence integrators for Stokes problems,
|
||||
* initial partial assembly mode for NonlinearForms.
|
||||
|
||||
- Diagonals of partially assembled operators can now be computed efficiently.
|
||||
See the new methods AssembleDiagonal in BilinearForm, AssembleDiagonalPA in
|
||||
BilinearFormIntegrator and the implementations in fem/bilininteg_*.cpp.
|
||||
|
||||
- In many examples, the partial assembly algorithms provide significantly
|
||||
improved performance, particularly in high-order 3D runs on GPUs.
|
||||
|
||||
Meshing improvements
|
||||
--------------------
|
||||
- The algorithms for mesh element numbering were changed to have significantly
|
||||
better caching and parallel partitioning properties. Both initial (see e.g.
|
||||
Mesh::GetHilbertElementOrdering) and ordering after uniform refinement were
|
||||
improved. NOTE: new ordering can have a round-off effect on solver results.
|
||||
|
||||
- Added support for non-conforming AMR on both prisms and tetrahedra, including
|
||||
coarsening and parallel load balancing. Anisotropic prism refinement is only
|
||||
available in the serial version at the moment.
|
||||
|
||||
- The TMOP mesh optimization algorithms were extended to support r-adaptivity.
|
||||
Target matrices can now be constructed either via a given analytical function
|
||||
(e.g. spatial dependence of size, aspect ratio, etc., for each element) or via
|
||||
a (Par)GridFunction specified on the original mesh.
|
||||
|
||||
- The TMOP algorithms were also improved to support non-conforming AMR meshes.
|
||||
|
||||
- Added support for creating refined versions of periodic meshes, making use of
|
||||
the new L2ElementRestriction class. This class also allows for computing
|
||||
geometric factors on periodic meshes using partial assembly.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Added support for GSLIB-FindPoints, a general high-order interpolation utility
|
||||
that can robustly evaluate a GridFunction in an arbitrary collection of points
|
||||
in physical space. See INSTALL for details on building MFEM with GSLIB, and
|
||||
miniapps/gslib for examples of how to use this feature.
|
||||
|
||||
- Added support for complex-valued finite element operators and fields using a
|
||||
2x2 block structured linear system to mimic complex arithmetic. New classes
|
||||
include: ComplexGridFunction, SesquilinearForm, ComplexLinearForm, and their
|
||||
parallel counterparts.
|
||||
|
||||
- Added second order derivatives of NURBS shape functions.
|
||||
|
||||
- Added support for serendipity elements of arbitrary order on affinely-mapped
|
||||
square elements. Basis functions for these elements can be visualized using
|
||||
an option in the display-basis miniapp.
|
||||
|
||||
- Two integrators related to Stokes problems, (Q grad u, v) and (Q div v, u),
|
||||
where u and the components of v are in H1, were added/modified to support full
|
||||
and partial assembly modes. See the new GradientIntegrator and the updated
|
||||
VectorDivergenceIntegrator classes in fem/bilininteg.hpp, as well as the PA
|
||||
kernels in fem/bilininteg_gradient.cpp and fem/bilininteg_divergence.cpp.
|
||||
|
||||
- Added a nonlinear vector valued convection integrator (Q u \cdot grad u, v)
|
||||
where u_i and v_i are in H1. This form occurs e.g. in the Navier-Stokes
|
||||
equations. The integrator supports the partial assembly mode for its
|
||||
action. In full assembly mode we also provide the GetGradient method that
|
||||
computes the linearized version of the integrator.
|
||||
|
||||
- Added a new method, MixedBilinearForm::FormRectangularLinearSystem, that can
|
||||
be used to impose boundary conditions on the non-square off-diagonal blocks of
|
||||
a block operator (similar to FormLinearSystem in the square case).
|
||||
|
||||
Linear and nonlinear solvers
|
||||
----------------------------
|
||||
- Added support for Ginkgo, a high-performance linear algebra library for GPU
|
||||
and manycore nodes, with a focus on sparse solution of linear systems. For
|
||||
more details see linalg/ginkgo.hpp and the example code in examples/gingko.
|
||||
|
||||
- Added support for HiOp, a lightweight HPC solver for nonlinear optimization
|
||||
problems, see class HiOpNLPOptimizer and the example codes in examples/hiop.
|
||||
|
||||
- Added a general interface for specifying and solving nonlinear constrained
|
||||
optimization problems through the new classes OptimizationProblem and
|
||||
OptimizationSolver, see linalg/solver.hpp.
|
||||
|
||||
- Added a block ILU(0) preconditioner for DG-type discretizations. Example 9
|
||||
(DG advection) now takes advantage of this for implicit time integration.
|
||||
|
||||
- New time integrators: Adams-Bashforth, Adams-Moulton and several integrators
|
||||
for 2nd order ODEs, see the new Example 23.
|
||||
|
||||
- Added a LinearSolve(A,X) convenience method to solve dense linear systems. In
|
||||
the trivial cases, i.e., square matrices of size 1 or 2, the system is solved
|
||||
directly, otherwise, LU factorization is employed.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added a collection of 7 playful miniapps in miniapps/toys that illustrate the
|
||||
meshing and visualization features of the library in more relaxed settings.
|
||||
The toys include simulations of cellular automata, Rubik's cube, Mandelbrot
|
||||
set, a tool to convert any image to mfem mesh, and more.
|
||||
|
||||
- Added 8 new example codes:
|
||||
* Example 22/22p demonstrates the use of the new complex-valued finite element
|
||||
operators by defining and solving a family of time-harmonic PDEs related to
|
||||
damped harmonic oscillators.
|
||||
* Example 23 solves a simple 2D/3D wave equation with the new second order
|
||||
time integrators.
|
||||
* Example 24/24p demonstrates usage of mixed finite element spaces in bilinear
|
||||
forms. Partial assembly is supported in this example.
|
||||
* A version of Example 1 in examples/ginkgo demonstrating the use of the
|
||||
Gingko interface to solve a linear system.
|
||||
* A version of Example 9/9p in examples/hiop demonstrating the nonlinear
|
||||
constrained optimization interface and use of the SLBQP and HiOp solvers.
|
||||
|
||||
- Added two new miniapps: Find Points and Field Diff in miniapps/gslib that show
|
||||
how GSLIB-FindPoints can be used to interpolate a (Par) GridFunction in an
|
||||
arbitrary number of physical space points in 2D and 3D. The GridFunction must
|
||||
be in H1 and in the same space as the mesh that is used to find the points.
|
||||
|
||||
- Added a simple miniapp, Get Values, that extracts field values at a set of
|
||||
points, from previously saved data via DataCollection classes.
|
||||
|
||||
- Several examples and miniapps were updated:
|
||||
* Added device support in Example 3/3p and Example 9/9p.
|
||||
* Example 1/1p and Example 3/3p now use diagonal preconditioning in partial
|
||||
assembly mode.
|
||||
* Example 9/9p now supports implicit time integration, using the new block
|
||||
ILU(0) solvers as preconditioners for the linear system.
|
||||
* The mesh-optimizer and pmesh-optimizer miniapps now include the new
|
||||
r-adaptivity capabilities of TMOP. They were also updated to support mesh
|
||||
optimization on non-conforming AMR meshes.
|
||||
* New options to reorder and partition the mesh and boundary attribute
|
||||
visualization (key 'b') are now available in the mesh-explorer miniapp.
|
||||
|
||||
- Collected object files from the miniapps/common directory into a new library,
|
||||
libmfem-common for the convenience of application developers. The new library
|
||||
is now used in several miniapps in the electromagnetic and tools directories.
|
||||
|
||||
Improved testing
|
||||
----------------
|
||||
- Added a large number of unit tests in the tests/unit directory, including
|
||||
several parallel unit tests.
|
||||
|
||||
- Added a new directory, tests/scripts, with several shell scripts that perform
|
||||
simple checks on the code including: code styling, documentation formatting,
|
||||
proper use of .gitignore, and preventing the accidental commit of large files.
|
||||
|
||||
- It is recommended that developers run the above tests scripts (via the runtest
|
||||
script) before pushing to GitHub. See the README file in tests/scripts.
|
||||
|
||||
- The Travis CI settings have been updated to include an initial Checks stage
|
||||
which currently runs the code-style, documentation and gitignore test scripts,
|
||||
as well as a final stage for optional checks/tests which currently runs the
|
||||
branch-history script.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Added support for output in the ParaView XML format. Both low-order and
|
||||
high-order Lagrange elements are supported. Output can be in ASCII or binary
|
||||
format. The binary output can be compressed if MFEM is compiled with zlib
|
||||
support (MFEM_USE_ZLIB). See the new ParaViewDataCollection class and the
|
||||
updated Examples 5/5p and 9/9p.
|
||||
|
||||
- Upgraded the SUNDIALS interface to utilize SUNDIALS 5.0. This necessitated a
|
||||
complete rework of the interface and requires changes at the application
|
||||
level. Example usage of the new interface can be found in examples/sundials.
|
||||
|
||||
- Switched from gzstream to zstr for the implementation of zlib-compressed C++
|
||||
output stream. The build system definition now uses MFEM_USE_ZLIB instead of
|
||||
MFEM_USE_GZSTREAM, but the code interface (e.g. ofgzstream) remains the same.
|
||||
|
||||
- Various other simplifications, extensions, and bugfixes in the code.
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- In the enum classes MemoryType and MemoryClass, "CUDA" was renamed to "DEVICE"
|
||||
which now denotes either "CUDA" or "HIP" depending on the build configuration.
|
||||
In the same enum classes, "CUDA_UVM" was renamed to "MANAGED".
|
||||
|
||||
|
||||
Version 4.0, released on May 24, 2019
|
||||
=====================================
|
||||
|
||||
Unlike previous MFEM releases, this version requires a C++11 compiler.
|
||||
|
||||
GPU support
|
||||
-----------
|
||||
- Added initial support for hardware devices, such as GPUs, and programming
|
||||
models, such as CUDA, OCCA, RAJA and OpenMP.
|
||||
|
||||
- The GPU/device support is based on MFEM's new backends and kernels working
|
||||
seamlessly with a new lightweight device/host memory manager. The kernels can
|
||||
be implemented either in OCCA, or as a simple wrapper around for-loops, which
|
||||
can then be dispatched to RAJA and native backends. See the files forall.hpp
|
||||
and mem_manager.hpp in the general/ directory for more details.
|
||||
|
||||
- Several of the MFEM example codes (ex1, ex1p, ex6, and ex6p) can now take
|
||||
advantage of GPU acceleration with the backend selectable at runtime. Many of
|
||||
the linear algebra and finite element operations (e.g. partially assembled
|
||||
bilinear forms) have been extended to take advantage of kernel acceleration by
|
||||
simply replacing loops with the MFEM_FORALL() macro.
|
||||
|
||||
- In addition to native CUDA kernels, the library currently supports OCCA, RAJA
|
||||
and OpenMP kernels, which could be mixed and matched in different parts of the
|
||||
same application. We plan on adding support for more programming models and
|
||||
devices in the future, without the need for significant modifications in user
|
||||
code. The list of current backends is: "occa-cuda", "raja-cuda", "cuda",
|
||||
"occa-omp", "raja-omp", "omp", "occa-cpu", "raja-cpu", and "cpu".
|
||||
|
||||
- GPU-related limitations:
|
||||
* Hypre preconditioners are not yet available in GPU mode, and in particular
|
||||
hypre must be built in CPU mode.
|
||||
* Only constant coefficients are currently supported on GPUs.
|
||||
* Optimized element assembly, and matrix-free bilinear forms are not
|
||||
implemented yet. Element batching is currently ignored.
|
||||
* In device mode, full assembly is performed on the host (but the matvec
|
||||
action is performed on the device).
|
||||
* Partial assembly kernels are not implemented yet for simplices.
|
||||
|
||||
Discretization improvements
|
||||
---------------------------
|
||||
- Partial assembled finite element operators are now available in the core
|
||||
library, based on the new classes PABilinearFormExtension, ElementRestriction,
|
||||
DofToQuad and GeometricFactors (associated with the classes BilinearForm,
|
||||
FiniteElementSpace, FiniteElement and Mesh, respectively). The kernels for
|
||||
partial assembled Setup/Assembly and Action/Mult are implemented in the
|
||||
BilinearFormIntegrator methods AssemblePA and AddMultPA.
|
||||
|
||||
- Added support for a general "low-order refined"-to-"high-order" transfer of
|
||||
GridFunction data from a "low-order refined" (LOR) space defined on a refined
|
||||
mesh to a "high-order" (HO) finite element space defined on a coarse mesh. See
|
||||
the new classes InterpolationGridTransfer and L2ProjectionGridTransfer and the
|
||||
new LOR Transfer miniapp: miniapps/tools/lor-transfer.cpp.
|
||||
|
||||
- Added element flux, and flux energy computation in class ElasticityIntegrator,
|
||||
allowing for the use of Zienkiewicz-Zhu type error estimators with the
|
||||
integrator. For an illustration of this addition, see the new Example 21.
|
||||
|
||||
- Added support for derefinement of vector (RT + ND) spaces.
|
||||
|
||||
- Added a variety of coefficients which are sums or products of existing
|
||||
coefficients as well as grid function coefficients which return the
|
||||
divergence, gradient, or curl of their GridFunctions.
|
||||
|
||||
Support for wedge elements and meshes with mixed element types
|
||||
--------------------------------------------------------------
|
||||
- Added support for wedge-shaped mesh elements of arbitrary order (with Geometry
|
||||
type PRISM) which have two triangular faces and three quadrilateral faces.
|
||||
Several examples of such meshes can be found in the data/ directory.
|
||||
|
||||
- Added support for mixed meshes containing triangles and quadrilaterals in 2D
|
||||
or tetrahedra, wedges, and hexahedra in 3D. This includes support for uniform
|
||||
refinement of such meshes. Several examples of such meshes can be found in the
|
||||
data/ directory.
|
||||
|
||||
- Added H1 and L2 finite elements of arbitrary order for Wedge elements.
|
||||
|
||||
- Added support for reading and writing linear and quadratic meshes containing
|
||||
wedge elements in VTK mesh format. Several examples of such meshes can be
|
||||
found in the data/ directory.
|
||||
|
||||
Other meshing improvements
|
||||
--------------------------
|
||||
- Improved the uniform refinement of tetrahedral meshes (also part of the
|
||||
uniform refinement of mixed 3D meshes). The previous refinement algorithm is
|
||||
still available as an option in Mesh::UniformRefinement. Both can be used in
|
||||
the updated Mesh Explorer miniapp.
|
||||
|
||||
- The local tetrahedral mesh refinement algorithm in serial and in parallel now
|
||||
follows precisely the paper:
|
||||
|
||||
D. Arnold, A. Mukherjee, and L. Pouly, "Locally Adapted Tetrahedral Meshes
|
||||
Using Bisection", SIAM J. Sci. Comput. 22 (2000), 431-448.
|
||||
|
||||
This guarantees that the shape regularity of the elements will be preserved
|
||||
under refinement.
|
||||
|
||||
- The TMOP mesh optimization algorithms were extended to support user-defined
|
||||
space-dependent limiting terms. Improved the TMOP objective functions by more
|
||||
accurate normalization of the different terms.
|
||||
|
||||
- Added support for parallel communication groups on non-conforming meshes.
|
||||
|
||||
- Improved parallel partitioning of non-conforming meshes. If the coarse mesh
|
||||
elements are ordered as a sequence of face-neighbors, the parallel partitions
|
||||
are now guaranteed to be continuous. To that end, inline quadrilateral and
|
||||
hexahedral meshes are now by default ordered along a space-filling curve.
|
||||
|
||||
- A boundary in a NURBS mesh can now be connected with another boundary. Such a
|
||||
periodic NURBS mesh is a simple way to impose periodic boundary conditions.
|
||||
|
||||
- Added support for reading linear and quadratic 2D quadrilateral and triangular
|
||||
Cubit meshes.
|
||||
|
||||
New and updated examples and miniapps
|
||||
-------------------------------------
|
||||
- Added a new meshing miniapp, Toroid, which can produce a variety of torus
|
||||
shaped meshes by twisting a stack of wedges or hexahedra.
|
||||
|
||||
- Added a new meshing miniapp, Extruder, that demonstrates the capability to
|
||||
produce 3D meshes by extruding 2D meshes.
|
||||
|
||||
- Added a simple miniapp, LOR Transfer, for visualizing the actions of the
|
||||
transfer operators between a high-order and a low-order refined spaces.
|
||||
|
||||
- Added a new example, Example 20/20p, that solves a system of 1D ODEs derived
|
||||
from a Hamiltonian. The example demonstrates the use of the variable order,
|
||||
symplectic integration algorithm implemented in class SIAVSolver.
|
||||
|
||||
- Added a new example, Example 21/21p, that illustrates the use of AMR to solve
|
||||
a linear elasticity problem. This is an extension of Example 2/2p.
|
||||
|
||||
New and improved solvers and preconditioners
|
||||
--------------------------------------------
|
||||
- Added support for parallel ILU preconditioning via hypre's Euclid solver.
|
||||
|
||||
- Added support for STRUMPACK v3 with a small API change in the class
|
||||
STRUMPACKSolver, see "API changes" below.
|
||||
|
||||
Miscellaneous
|
||||
-------------
|
||||
- Added unit tests based on the Catch++ library in the test/ directory.
|
||||
|
||||
- Renamed the option MFEM_USE_OPENMP to MFEM_USE_LEGACY_OPENMP. This legacy
|
||||
option is deprecated and planned for removal in a future release. The original
|
||||
option name, MFEM_USE_OPENMP, is now used to enable the new OpenMP backends in
|
||||
the new kernels.
|
||||
|
||||
- In SparseMatrix added the option to perform MultTranspose() by matvec with
|
||||
computed and stored transpose matrix. This is required for deterministic
|
||||
results when using devices such as CUDA and OpenMP.
|
||||
|
||||
- Altered the way FGMRES counts its iterations so that it matches GMRES.
|
||||
|
||||
- Various other simplifications, extensions, and bugfixes in the code.
|
||||
|
||||
- Construct abstract parallel rectangular truedof-to-truedof operators via
|
||||
Operator::FormDiscreteOperator().
|
||||
|
||||
API changes
|
||||
-----------
|
||||
- In multiple places, use Geometry::Type instead of int, where appropriate.
|
||||
- In multiple places, use Element::Type instead of int, where appropriate.
|
||||
- The Mesh methods GetElementBaseGeometry and GetBdrElementBaseGeometry no
|
||||
longer have a default value for their parameter, they only work with an
|
||||
explicitly given index.
|
||||
- In class Mesh, added methods useful for queries regarding the types of
|
||||
elements present in the mesh: HasGeometry, GetNumGeometries, GetGeometries,
|
||||
and class Mesh::GeometryList.
|
||||
- The struct CoarseFineTransformations (returned by the Mesh method
|
||||
GetRefinementTransforms) now stores the embedding matrices separately for each
|
||||
Geometry::Type.
|
||||
- In class ParMesh, replaced the method GroupNFaces with two new methods:
|
||||
GroupNTriangles and GroupNQuadrilaterals. Also, replaced GroupFace with two
|
||||
methods: GroupTriangle and GroupQuadrilateral.
|
||||
- In class ParMesh, made the two RefineGroups methods protected.
|
||||
- Removed the virtual method Element::GetRefinementFlag, it is only used by the
|
||||
derived class Tetrahedron.
|
||||
- Added new methods: Array::CopyTo, Tetrahedron::Init.
|
||||
- In class STRUMPACKSolver, the method SetMC64Job() was replaced by the new
|
||||
methods: DisableMatching(), EnableMatching(), and EnableParallelMatching().
|
||||
|
||||
|
||||
Version 3.4, released on May 29, 2018
|
||||
=====================================
|
||||
@@ -978,7 +68,7 @@ Discretization improvements
|
||||
|
||||
- New specialized time integrators: symplectic integrators of orders 1-4 for
|
||||
systems of first order ODEs derived from a Hamiltonian and generalized-alpha
|
||||
ODE solver for the filtered Navier-Stokes equations with stabilization. See
|
||||
ODE solver for the filtered Navier–Stokes equations with stabilization. See
|
||||
classes SIASolver and GeneralizedAlphaSolver in linalg/ode.hpp.
|
||||
|
||||
- Inherit finite element classes from the new base class TensorBasisElement,
|
||||
@@ -992,9 +82,6 @@ New and updated examples and miniapps
|
||||
incompressible hyperelastic equations. The example demonstrates the use of
|
||||
block nonlinear forms as well as custom block preconditioners.
|
||||
|
||||
- Added a new serial example (ex23) to demonstrate the use of second order
|
||||
time integration to solve the wave equation.
|
||||
|
||||
- Added a new electromagnetics miniapp, Maxwell, for simulating time-domain
|
||||
electromagnetics phenomena as a coupled first order system of equations.
|
||||
|
||||
@@ -1128,7 +215,7 @@ New and updated examples and miniapps
|
||||
|
||||
- Added a new meshing miniapp, Shaper, that can be used to resolve complicated
|
||||
material interfaces by mesh refinement, e.g. as a tool for initial mesh
|
||||
generation from prescribed "material()" function. Both conforming and
|
||||
generation from prescribed "material()" function. Both conforming and
|
||||
non-conforming (isotropic and anisotropic) refinements are supported.
|
||||
|
||||
- Added a new meshing miniapp, Mesh Optimizer, that demonstrates the use of TMOP
|
||||
@@ -1140,7 +227,7 @@ Discretization improvements
|
||||
---------------------------
|
||||
- Added a FindPoints method of the Mesh and ParMesh classes that returns the
|
||||
elements that contain a given set of points, together with the coordinates of
|
||||
the points in the reference space of the corresponding element. In parallel,
|
||||
the points in the reference space of the corresponding element. In parallel,
|
||||
if a point is shared by multiple processors, only one of them will mark that
|
||||
point as found. Note that the current implementation of this method is not
|
||||
optimal and/or 100% reliable. See the mesh-explorer miniapp for an example.
|
||||
@@ -1467,8 +554,8 @@ New and improved linear solvers
|
||||
which is a sparse direct solver for distributed memory architectures. As such
|
||||
it can only be enabled along with MFEM_USE_MPI. When MFEM is configured with
|
||||
MFEM_USE_SUPERLU, one also needs to alter the version of METIS, since SuperLU
|
||||
requires ParMETIS (which comes packaged with a serial version of METIS). See
|
||||
http://crd-legacy.lbl.gov/~xiaoye/SuperLU for SuperLU_DIST details.
|
||||
requires ParMETIS (which comes packaged with a serial version of METIS). See
|
||||
http://http://crd-legacy.lbl.gov/~xiaoye/SuperLU for SuperLU_DIST details.
|
||||
|
||||
- Added a wrapper for the KLU solver in SuiteSparse see
|
||||
http://faculty.cse.tamu.edu/davis/suitesparse.html for details of KLU.
|
||||
@@ -1878,7 +965,7 @@ New and updated examples
|
||||
(ADS) in hypre.
|
||||
|
||||
- Modified Example 1 to use isoparametric discretization (use the FE space from
|
||||
the mesh) including NURBS meshes and spaces. Updated Example 2 to support
|
||||
the mesh) including NURBS meshes and spaces. Updated Example 2 to support
|
||||
arbitrary order spaces. Updated all examples to work with NURBS meshes and
|
||||
spaces, as well as to not use projection onto discontinuous polynomial spaces
|
||||
for visualization (this is now handled directly in GLVis when necessary).
|
||||
@@ -1970,7 +1057,7 @@ Version 1.1, released on Sep 13, 2010
|
||||
New MFEM format for general meshes
|
||||
----------------------------------
|
||||
- New MFEM mesh v1.0 format with uniform structure for any dimension and support
|
||||
for curved meshes including in 3D. Class Mesh will recognize and read the new
|
||||
for curved meshes including in 3D. Class Mesh will recognize and read the new
|
||||
format (in addition to all previously used formats) and Mesh::Print uses the
|
||||
new format by default. The old print function was renamed to Mesh::PrintXG.
|
||||
|
||||
|
||||
+45
-222
@@ -1,27 +1,18 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# The variable CMAKE_CXX_STANDARD and related were introduced in CMake v3.1
|
||||
cmake_minimum_required(VERSION 3.1)
|
||||
cmake_minimum_required(VERSION 2.8.11)
|
||||
set(USER_CONFIG "${CMAKE_CURRENT_SOURCE_DIR}/config/user.cmake" CACHE PATH
|
||||
"Path to optional user configuration file.")
|
||||
|
||||
# Require C++11 and disable compiler-specific extensions
|
||||
set(CMAKE_CXX_STANDARD 11)
|
||||
if (MFEM_USE_GINKGO)
|
||||
set(CMAKE_CXX_STANDARD 14)
|
||||
endif()
|
||||
set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||
set(CMAKE_CXX_EXTENSIONS OFF)
|
||||
|
||||
# Load user settings before the defaults - this way the defaults will not
|
||||
# overwrite the user set options. If the user has not set all options, we still
|
||||
# have the defaults.
|
||||
@@ -54,7 +45,7 @@ project(mfem NONE)
|
||||
# Current version of MFEM, see also `makefile`.
|
||||
# mfem_VERSION = (string)
|
||||
# MFEM_VERSION = (int) [automatically derived from mfem_VERSION]
|
||||
set(${PROJECT_NAME}_VERSION 4.2.1)
|
||||
set(${PROJECT_NAME}_VERSION 3.4.1)
|
||||
|
||||
# Prohibit in-source build
|
||||
if (${PROJECT_SOURCE_DIR} STREQUAL ${PROJECT_BINARY_DIR})
|
||||
@@ -90,49 +81,6 @@ include("${CMAKE_CURRENT_SOURCE_DIR}/config/XSDKDefaults.cmake")
|
||||
|
||||
# Enable languages.
|
||||
enable_language(CXX)
|
||||
if (MFEM_USE_CUDA)
|
||||
if (MFEM_USE_HIP)
|
||||
message(FATAL_ERROR " *** MFEM_USE_HIP cannot be combined with MFEM_USE_CUDA.")
|
||||
endif()
|
||||
# MFEM_USE_CUDA requires CMake 3.8 or newer (for direct CUDA support)
|
||||
cmake_minimum_required(VERSION 3.8 FATAL_ERROR)
|
||||
# Use ${CMAKE_CXX_COMPILER} as the cuda host compiler.
|
||||
if (NOT CMAKE_CUDA_HOST_COMPILER)
|
||||
set(CMAKE_CUDA_HOST_COMPILER ${CMAKE_CXX_COMPILER})
|
||||
endif()
|
||||
enable_language(CUDA)
|
||||
set(CMAKE_CUDA_STANDARD 11)
|
||||
if (MFEM_USE_GINKGO)
|
||||
set(CMAKE_CUDA_STANDARD 14)
|
||||
endif()
|
||||
set(CMAKE_CUDA_STANDARD_REQUIRED ON)
|
||||
set(CMAKE_CUDA_EXTENSIONS OFF)
|
||||
set(CUDA_FLAGS "--expt-extended-lambda")
|
||||
if (CMAKE_VERSION VERSION_LESS 3.18.0)
|
||||
set(CUDA_FLAGS "-arch=${CUDA_ARCH} ${CUDA_FLAGS}")
|
||||
elseif (NOT CMAKE_CUDA_ARCHITECTURES)
|
||||
string(REGEX REPLACE "^sm_" "" ARCH_NUMBER "${CUDA_ARCH}")
|
||||
if ("${CUDA_ARCH}" STREQUAL "sm_${ARCH_NUMBER}")
|
||||
set(CMAKE_CUDA_ARCHITECTURES "${ARCH_NUMBER}")
|
||||
else()
|
||||
message(FATAL_ERROR "Unknown CUDA_ARCH: ${CUDA_ARCH}")
|
||||
endif()
|
||||
else()
|
||||
set(CUDA_ARCH "CMAKE_CUDA_ARCHITECTURES: ${CMAKE_CUDA_ARCHITECTURES}")
|
||||
endif()
|
||||
message(STATUS "Using CUDA architecture: ${CUDA_ARCH}")
|
||||
if (CMAKE_VERSION VERSION_LESS 3.12.0)
|
||||
# CMake versions 3.8 and 3.9 require this to work; 3.10 and 3.11 are not
|
||||
# tested and may not actually need this (but should be ok to keep).
|
||||
set(CUDA_FLAGS "-ccbin=${CMAKE_CXX_COMPILER} ${CUDA_FLAGS}")
|
||||
set(CMAKE_CUDA_HOST_LINK_LAUNCHER ${CMAKE_CXX_COMPILER})
|
||||
endif()
|
||||
set(CMAKE_CUDA_FLAGS "${CUDA_FLAGS}" CACHE STRING
|
||||
"CUDA flags set for MFEM" FORCE)
|
||||
set(CUSPARSE_FOUND TRUE)
|
||||
set(CUSPARSE_LIBRARIES "cusparse")
|
||||
endif()
|
||||
|
||||
if (XSDK_ENABLE_C)
|
||||
enable_language(C)
|
||||
endif()
|
||||
@@ -179,12 +127,6 @@ endif()
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(MPI REQUIRED)
|
||||
set(MPI_CXX_INCLUDE_DIRS ${MPI_CXX_INCLUDE_PATH})
|
||||
if (MFEM_MPIEXEC)
|
||||
set(MPIEXEC ${MFEM_MPIEXEC})
|
||||
endif()
|
||||
if (MFEM_MPIEXEC_NP)
|
||||
set(MPIEXEC_NUMPROC_FLAG ${MFEM_MPIEXEC_NP})
|
||||
endif()
|
||||
# Parallel MFEM depends on hypre
|
||||
find_package(HYPRE REQUIRED)
|
||||
set(MFEM_HYPRE_VERSION ${HYPRE_VERSION})
|
||||
@@ -195,13 +137,9 @@ if (MFEM_USE_MPI)
|
||||
message(FATAL_ERROR "PETSc version >= 3.8.0 is required")
|
||||
endif()
|
||||
set(PETSC_INCLUDE_DIRS ${PETSC_INCLUDES})
|
||||
if (MFEM_USE_SLEPC)
|
||||
find_package(SLEPc REQUIRED config)
|
||||
message(STATUS "Found SLEPc version ${SLEPC_VERSION}")
|
||||
endif()
|
||||
endif()
|
||||
else()
|
||||
set(PKGS_NEED_MPI SUPERLU MUMPS PETSC SLEPC STRUMPACK PUMI)
|
||||
set(PKGS_NEED_MPI SUPERLU PETSC STRUMPACK PUMI)
|
||||
foreach(PKG IN LISTS PKGS_NEED_MPI)
|
||||
if (MFEM_USE_${PKG})
|
||||
message(STATUS "Disabling package ${PKG} - requires MPI")
|
||||
@@ -214,17 +152,8 @@ if (MFEM_USE_METIS)
|
||||
find_package(METIS REQUIRED)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_GINKGO)
|
||||
find_package(Ginkgo REQUIRED)
|
||||
if (Ginkgo_FOUND)
|
||||
get_target_property(Ginkgo_INCLUDE_DIRS
|
||||
Ginkgo::ginkgo INTERFACE_INCLUDE_DIRECTORIES)
|
||||
set(Ginkgo_LIBRARIES Ginkgo::ginkgo)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# zlib
|
||||
if (MFEM_USE_ZLIB)
|
||||
# GZSTREAM -> zlib
|
||||
if (MFEM_USE_GZSTREAM)
|
||||
find_package(ZLIB REQUIRED)
|
||||
endif()
|
||||
|
||||
@@ -241,11 +170,12 @@ if (MFEM_USE_LAPACK)
|
||||
endif()
|
||||
|
||||
# OpenMP
|
||||
if (MFEM_USE_OPENMP OR MFEM_USE_LEGACY_OPENMP)
|
||||
if (NOT MFEM_THREAD_SAFE AND MFEM_USE_LEGACY_OPENMP)
|
||||
message(FATAL_ERROR " *** MFEM_USE_LEGACY_OPENMP requires MFEM_THREAD_SAFE=ON.")
|
||||
if (MFEM_USE_OPENMP)
|
||||
if (MFEM_THREAD_SAFE)
|
||||
find_package(OpenMP REQUIRED)
|
||||
else()
|
||||
message(FATAL_ERROR " *** MFEM_USE_OPENMP requires MFEM_THREAD_SAFE=ON.")
|
||||
endif()
|
||||
find_package(OpenMP REQUIRED)
|
||||
endif()
|
||||
|
||||
# SuiteSparse (before SUNDIALS which may depend on KLU)
|
||||
@@ -256,14 +186,12 @@ endif()
|
||||
|
||||
# SUNDIALS
|
||||
if (MFEM_USE_SUNDIALS)
|
||||
set(SUNDIALS_COMPONENTS CVODES ARKODE KINSOL NVector_Serial)
|
||||
if (MFEM_USE_MPI)
|
||||
list(APPEND SUNDIALS_COMPONENTS NVector_Parallel NVector_MPIPlusX)
|
||||
if (NOT MFEM_USE_MPI)
|
||||
find_package(SUNDIALS REQUIRED NVector_Serial CVODE ARKODE KINSOL)
|
||||
else()
|
||||
find_package(SUNDIALS REQUIRED
|
||||
NVector_Serial NVector_Parallel NVector_ParHyp CVODE ARKODE KINSOL)
|
||||
endif()
|
||||
if (MFEM_USE_CUDA)
|
||||
list(APPEND SUNDIALS_COMPONENTS NVector_Cuda)
|
||||
endif()
|
||||
find_package(SUNDIALS REQUIRED ${SUNDIALS_COMPONENTS})
|
||||
endif()
|
||||
|
||||
# Mesquite
|
||||
@@ -280,15 +208,6 @@ if (MFEM_USE_SUPERLU)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# MUMPS can only be enabled in parallel
|
||||
if (MFEM_USE_MUMPS)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(MUMPS REQUIRED mumps_common pord)
|
||||
else()
|
||||
message(FATAL_ERROR " *** MUMPS requires that MPI be enabled.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# STRUMPACK can only be enabled in parallel
|
||||
if (MFEM_USE_STRUMPACK)
|
||||
if (MFEM_USE_MPI)
|
||||
@@ -298,16 +217,16 @@ if (MFEM_USE_STRUMPACK)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Gecko
|
||||
if (MFEM_USE_GECKO)
|
||||
find_package(Gecko REQUIRED)
|
||||
endif()
|
||||
|
||||
# GnuTLS
|
||||
if (MFEM_USE_GNUTLS)
|
||||
find_package(_GnuTLS REQUIRED)
|
||||
endif()
|
||||
|
||||
# GSLIB
|
||||
if (MFEM_USE_GSLIB)
|
||||
find_package(GSLIB REQUIRED)
|
||||
endif()
|
||||
|
||||
# NetCDF
|
||||
if (MFEM_USE_NETCDF)
|
||||
find_package(NetCDF REQUIRED)
|
||||
@@ -318,21 +237,13 @@ if (MFEM_USE_MPFR)
|
||||
find_package(MPFR REQUIRED)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_CEED)
|
||||
find_package(libCEED REQUIRED)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_AMGX)
|
||||
find_package(AMGX REQUIRED)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_CONDUIT)
|
||||
find_package(Conduit REQUIRED conduit relay blueprint )
|
||||
endif()
|
||||
|
||||
# Axom/Sidre
|
||||
if (MFEM_USE_SIDRE)
|
||||
find_package(Axom REQUIRED Axom)
|
||||
find_package(Axom REQUIRED Sidre SLIC axom_utils)
|
||||
endif()
|
||||
|
||||
# PUMI
|
||||
@@ -351,55 +262,6 @@ if (MFEM_USE_PUMI)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# HiOp optimizer
|
||||
if (MFEM_USE_HIOP)
|
||||
find_package(HIOP REQUIRED)
|
||||
# find_package updates HIOP_FOUND, HIOP_INCLUDE_DIRS, HIOP_LIBRARIES
|
||||
endif()
|
||||
|
||||
# OCCA
|
||||
if (MFEM_USE_OCCA)
|
||||
find_package(OCCA REQUIRED)
|
||||
endif()
|
||||
|
||||
# RAJA
|
||||
if (MFEM_USE_RAJA)
|
||||
find_package(RAJA REQUIRED)
|
||||
endif()
|
||||
|
||||
# UMPIRE
|
||||
if (MFEM_USE_UMPIRE)
|
||||
find_package(UMPIRE REQUIRED)
|
||||
endif()
|
||||
|
||||
# Caliper
|
||||
if (MFEM_USE_CALIPER)
|
||||
find_package(Caliper REQUIRED)
|
||||
endif()
|
||||
|
||||
# AMD HIP
|
||||
if (MFEM_USE_HIP)
|
||||
find_package(HIP REQUIRED)
|
||||
if (HIP_ARCH)
|
||||
message(STATUS "Using HIP architecture: ${HIP_ARCH}")
|
||||
list(APPEND HIP_HIPCC_FLAGS "--amdgpu-target=${HIP_ARCH}")
|
||||
if (MFEM_USE_GINKGO)
|
||||
list(APPEND HIP_HIPCC_FLAGS "-std=c++14")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# ADIOS2 for parallel I/O
|
||||
if (MFEM_USE_ADIOS2)
|
||||
find_package(ADIOS2 REQUIRED)
|
||||
endif()
|
||||
|
||||
if (MFEM_USE_MKL_CPARDISO)
|
||||
if (MFEM_USE_MPI)
|
||||
find_package(MKL_CPARDISO REQUIRED MKL_SEQUENTIAL MKL_LP64 MKL_MPI_WRAPPER)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# MFEM_TIMER_TYPE
|
||||
if (NOT DEFINED MFEM_TIMER_TYPE)
|
||||
if (APPLE)
|
||||
@@ -423,10 +285,9 @@ endif()
|
||||
# With newer versions of SuiteSparse which include METIS header using 64-bit
|
||||
# integers, the METIS header (with 32-bit indices, as used by mfem) needs to
|
||||
# be before SuiteSparse.
|
||||
set(MFEM_TPLS MPI_CXX OPENMP HYPRE BLAS LAPACK SuperLUDist METIS SuiteSparse SUNDIALS PETSC
|
||||
SLEPC MESQUITE MUMPS STRUMPACK AXOM CONDUIT Ginkgo GNUTLS GSLIB NETCDF
|
||||
MPFR PUMI HIOP POSIXCLOCKS MFEMBacktrace ZLIB OCCA CEED RAJA UMPIRE ADIOS2
|
||||
CUSPARSE MKL_CPARDISO AMGX CALIPER)
|
||||
set(MFEM_TPLS MPI_CXX OPENMP BLAS LAPACK METIS HYPRE SuiteSparse SUNDIALS PETSC
|
||||
MESQUITE SuperLUDist STRUMPACK AXOM CONDUIT GECKO GNUTLS NETCDF MPFR PUMI
|
||||
POSIXCLOCKS MFEMBacktrace ZLIB)
|
||||
# Add all *_FOUND libraries in the variable TPL_LIBRARIES.
|
||||
set(TPL_LIBRARIES "")
|
||||
set(TPL_INCLUDE_DIRS "")
|
||||
@@ -451,6 +312,9 @@ message(STATUS "MFEM build type: CMAKE_BUILD_TYPE = ${CMAKE_BUILD_TYPE}")
|
||||
message(STATUS "MFEM version: v${MFEM_VERSION_STRING}")
|
||||
message(STATUS "MFEM git string: ${MFEM_GIT_STRING}")
|
||||
|
||||
# Windows specific
|
||||
set(_USE_MATH_DEFINES ${WIN32})
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Define and configure the MFEM library
|
||||
#-------------------------------------------------------------------------------
|
||||
@@ -462,13 +326,6 @@ set(MFEM_SOURCE_DIRS general linalg mesh fem)
|
||||
foreach(DIR IN LISTS MFEM_SOURCE_DIRS)
|
||||
add_subdirectory(${DIR})
|
||||
endforeach()
|
||||
|
||||
if (MFEM_USE_CUDA)
|
||||
set_source_files_properties(${SOURCES} PROPERTIES LANGUAGE CUDA)
|
||||
elseif(MFEM_USE_HIP)
|
||||
set_source_files_properties(${SOURCES} PROPERTIES HIP_SOURCE_PROPERTY_FORMAT TRUE)
|
||||
endif()
|
||||
|
||||
add_subdirectory(config)
|
||||
set(MASTER_HEADERS
|
||||
${PROJECT_SOURCE_DIR}/mfem.hpp
|
||||
@@ -479,13 +336,8 @@ set(CMAKE_INSTALL_RPATH_USE_LINK_PATH ON CACHE BOOL "")
|
||||
set(CMAKE_INSTALL_RPATH "${_lib_path}" CACHE PATH "")
|
||||
set(CMAKE_INSTALL_NAME_DIR "${_lib_path}" CACHE PATH "")
|
||||
|
||||
set(MFEM_SOURCE_DIR ${CMAKE_CURRENT_SOURCE_DIR} CACHE PATH
|
||||
"The MFEM source directory" FORCE)
|
||||
set(MFEM_INSTALL_DIR ${CMAKE_INSTALL_PREFIX} CACHE PATH
|
||||
"The MFEM install directory" FORCE)
|
||||
|
||||
# Declaring the library
|
||||
mfem_add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
|
||||
add_library(mfem ${SOURCES} ${HEADERS} ${MASTER_HEADERS})
|
||||
# message(STATUS "TPL_LIBRARIES = ${TPL_LIBRARIES}")
|
||||
if (CMAKE_VERSION VERSION_GREATER 2.8.11)
|
||||
target_link_libraries(mfem PUBLIC ${TPL_LIBRARIES})
|
||||
@@ -498,11 +350,11 @@ endif()
|
||||
set_target_properties(mfem PROPERTIES VERSION "${mfem_VERSION}")
|
||||
set_target_properties(mfem PROPERTIES SOVERSION "${mfem_VERSION}")
|
||||
|
||||
# If building out-of-source, define MFEM_CONFIG_FILE to point to the config file
|
||||
# inside the build directory.
|
||||
# If building out-of-source, define MFEM_BUILD_DIR to point to the build
|
||||
# directory.
|
||||
if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
|
||||
target_compile_definitions(mfem PRIVATE
|
||||
"MFEM_CONFIG_FILE=\"${PROJECT_BINARY_DIR}/config/_config.hpp\"")
|
||||
"MFEM_BUILD_DIR=${PROJECT_BINARY_DIR}")
|
||||
endif()
|
||||
|
||||
# Generate configuration file in the build directory: config/_config.hpp.
|
||||
@@ -518,7 +370,7 @@ if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
|
||||
"Writing substitute header --> \"${Header}\"")
|
||||
file(WRITE "${PROJECT_BINARY_DIR}/${Header}"
|
||||
"// Auto-generated file.
|
||||
#define MFEM_CONFIG_FILE \"${PROJECT_BINARY_DIR}/config/_config.hpp\"
|
||||
#define MFEM_BUILD_DIR ${PROJECT_BINARY_DIR}
|
||||
#include \"${PROJECT_SOURCE_DIR}/${Header}\"
|
||||
")
|
||||
# This version will be installed in the top include directory:
|
||||
@@ -529,8 +381,6 @@ if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
set(MFEM_CUSTOM_TARGET_PREFIX CACHE STRING "")
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Examples, miniapps, and testing
|
||||
#-------------------------------------------------------------------------------
|
||||
@@ -538,9 +388,6 @@ set(MFEM_CUSTOM_TARGET_PREFIX CACHE STRING "")
|
||||
# Enable testing if required
|
||||
if (MFEM_ENABLE_TESTING)
|
||||
enable_testing()
|
||||
set(MFEM_ALL_TESTS_TARGET_NAME tests)
|
||||
add_mfem_target(${MFEM_ALL_TESTS_TARGET_NAME} OFF)
|
||||
add_subdirectory(tests EXCLUDE_FROM_ALL)
|
||||
endif()
|
||||
|
||||
# Define a target that all examples and miniapps will depend on.
|
||||
@@ -560,9 +407,7 @@ add_subdirectory(miniapps EXCLUDE_FROM_ALL)
|
||||
# Target to build all executables, i.e. everything.
|
||||
add_custom_target(exec)
|
||||
add_dependencies(exec
|
||||
${MFEM_ALL_EXAMPLES_TARGET_NAME}
|
||||
${MFEM_ALL_MINIAPPS_TARGET_NAME}
|
||||
${MFEM_ALL_TESTS_TARGET_NAME})
|
||||
${MFEM_ALL_EXAMPLES_TARGET_NAME} ${MFEM_ALL_MINIAPPS_TARGET_NAME})
|
||||
# Here, we want to "add_dependencies(test exec)". However, dependencies for
|
||||
# 'test' (and other built-in targets) can not be added with add_dependencies():
|
||||
# - https://gitlab.kitware.com/cmake/cmake/issues/8438
|
||||
@@ -581,17 +426,16 @@ if (NOT ("${PROJECT_SOURCE_DIR}" STREQUAL "${PROJECT_BINARY_DIR}"))
|
||||
endif()
|
||||
|
||||
# Add 'check' target - quick test
|
||||
set(MFEM_CHECK_TARGET_NAME ${MFEM_CUSTOM_TARGET_PREFIX}check)
|
||||
if (NOT MFEM_USE_MPI)
|
||||
add_custom_target(${MFEM_CHECK_TARGET_NAME}
|
||||
${CMAKE_CTEST_COMMAND} -R \"^ex1_ser\" -C ${CMAKE_CFG_INTDIR}
|
||||
add_custom_target(check
|
||||
${CMAKE_CTEST_COMMAND} -R '^ex1_ser' -C ${CMAKE_CFG_INTDIR}
|
||||
USES_TERMINAL)
|
||||
add_dependencies(${MFEM_CHECK_TARGET_NAME} ex1)
|
||||
add_dependencies(check ex1)
|
||||
else()
|
||||
add_custom_target(${MFEM_CHECK_TARGET_NAME}
|
||||
${CMAKE_CTEST_COMMAND} -R \"^ex1p\" -C ${CMAKE_CFG_INTDIR}
|
||||
add_custom_target(check
|
||||
${CMAKE_CTEST_COMMAND} -R '^ex1p' -C ${CMAKE_CFG_INTDIR}
|
||||
USES_TERMINAL)
|
||||
add_dependencies(${MFEM_CHECK_TARGET_NAME} ex1p)
|
||||
add_dependencies(check ex1p)
|
||||
endif()
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
@@ -634,20 +478,6 @@ install(DIRECTORY ${MFEM_SOURCE_DIRS}
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
|
||||
FILES_MATCHING PATTERN "*.hpp")
|
||||
|
||||
# Install the okl files
|
||||
if (MFEM_USE_OCCA)
|
||||
install(DIRECTORY ${MFEM_SOURCE_DIRS}
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
|
||||
FILES_MATCHING PATTERN "*.okl")
|
||||
endif()
|
||||
|
||||
# Install the libCEED files
|
||||
if (MFEM_USE_CEED)
|
||||
install(DIRECTORY ${MFEM_SOURCE_DIRS}
|
||||
DESTINATION ${INSTALL_INCLUDE_DIR}/mfem
|
||||
FILES_MATCHING PATTERN "fem/ceed/*.h")
|
||||
endif()
|
||||
|
||||
# Install ${HEADERS}
|
||||
# ---
|
||||
# foreach (HDR ${HEADERS})
|
||||
@@ -713,10 +543,3 @@ install(FILES
|
||||
# Install the export set for use with the install-tree
|
||||
install(EXPORT ${PROJECT_NAME_UC}Targets
|
||||
DESTINATION ${INSTALL_CMAKE_DIR})
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Create 'config.mk' from 'config.mk.in' for the build and install locations and
|
||||
# define install rules for 'config.mk' and 'test.mk'
|
||||
#-------------------------------------------------------------------------------
|
||||
|
||||
mfem_export_mk_files()
|
||||
|
||||
+120
-242
@@ -1,41 +1,33 @@
|
||||
<p align="center">
|
||||
<a href="https://mfem.org/"><img alt="mfem" src="https://mfem.org/img/logo-300.png"></a>
|
||||
<a href="http://mfem.org/"><img alt="mfem" src="http://mfem.org/img/logo-300.png"></a>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://github.com/mfem/mfem/blob/master/LICENSE"><img alt="License" src="https://img.shields.io/badge/License-BSD-brightgreen.svg"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Arepo-check+branch%3Amaster"><img alt="Repo check" src="https://github.com/mfem/mfem/actions/workflows/repo-check.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuild-analysis+branch%3Amaster"><img alt="Build Analysis" src="https://github.com/mfem/mfem/actions/workflows/mfem-analysis.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/actions?query=workflow%3Abuilds-and-tests+branch%3Amaster"><img alt="Builds and Tests" src="https://github.com/mfem/mfem/actions/workflows/builds-and-tests.yml/badge.svg?branch=master"></a>
|
||||
<a href="https://github.com/mfem/mfem/blob/master/COPYRIGHT"><img alt="License" src="https://img.shields.io/badge/License-LGPL--2.1-brightgreen.svg"></a>
|
||||
<a href="https://travis-ci.org/mfem/mfem"><img alt="Build Status" src="https://travis-ci.org/mfem/mfem.svg?branch=master"></a>
|
||||
<a href="https://ci.appveyor.com/project/mfem/mfem"><img alt="Build Status" src="https://ci.appveyor.com/api/projects/status/19non9sqm6msi2wy?svg=true"></a>
|
||||
<a href="https://mfem.github.io/doxygen/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
<a href="http://mfem.github.io/doxygen/html/index.html"><img alt="Doxygen" src="https://img.shields.io/badge/code-documented-brightgreen.svg"></a>
|
||||
</p>
|
||||
|
||||
|
||||
# How to Contribute
|
||||
|
||||
The MFEM team welcomes contributions at all levels: bugfixes; code improvements;
|
||||
simplifications; new mesh, discretization or solver capabilities; improved
|
||||
documentation; new examples and miniapps; HPC performance improvements; etc.
|
||||
The MFEM team welcomes contributions at all levels: bugfixes; code
|
||||
improvements; simplifications; new mesh, discretization or solver
|
||||
capabilities; improved documentation; new examples and miniapps;
|
||||
HPC performance improvements; ...
|
||||
|
||||
MFEM is distributed under the terms of the BSD-3 license. All new contributions
|
||||
must be made under this license.
|
||||
|
||||
If you plan on contributing to MFEM, consider reviewing the
|
||||
[issue tracker](https://github.com/mfem/mfem/issues) first to check if a thread
|
||||
already exists for your desired feature or the bug you ran into. Use a pull
|
||||
request (PR) toward the `mfem:master` branch to propose your contribution. If
|
||||
you are planning significant code changes or have questions, you may want to
|
||||
open an [issue](https://github.com/mfem/mfem/issues) before issuing a PR. In
|
||||
addition to technical contributions, we are also interested in your results and
|
||||
[simulation images](https://mfem.org/gallery/), which you can share via a pull
|
||||
request in the [mfem/web](https://github.com/mfem/web) repo.
|
||||
Use a pull request (PR) toward the `mfem:master` branch to propose your
|
||||
contribution. If you are planning significant code changes, or have any
|
||||
questions, you can also open an [issue](https://github.com/mfem/mfem/issues)
|
||||
before issuing a PR. We also welcome your [simulation
|
||||
images](http://mfem.org/gallery/), which you can submit via a pull request in
|
||||
[mfem/web](https://github.com/mfem/web).
|
||||
|
||||
See the [Quick Summary](#quick-summary) section for the main highlights of our
|
||||
GitHub workflow. For more details, consult the following sections and refer
|
||||
back to them before issuing pull requests:
|
||||
|
||||
- [Quick Summary](#quick-summary)
|
||||
- [Code Overview](#code-overview)
|
||||
- [GitHub Workflow](#github-workflow)
|
||||
- [MFEM Organization](#mfem-organization)
|
||||
@@ -53,7 +45,7 @@ back to them before issuing pull requests:
|
||||
Contributing to MFEM requires knowledge of Git and, likely, finite elements. If
|
||||
you are new to Git, see the [GitHub learning
|
||||
resources](https://help.github.com/articles/git-and-github-learning-resources/).
|
||||
To learn more about the finite element method, see our [FEM page](https://mfem.org/fem).
|
||||
To learn more about the finite element method, see our [FEM page](http://mfem.org/fem).
|
||||
|
||||
*By submitting a pull request, you are affirming the [Developer's Certificate of
|
||||
Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
@@ -65,17 +57,8 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
development branches off `mfem:master`.
|
||||
- Please follow the [developer guidelines](#developer-guidelines), in particular
|
||||
with regards to documentation and code styling.
|
||||
- Please do not commit large/binary files to the central repository (use a fork
|
||||
instead).
|
||||
- Pull requests should be issued toward `mfem:master`. Make sure
|
||||
to check the items off the [Pull Request Checklist](#pull-request-checklist).
|
||||
- When your contribution is fully working and ready to be reviewed, add
|
||||
the `ready-for-review` label.
|
||||
- PRs are treated similarly to journal submission with an "editor" assigning two
|
||||
reviewers to evaluate the changes.
|
||||
- The reviewers have 3 weeks to evaluate the PR and work with the author to
|
||||
fix issues and implement improvements.
|
||||
- During review there should be no force pushes/rewriting history in the branch.
|
||||
- After approval, MFEM developers merge the PR manually in the [mfem:next branch](#masternext-workflow).
|
||||
- After a week of testing in `mfem:next`, the original PR is merged in `mfem:master`.
|
||||
- We use [milestones](https://github.com/mfem/mfem/milestones) to coordinate the
|
||||
@@ -85,141 +68,98 @@ Origin](#developers-certificate-of-origin-11) at the end of this file.*
|
||||
|
||||
### Code Overview
|
||||
|
||||
#### Source code structure:
|
||||
- The MFEM library uses object-orient design principles which reflect, in code,
|
||||
the independent mathematical concepts of meshing, linear algebra and finite
|
||||
element spaces and operators.
|
||||
|
||||
The MFEM library uses object-oriented design principles which reflect, in code,
|
||||
the independent mathematical concepts of meshing, linear algebra and finite
|
||||
element spaces and operators.
|
||||
|
||||
The MFEM source code has the following structure:
|
||||
|
||||
```
|
||||
- The MFEM source code has the following structure:
|
||||
```
|
||||
.
|
||||
├── config
|
||||
│ ├── cmake
|
||||
│ │ └── ...
|
||||
│ └── githooks
|
||||
│ └── cmake
|
||||
│ └── modules
|
||||
├── data
|
||||
├── doc
|
||||
│ └── web
|
||||
│ └── examples
|
||||
├── examples
|
||||
│ ├── amgx
|
||||
│ ├── caliper
|
||||
│ ├── ginkgo
|
||||
│ ├── hiop
|
||||
│ ├── petsc
|
||||
│ ├── pumi
|
||||
│ ├── sundials
|
||||
| └── superlu
|
||||
│ └── sundials
|
||||
├── fem
|
||||
│ ├── ceed
|
||||
│ ├── qinterp
|
||||
│ └── tmop
|
||||
├── general
|
||||
├── linalg
|
||||
│ └── simd
|
||||
├── mesh
|
||||
├── miniapps
|
||||
│ ├── adjoint
|
||||
│ ├── common
|
||||
│ ├── electromagnetics
|
||||
│ ├── gslib
|
||||
│ ├── meshing
|
||||
│ ├── mtop
|
||||
│ ├── navier
|
||||
│ ├── nurbs
|
||||
│ ├── performance
|
||||
│ ├── shifted
|
||||
│ ├── solvers
|
||||
│ ├── tools
|
||||
│ └── toys
|
||||
└── tests
|
||||
├── convergence
|
||||
├── gitlab
|
||||
├── par-mesh-format
|
||||
├── scripts
|
||||
└── unit
|
||||
└── ...
|
||||
```
|
||||
└── miniapps
|
||||
├── common
|
||||
├── electromagnetics
|
||||
├── meshing
|
||||
├── nurbs
|
||||
├── performance
|
||||
└── tools
|
||||
```
|
||||
|
||||
#### Main directories and classes
|
||||
|
||||
The main directories are `fem/`, `mesh/` and `linalg/` containing the C++
|
||||
classes implementing the finite element, mesh and linear algebra concepts
|
||||
respectively.
|
||||
- The main directories are `fem/`, `mesh/` and `linalg/` containing the C++
|
||||
classes implementing the finite element, mesh and linear algebra concepts
|
||||
respectively.
|
||||
|
||||
- The main mesh classes are:
|
||||
+ [`Mesh`](https://mfem.github.io/doxygen/html/classmfem_1_1Mesh.html)
|
||||
+ [`NCMesh`](https://mfem.github.io/doxygen/html/classmfem_1_1NCMesh.html)
|
||||
+ [`Element`](https://mfem.github.io/doxygen/html/classmfem_1_1Element.html)
|
||||
+ [`ElementTransformation`](https://mfem.github.io/doxygen/html/classmfem_1_1ElementTransformation.html)
|
||||
+ [`Mesh`](http://mfem.github.io/doxygen/html/classmfem_1_1Mesh.html)
|
||||
+ [`NCMesh`](http://mfem.github.io/doxygen/html/classmfem_1_1NCMesh.html)
|
||||
+ [`Element`](http://mfem.github.io/doxygen/html/classmfem_1_1Element.html)
|
||||
+ [`ElementTransformation`](http://mfem.github.io/doxygen/html/classmfem_1_1ElementTransformation.html)
|
||||
|
||||
- The main finite element classes are:
|
||||
+ [`FiniteElement`](https://mfem.github.io/doxygen/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementCollection`](https://mfem.github.io/doxygen/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementSpace`](https://mfem.github.io/doxygen/html/classmfem_1_1FiniteElementSpace.html)
|
||||
+ [`GridFunction`](https://mfem.github.io/doxygen/html/classmfem_1_1GridFunction.html)
|
||||
+ [`BilinearFormIntegrator`](https://mfem.github.io/doxygen/html/classmfem_1_1BilinearFormIntegrator.html) and [`LinearFormIntegrator`](https://mfem.github.io/doxygen/html/classmfem_1_1LinearFormIntegrator.html)
|
||||
+ [`LinearForm`](https://mfem.github.io/doxygen/html/classmfem_1_1LinearFormIntegrator.html), [`BilinearForm`](https://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](https://mfem.github.io/doxygen/html/classmfem_1_1MixedBilinearForm.html)
|
||||
+ [`FiniteElement`](http://mfem.github.io/doxygen/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementCollection`](http://mfem.github.io/doxygen/html/classmfem_1_1FiniteElement.html)
|
||||
+ [`FiniteElementSpace`](http://mfem.github.io/doxygen/html/classmfem_1_1FiniteElementSpace.html)
|
||||
+ [`GridFunction`](http://mfem.github.io/doxygen/html/classmfem_1_1GridFunction.html)
|
||||
+ [`BilinearFormIntegrator`](http://mfem.github.io/doxygen/html/classmfem_1_1BilinearFormIntegrator.html) and [`LinearFormIntegrator`](http://mfem.github.io/doxygen/html/classmfem_1_1LinearFormIntegrator.html)
|
||||
+ [`LinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1LinearFormIntegrator.html), [`BilinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html) and [`MixedBilinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1MixedBilinearForm.html)
|
||||
|
||||
- The main linear algebra classes and sources are
|
||||
+ [`Operator`](https://mfem.github.io/doxygen/html/classmfem_1_1Operator.html) and [`BilinearForm`](https://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html)
|
||||
+ [`Vector`](https://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html) and [`LinearForm`](https://mfem.github.io/doxygen/html/classmfem_1_1LinearForm.html)
|
||||
+ [`DenseMatrix`](https://mfem.github.io/doxygen/html/classmfem_1_1DenseMatrix.html) and [`SparseMatrix`](https://mfem.github.io/doxygen/html/classmfem_1_1SparseMatrix.html)
|
||||
+ Sparse [smoothers](https://mfem.github.io/doxygen/html/sparsesmoothers_8hpp.html) and linear [solvers](https://mfem.github.io/doxygen/html/solvers_8hpp.html)
|
||||
+ [`Operator`](http://mfem.github.io/doxygen/html/classmfem_1_1Operator.html) and [`BilinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html)
|
||||
+ [`Vector`](http://mfem.github.io/doxygen/html/classmfem_1_1BilinearForm.html) and [`LinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1LinearForm.html)
|
||||
+ [`DenseMatrix`](http://mfem.github.io/doxygen/html/classmfem_1_1DenseMatrix.html) and [`SparseMatrix`](http://mfem.github.io/doxygen/html/classmfem_1_1SparseMatrix.html)
|
||||
+ Sparse [smoothers](http://mfem.github.io/doxygen/html/sparsesmoothers_8hpp.html) and linear [solvers](http://mfem.github.io/doxygen/html/solvers_8hpp.html)
|
||||
|
||||
#### Parallel implementation
|
||||
|
||||
Parallel MPI objects in MFEM inherit their serial counterparts, so a parallel
|
||||
mesh for example is just a serial mesh on each task plus the information on
|
||||
shared geometric entities between different tasks. The parallel source files
|
||||
have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
|
||||
- Parallel MPI objects in MFEM inherit their serial counterparts, so a parallel
|
||||
mesh for example is just a serial mesh on each task plus the information on
|
||||
shared geometric entities between different tasks. The parallel source files
|
||||
have a `p` prefix, e.g. `pmesh.cpp` vs. the serial `mesh.cpp`.
|
||||
|
||||
- The main parallel classes are
|
||||
+ [`ParMesh`](https://mfem.github.io/doxygen/html/solvers_8hpp.html)
|
||||
+ [`ParNCMesh`](https://mfem.github.io/doxygen/html/classmfem_1_1ParMesh.html)
|
||||
+ [`ParFiniteElementSpace`](https://mfem.github.io/doxygen/html/classmfem_1_1ParFiniteElementSpace.html)
|
||||
+ [`ParGridFunction`](https://mfem.github.io/doxygen/html/classmfem_1_1ParGridFunction.html)
|
||||
+ [`ParBilinearForm`](https://mfem.github.io/doxygen/html/classmfem_1_1ParBilinearForm.html) and [`ParLinearForm`](https://mfem.github.io/doxygen/html/classmfem_1_1ParLinearForm.html)
|
||||
+ [`HypreParMatrix`](https://mfem.github.io/doxygen/html/classmfem_1_1HypreParMatrix.html) and [`HypreParVector`](https://mfem.github.io/doxygen/html/classmfem_1_1HypreParVector.html)
|
||||
+ [`HypreSolver`](https://mfem.github.io/doxygen/html/classmfem_1_1HypreSolver.html) and other [hypre classes](https://mfem.github.io/doxygen/html/hypre_8hpp.html)
|
||||
+ [`ParMesh`](http://mfem.github.io/doxygen/html/solvers_8hpp.html)
|
||||
+ [`ParNCMesh`](http://mfem.github.io/doxygen/html/classmfem_1_1ParMesh.html)
|
||||
+ [`ParFiniteElementSpace`](http://mfem.github.io/doxygen/html/classmfem_1_1ParFiniteElementSpace.html)
|
||||
+ [`ParGridFunction`](http://mfem.github.io/doxygen/html/classmfem_1_1ParGridFunction.html)
|
||||
+ [`ParBilinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1ParBilinearForm.html) and [`ParLinearForm`](http://mfem.github.io/doxygen/html/classmfem_1_1ParLinearForm.html)
|
||||
+ [`HypreParMatrix`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreParMatrix.html) and [`HypreParVector`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreParVector.html)
|
||||
+ [`HypreSolver`](http://mfem.github.io/doxygen/html/classmfem_1_1HypreSolver.html) and other [hypre classes](http://mfem.github.io/doxygen/html/hypre_8hpp.html)
|
||||
|
||||
#### GPU and general device support
|
||||
|
||||
GPU and multi-core CPU support is based on device kernels supporting different
|
||||
backends (CUDA, OCCA, RAJA, OpenMP, etc.) and an internal lightweight
|
||||
device/host memory manager.
|
||||
|
||||
- The main device-relevant classes and sources are:
|
||||
+ [`Device`](https://mfem.github.io/doxygen/html/device_8hpp.html)
|
||||
+ [`MemoryManager`](https://mfem.github.io/doxygen/html/mem_manager_8hpp.html)
|
||||
+ the [`MFEM_FORALL`](https://mfem.github.io/doxygen/html/forall_8hpp.html) macro
|
||||
+ the [`cuda.hpp`](https://mfem.github.io/doxygen/html/cuda_8hpp.html) and [`occa.hpp`](https://mfem.github.io/doxygen/html/occa_8hpp.html) files
|
||||
|
||||
#### Utilities, building and documentation
|
||||
- The `general/` directory contains C++ classes that serve as utilities for
|
||||
communication, error handling, arrays, (Boolean) tables, timing, etc.
|
||||
|
||||
- The `config/` directory contains build-related files, both for the plain
|
||||
Makefile and the CMake build options.
|
||||
|
||||
- The `doc/` directory contains configuration for the Doxygen code documentation
|
||||
that can either be built locally or browsed online at
|
||||
https://mfem.github.io/doxygen/html/index.html.
|
||||
that can either be build locally, or browsed online at
|
||||
http://mfem.github.io/doxygen/html/index.html.
|
||||
|
||||
#### Examples and tests
|
||||
- `examples` and `miniapps` respectively gather simple and more fully-featured
|
||||
demonstrations of the usage on MFEM. They both rely on `data/` for the
|
||||
collection of meshes.
|
||||
- The `tests/` directory contains a unit test suite and will later contain more
|
||||
tests that run example codes.
|
||||
- The `data/` directory contains a collection of small mesh files, that are used
|
||||
in the simple example codes and more fully-featured mini applications in the
|
||||
`examples/` and `miniapps/` directories.
|
||||
|
||||
See also the [code overview](https://mfem.org/code-overview/) section on the MFEM
|
||||
website.
|
||||
- See also the [code overview](http://mfem.org/code-overview/) section on the
|
||||
MFEM website.
|
||||
|
||||
## GitHub Workflow
|
||||
|
||||
The MFEM GitHub organization: https://github.com/mfem, is the main developer hub
|
||||
for the MFEM project.
|
||||
The GitHub organization, https://github.com/mfem, is the main developer hub for
|
||||
the MFEM project.
|
||||
|
||||
If you plan to make contributions or would like to stay up-to-date with changes
|
||||
If you plan to make contributions or will like to stay up-to-date with changes
|
||||
in the code, *we strongly encourage you to [join the MFEM organization](#mfem-organization)*.
|
||||
|
||||
This will simplify the workflow (by providing you additional permissions), and
|
||||
@@ -228,31 +168,34 @@ will allow us to reach you directly with project announcements.
|
||||
|
||||
### MFEM Organization
|
||||
|
||||
#### Getting started (GitHub)
|
||||
Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
+ Create the account at: [github.com/join](https://github.com/join).
|
||||
+ For easy identification, please add your full name and maybe a picture of you at:
|
||||
https://github.com/settings/profile.
|
||||
- Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
+ Create the account at: github.com/join.
|
||||
+ For easy identification, please add your name and maybe a picture of you at: https://github.com/settings/profile.
|
||||
+ To receive notification, set a primary email at: https://github.com/settings/emails.
|
||||
+ For password-less pull/push over SSH, add your SSH keys at: https://github.com/settings/keys.
|
||||
|
||||
#### Joining
|
||||
- [Contact us](#contact-information) for an invitation to join the MFEM GitHub
|
||||
organization. You will receive an invitation email, which you can directly accept.
|
||||
organization.
|
||||
|
||||
- You should receive an invitation email, which you can directly accept.
|
||||
Alternatively, *after logging into GitHub*, you can accept the invitation at
|
||||
the top of https://github.com/mfem.
|
||||
|
||||
- Consider making your membership public by going to https://github.com/orgs/mfem/people
|
||||
and clicking on the organization visibility drop box next to your name.
|
||||
and clicking on the organization visibility dropbox next to your name.
|
||||
|
||||
- Project discussions and announcements will be posted at
|
||||
https://github.com/orgs/mfem/teams/everyone.
|
||||
|
||||
#### Structure
|
||||
- The MFEM source code is in the [mfem](https://github.com/mfem/mfem)
|
||||
repository.
|
||||
|
||||
- The website and corresponding documentation are in the
|
||||
[web](https://github.com/mfem/web) repository.
|
||||
|
||||
- The [PyMFEM](https://github.com/mfem/PyMFEM) repository contains a Python
|
||||
wrapper for MFEM.
|
||||
|
||||
- The [data](https://github.com/mfem/data) repository contains additional
|
||||
(large) datafiles for MFEM.
|
||||
|
||||
@@ -260,9 +203,9 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
### New Feature Development
|
||||
|
||||
- A new feature should be important enough that at least one person, the
|
||||
author, is willing to work on it and be its champion.
|
||||
proposer, is willing to work on it and be its champion.
|
||||
|
||||
- The author creates a branch for the new feature (with suffix `-dev`), off
|
||||
- The proposer creates a branch for the new feature (with suffix `-dev`), off
|
||||
the `master` branch, or another existing feature branch, for example:
|
||||
|
||||
```
|
||||
@@ -305,13 +248,12 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
- Lean code base is easier to understand by new collaborators.
|
||||
- New features should be added only if they are necessary or generally useful.
|
||||
- Introduction of language constructions not currently used in MFEM should be
|
||||
justified and generally avoided (to maintain portability to various systems
|
||||
and compilers, including early access hardware).
|
||||
justified and generally avoided (so we can build on cutting-edge systems).
|
||||
- We prefer basic C++ and the C++03 standard, to keep the code readable by
|
||||
a large audience and to make sure it compiles anywhere.
|
||||
|
||||
- *Keep the code general and reasonably efficient*
|
||||
- The main goal is fast prototyping for research and application development.
|
||||
- Main goal is fast prototyping for research.
|
||||
- When in doubt, generality wins over efficiency.
|
||||
- Respect the needs of different users (current and/or future).
|
||||
|
||||
@@ -354,49 +296,15 @@ Before you can start, you need a GitHub account, here are a few suggestions:
|
||||
|
||||
`[DISCUSS] Hybridized DG [hdg-dev]`
|
||||
|
||||
- If the PR is still a work in progress, add the `WIP` label to it and
|
||||
optionally the `[WIP]` prefix in the title.
|
||||
|
||||
- Add a description, appropriate labels and assign yourself to the PR. The MFEM
|
||||
team will add reviewers as appropriate.
|
||||
|
||||
- List outstanding TODO items in the description, see PR #222 for an example.
|
||||
|
||||
- When your contribution is fully working and ready to be reviewed, add
|
||||
the `ready-for-review` label.
|
||||
- Track the Travis CI and Appveyor [continuous integration](#automated-testing)
|
||||
builds at the end of the PR. These should run clean, so address any errors as
|
||||
soon as possible.
|
||||
|
||||
- PRs are treated similarly to journal submission with an "editor" assigning
|
||||
two reviewers to evaluate the changes. The reviewers have 3 weeks to evaluate
|
||||
the PR and work with the author to implement improvements and fix issues.
|
||||
|
||||
- Once the `ready-for-review` label has been applied and reviewers have been
|
||||
assigned, the PR is considered under review. To help with the review process
|
||||
there should be no force pushes/rewriting history in the branch.
|
||||
|
||||
- After approval, the PR is [tested](#masternext-workflow) for a week with
|
||||
other approved PRs in the `mfem:next` branch.
|
||||
|
||||
- Consider manually running the tests in `tests/scripts` before merging in
|
||||
`mfem:next`, see the [README](tests/scripts/README) file in that directory
|
||||
for more details.
|
||||
|
||||
- Track the GitHub Actions and Appveyor [continuous integration](#automated-testing)
|
||||
builds at the end of the PR. These should generally run clean, so address any
|
||||
errors as soon as possible. Please ask if you are unsure how to do that.
|
||||
|
||||
- Note that some tests, such as the `branch-history` check in GitHub Actions
|
||||
are safeguards that are allowed to fail in certain cases.
|
||||
|
||||
- Other tests, such as the `code-style`, `documentation` and `gitignore`
|
||||
checks in GitHub Actions enforce MFEM-specific rules which are explained in
|
||||
the error messages and the `tests/scripts` directory.
|
||||
|
||||
- Also note that the tests `branch-history` and `repos-checks` found in GitHub
|
||||
Actions can be triggered automatically before each push using git hooks. See
|
||||
the [git hooks README](config/githooks/README.md) for a detailed explanation.
|
||||
|
||||
- If triggered, track the status of the LLNL GitLab tests. If failing, ask
|
||||
one of the _LLNL developers_ for details.
|
||||
|
||||
### Pull Request Checklist
|
||||
|
||||
@@ -408,18 +316,14 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Is this a new feature users need to be aware of? New or updated example or miniapp?
|
||||
- [ ] Does it make sense to create a new section in the `CHANGELOG` to group with other related features?
|
||||
- [ ] Update `INSTALL`:
|
||||
- [ ] Had a new optional library been added? If so, what range of versions of this library are required? (*Make sure the external library is compatible with our BSD license, e.g. it is not licensed under GPL!*)
|
||||
- [ ] Have the version ranges for any required or optional libraries changed?
|
||||
- [ ] Has a new optional library been added? (*Make sure the external library is licensed under LGPL, not GPL!*)
|
||||
- [ ] Does `make` or `cmake` have a new target?
|
||||
- [ ] Did the requirements or the installation process change? *(rare)*
|
||||
- [ ] Update continuous integration server configurations if necessary (e.g. with new version requirements for each of MFEM's dependencies)
|
||||
- [ ] `.github`
|
||||
- [ ] `.appveyor.yml`
|
||||
- [ ] Did the requirements or the installation process change? *(rare)*.
|
||||
- [ ] Update `.gitignore`:
|
||||
- [ ] Check if `make distclean; git status` shows any files that were generated from the source by the project (not an IDE) but we don't want to track in the repository.
|
||||
- [ ] Check if `make distclean; git status` shows any files that are generated from the source but we don't want to track in the repository.
|
||||
- [ ] Add new patterns (just for the new files above) and re-run the above test.
|
||||
- [ ] New examples:
|
||||
- [ ] All sample runs at the top of the example source file work.
|
||||
- [ ] All sample runs at the top of the example work.
|
||||
- [ ] Update `examples/makefile`:
|
||||
- [ ] Add the example code to the appropriate `SEQ_EXAMPLES` and `PAR_EXAMPLES` variables.
|
||||
- [ ] Add any files generated by it to the `clean` target.
|
||||
@@ -428,14 +332,13 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Add the example code to the `ALL_EXE_SRCS` variable.
|
||||
- [ ] Make sure `THIS_TEST_OPTIONS` is set correctly for the new example.
|
||||
- [ ] List the new example in `doc/CodeDocumentation.dox`.
|
||||
- [ ] If new examples directory (e.g.`examples/pumi`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add example-specific documentation, see e.g. the `src/examples.md`.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
- [ ] In `examples.md`, list the example under the appropriate categories, add new categories if necessary.
|
||||
- [ ] Add a short description of the example in the "Extensive Examples" section of `features.md`.
|
||||
- [ ] New miniapps:
|
||||
- [ ] All sample runs at the top of the miniapp source file work.
|
||||
- [ ] All sample runs at the top of the miniapp work.
|
||||
- [ ] Update top-level `makefile` and `makefile` in corresponding miniapp directory.
|
||||
- [ ] Add the miniapp binary and any files generated by it to the top-level `.gitignore` file.
|
||||
- [ ] Update CMake build system:
|
||||
@@ -443,8 +346,6 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] Add/update the `CMakeLists.txt` file in the new miniapp directory.
|
||||
- [ ] Consider adding a new test for the new miniapp.
|
||||
- [ ] List the new miniapp in `doc/CodeDocumentation.dox`
|
||||
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), add it to `MINIAPP_SUBDIRS` in the `makefile`.
|
||||
- [ ] If new miniapps directory (e.g.`miniapps/nurbs`), list it in `doc/CodeDocumentation.conf.in`
|
||||
- [ ] Companion pull request for documentation in [mfem/web](https://github.com/mfem/web) repo:
|
||||
- [ ] Update or add miniapp-specific documentation, see e.g. the `src/meshing.md` and `src/electromagnetics.md` files.
|
||||
- [ ] Add the description, labels and screenshots in `src/examples.md` and `src/img`.
|
||||
@@ -454,13 +355,18 @@ Before a PR can be merged, it should satisfy the following:
|
||||
- [ ] All significant new classes, methods and functions have Doxygen-style documentation in source comments.
|
||||
- [ ] Consider adding new sample runs in existing examples to highlight the new capability.
|
||||
- [ ] Consider saving cool simulation pictures with the new capability in the Confluence gallery (LLNL only) or submitting them, via pull request, to the gallery section of the `mfem/web` repo.
|
||||
- [ ] If this is a major new feature, consider mentioning it in the short summary inside `README` *(rare)*.
|
||||
- [ ] If this is a major new feature, consider mentioning in the short summary inside `README` *(rare)*.
|
||||
- [ ] List major new classes in `doc/CodeDocumentation.dox` *(rare)*.
|
||||
- [ ] Update this checklist, if the new pull request affects it.
|
||||
- [ ] Run `make unittest` to make sure all unit tests pass.
|
||||
- [ ] Run the tests in `tests/scripts`.
|
||||
- [ ] (LLNL only) Clone the `tests` repository and run the following tests, see `mfem/tests/README.md`:
|
||||
- [ ] `compilers`
|
||||
- [ ] `memcheck`
|
||||
- [ ] `unit-test`
|
||||
- [ ] `documentation`
|
||||
- [ ] (LLNL only) After merging:
|
||||
- [ ] Update internal tests to include the new features.
|
||||
- [ ] Regenerate `README.html` files from companion documentation pull requests.
|
||||
- [ ] Update the `baseline` and `compiler` tests, add new tests if necessary.
|
||||
- [ ] Consider updating the script `mfem/tests/sample-runs` (`sample-runs-serial` and `sample-runs-parallel`).
|
||||
|
||||
### Master/Next Workflow
|
||||
|
||||
@@ -519,7 +425,7 @@ MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
(replace `3.3.2` with current release):
|
||||
- Rename the current `next` branch to `next-pre-v3.3.2`.
|
||||
- Create a new `next` branch starting from the `v3.3.2` release.
|
||||
- Local copies of `next` can then be updated with `git fetch origin next && git checkout -B next origin/next`.
|
||||
- Local copies of `next` can then be updated with `git checkout -B next origin/next`.
|
||||
|
||||
### Release Checklist
|
||||
|
||||
@@ -528,14 +434,7 @@ MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
- [ ] `makefile`
|
||||
- [ ] `CMakeLists.txt`
|
||||
- [ ] `doc/CodeDocumentation.conf.in`
|
||||
- [ ] Check that version requirements for each of MFEM's dependencies are documented in `INSTALL` and up-to-date
|
||||
- [ ] Check that continuous integration server configurations reflect the dependency version requirements of the new release
|
||||
- [ ] `.github`
|
||||
- [ ] `.appveyor.yml`
|
||||
- [ ] Update the `CHANGELOG` to organize all release contributions
|
||||
- [ ] Review the whole source code once over
|
||||
- [ ] Ask MFEM-based applications to test the pre-release branch
|
||||
- [ ] Test on additional platforms and compilers
|
||||
- [ ] (LLNL only) Make sure all `README.html` files in the source repo are up to date.
|
||||
- [ ] Tag the repository:
|
||||
|
||||
```
|
||||
@@ -545,17 +444,16 @@ MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
- [ ] Create the release tarball and push to `mfem/releases`.
|
||||
- [ ] Recreate the `next` branch as described in previous section.
|
||||
- [ ] Update and push documentation to `mfem/doxygen`.
|
||||
- [ ] Update URL shortlinks:
|
||||
- [ ] Create a shortlink at [http://bit.ly/](http://bit.ly/) for the release tarball, e.g. https://mfem.github.io/releases/mfem-3.1.tgz.
|
||||
- [ ] (LLNL only) Add and commit the new shortlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
|
||||
- [ ] Update URL shorlinks:
|
||||
- [ ] Create a shortlink at [https://goo.gl/](https://goo.gl/) for the release tarball, e.g. http://mfem.github.io/releases/mfem-3.1.tgz.
|
||||
- [ ] (LLNL only) Add and commit the new shorlink in the `links` and `links-mfem` files of the internal `mfem/downloads` repo.
|
||||
- [ ] Add the new shortlinks to the MFEM packages in `spack`, `homebrew/science`, `VisIt`, etc.
|
||||
- [ ] Update website in `mfem/web` repo:
|
||||
- Update version and shortlinks in `src/index.md` and `src/download.md`.
|
||||
- Use [cloc-1.62.pl](http://cloc.sourceforge.net/) and `ls -lh` to estimate the SLOC and the tarball size in `src/download.md`.
|
||||
|
||||
## LLNL Workflow
|
||||
|
||||
### Mirroring on Bitbucket
|
||||
## LLNL Workflow
|
||||
|
||||
- The GitHub `master` and `next` branches are mirrored to the LLNL institutional
|
||||
Bitbucket repository as `gh-master` and `gh-next`.
|
||||
@@ -574,34 +472,18 @@ MFEM uses a `master`/`next`-branch workflow as described below:
|
||||
- `mfem:gh-next` -- Bleeding-edge development version, may be broken, use at
|
||||
your own risk.
|
||||
|
||||
### Mirroring on GitLab
|
||||
|
||||
- MFEM repository is also mirrored on the LLNL GitLab instance, in a
|
||||
semi-automated manner.
|
||||
|
||||
- This instance is meant to complete CI testing with tests on Livermore
|
||||
Computing systems. Gitlab pipeline status is reported in the corresponding
|
||||
GitHub pull request.
|
||||
|
||||
- In Gitlab pipelines, TPLs (dependencies) are built using Spack, driven by Uberenv.
|
||||
|
||||
- No change to the MFEM repo can be made on this instance.
|
||||
|
||||
## Automated Testing
|
||||
|
||||
MFEM has several levels of automated testing running on GitHub, as well as on
|
||||
local Mac and Linux workstations, and Livermore Computing clusters at LLNL.
|
||||
|
||||
In addition, developers can set local git hooks to run some quick checks on
|
||||
commit or push, see the [README](config/githooks/README.md) in the `config/githooks`
|
||||
directory.
|
||||
|
||||
### Linux and Mac smoke tests
|
||||
We use GitHub Actions to drive the default tests on the `master` and `next`
|
||||
branches. See the `.github/workflows` files and the logs at
|
||||
[https://github.com/mfem/mfem/actions](https://github.com/mfem/mfem/actions).
|
||||
We use Travis CI to drive the default tests on the `master` and `next`
|
||||
branches. See the `.travis` file and the logs at
|
||||
[https://travis-ci.org/mfem/mfem](https://travis-ci.org/mfem/mfem).
|
||||
|
||||
Testing using GitHub Actions should be kept lightweight, as there is a time
|
||||
Testing using Travis CI should be kept lightweight, as there is a 50 minute time
|
||||
constraint on jobs. Two virtual machines are configured - Mac (OS X) and Linux.
|
||||
|
||||
- Tests on the `master` branch are triggered whenever a PR is issued on this branch.
|
||||
@@ -617,15 +499,11 @@ CMake is used to generate the MSVC Project files and drive the build. A release
|
||||
and debug build is performed with a simple run of `ex1` to verify the executable.
|
||||
|
||||
### Tests at LLNL
|
||||
At LLNL, we mirror the `master` and `next` branches internally (to `gh-master`
|
||||
and `gh-next`) and run longer nightly tests via cron. On the weekends, a more
|
||||
extensive test is run which extracts and executes all the different sample runs
|
||||
from each example.
|
||||
|
||||
- We mirror the `master` and `next` branches internally (to `gh-master` and
|
||||
`gh-next`) and run longer nightly tests via cron. On the weekends, a more
|
||||
extensive test is run which extracts and executes all the different sample
|
||||
runs from each example.
|
||||
|
||||
- We also mirror PRs on the LLNL GitLab instance. PR mirroring can only be
|
||||
triggered by _LLNL developers_, but test status is publicly available. Only
|
||||
_LLNL developers_ can access the detailed test report.
|
||||
|
||||
## Contact Information
|
||||
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
COPYRIGHT
|
||||
|
||||
The following copyright applies to each file in the MFEM distribution, unless
|
||||
otherwise stated in the file:
|
||||
|
||||
Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
Lawrence Livermore National Laboratory under Contract No. DE-AC52-07NA27344 with
|
||||
the DOE.
|
||||
|
||||
Neither the United States Government nor Lawrence Livermore National Security,
|
||||
LLC nor any of their employees, makes any warranty, express or implied, or
|
||||
assumes any liability or responsibility for the accuracy, completeness, or
|
||||
usefulness of any information, apparatus, product, or process disclosed, or
|
||||
represents that its use would not infringe privately-owned rights.
|
||||
|
||||
Also, reference herein to any specific commercial products, process, or services
|
||||
by trade name, trademark, manufacturer or otherwise does not necessarily
|
||||
constitute or imply its endorsement, recommendation, or favoring by the United
|
||||
States Government or Lawrence Livermore National Security, LLC. The views and
|
||||
opinions of authors expressed herein do not necessarily state or reflect those
|
||||
of the United States Government or Lawrence Livermore National Security, LLC,
|
||||
and shall not be used for advertising or product endorsement purposes.
|
||||
|
||||
|
||||
LICENSE
|
||||
|
||||
MFEM is free software; you can redistribute it and/or modify it under the terms
|
||||
of the GNU Lesser General Public License (as published by the Free Software
|
||||
Foundation) version 2.1 dated February 1999, with the following EXCEPTIONS:
|
||||
|
||||
1. Subclasses of MFEM classes do not constitute a derivative work.
|
||||
|
||||
2. Static linking of applications to the MFEM library does not constitute a
|
||||
derivative work and does not require the author to provide source code for
|
||||
the application, use the shared MFEM library, or link their applications
|
||||
against a user-supplied version of MFEM.
|
||||
|
||||
MFEM is distributed in the hope that it will be useful, but WITHOUT ANY
|
||||
WARRANTY; without even the IMPLIED WARRANTY OF MERCHANTABILITY or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. See the terms and conditions of the GNU Lesser General
|
||||
Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Lesser General Public License along
|
||||
with this library (file LICENSE); if not, write to the Free Software Foundation,
|
||||
Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
|
||||
|
||||
CONTACT INFORMATION
|
||||
|
||||
The software is released under LLNL-CODE-443211. Please see
|
||||
|
||||
http://mfem.org
|
||||
|
||||
for more information and source code availability.
|
||||
@@ -5,7 +5,7 @@
|
||||
| | | | | || _|| __/| | | | | |
|
||||
|_| |_| |_||_| \___||_| |_| |_|
|
||||
|
||||
https://mfem.org
|
||||
http://mfem.org
|
||||
|
||||
The MFEM library has a serial and an MPI-based parallel version, which largely
|
||||
share the same code base. The only prerequisite for building the serial version
|
||||
@@ -13,42 +13,14 @@ of MFEM is a (modern) C++ compiler, such as g++. The parallel version of MFEM
|
||||
requires an MPI C++ compiler, as well as the following external libraries:
|
||||
|
||||
- hypre (a library of high-performance preconditioners)
|
||||
https://github.com/hypre-space/hypre
|
||||
http://www.llnl.gov/CASC/hypre
|
||||
|
||||
- METIS (a family of multilevel partitioning algorithms)
|
||||
http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
|
||||
|
||||
The hypre dependency can be downloaded as a tarball from GitHub or from the
|
||||
project webpage https://www.llnl.gov/casc/hypre. For example, the 2.20.0 release
|
||||
of hypre is available at
|
||||
|
||||
https://github.com/hypre-space/hypre/archive/v2.20.0.tar.gz
|
||||
|
||||
The METIS dependency can be disabled but that is not generally recommended, see
|
||||
the option MFEM_USE_METIS.
|
||||
|
||||
MFEM also includes support for devices such as GPUs, and programming models such
|
||||
as CUDA, HIP, OCCA, OpenMP and RAJA.
|
||||
|
||||
- Starting with version 4.0, MFEM requires a C++11 compiler. We recommend using
|
||||
a newer compiler, e.g. GCC version 4.9 or higher.
|
||||
|
||||
- CUDA support requires an NVIDIA GPU and an installation of the CUDA Toolkit
|
||||
https://developer.nvidia.com/cuda-toolkit
|
||||
|
||||
- HIP support requires an AMD GPU and an installation of the ROCm software stack
|
||||
https://rocm.github.io/ROCmInstall.html#installing-from-amd-rocm-repositories
|
||||
|
||||
- OCCA support requires the OCCA library
|
||||
https://libocca.org
|
||||
|
||||
- OpenMP support requires a compiler implementing the OpenMP API
|
||||
https://www.openmp.org
|
||||
|
||||
- RAJA support requires installation of the RAJA performance portability layer
|
||||
with (optionally) support for CUDA and OpenMP
|
||||
https://github.com/LLNL/RAJA
|
||||
|
||||
The library supports two build systems: one based on GNU make, and a second one
|
||||
based on CMake. Both build systems are described below. Some hints for building
|
||||
without GNU make or CMake can be found at the end of this file.
|
||||
@@ -58,12 +30,11 @@ following package managers:
|
||||
|
||||
- Spack, https://github.com/spack/spack
|
||||
- OpenHPC, http://openhpc.community
|
||||
- Conda-forge, https://conda-forge.org (pre-built binaries linked with OpenMPI/MPICH, hypre, and METIS)
|
||||
- Homebrew/Science, https://github.com/Homebrew/homebrew-science (deprecated)
|
||||
- Homebrew/Science, https://github.com/Homebrew/homebrew-science
|
||||
|
||||
We also recommend downloading and building the MFEM-based GLVis visualization
|
||||
tool which can be used to visualize the meshes and solution in MFEM's examples
|
||||
and miniapps. See https://glvis.org and https://mfem.org/building.
|
||||
and miniapps. See http://glvis.org and http://mfem.org/building.
|
||||
|
||||
Quick start with GNU make
|
||||
=========================
|
||||
@@ -71,19 +42,11 @@ Serial build:
|
||||
make serial -j 4
|
||||
|
||||
Parallel build:
|
||||
(download hypre and METIS 4 from above URLs)
|
||||
(download hypre 2.10.0b and METIS 4 from above URLs)
|
||||
(build METIS 4 in ../metis-4.0 relative to mfem/)
|
||||
(build hypre in ../hypre relative to mfem/)
|
||||
(build hypre 2.10.0b in ../hypre-2.10.0b relative to mfem/)
|
||||
make parallel -j 4
|
||||
|
||||
CUDA build:
|
||||
make cuda -j 4
|
||||
(build for a specific compute capability: 'make cuda -j 4 CUDA_ARCH=sm_30')
|
||||
|
||||
HIP build:
|
||||
make hip -j 4
|
||||
(build for a specific AMD GPU chip: 'make hip -j 4 HIP_ARCH=gfx900')
|
||||
|
||||
Example codes (serial/parallel, depending on the build):
|
||||
cd examples
|
||||
make -j 4
|
||||
@@ -94,6 +57,7 @@ Build everything (library, examples and miniapps) with current configuration:
|
||||
Quick-check the build by running Example 1/1p (optional):
|
||||
make check
|
||||
|
||||
|
||||
Quick start with CMake
|
||||
======================
|
||||
Serial build:
|
||||
@@ -102,19 +66,13 @@ Serial build:
|
||||
make -j 4 (assuming "UNIX Makefiles" generator)
|
||||
|
||||
Parallel build:
|
||||
(download hypre and METIS 4 from above URLs)
|
||||
(download hypre 2.10.0b and METIS 4 from above URLs)
|
||||
(build METIS 4 in ../metis-4.0 relative to mfem/)
|
||||
(build hypre in ../hypre relative to mfem/)
|
||||
(build hypre 2.10.0b in ../hypre-2.10.0b relative to mfem/)
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES
|
||||
make -j 4
|
||||
|
||||
CUDA build:
|
||||
(this build requires CMake 3.8 or newer)
|
||||
mkdir <mfem-build-dir> ; cd <mfem-build-dir>
|
||||
cmake <mfem-source-dir> -DMFEM_USE_CUDA=YES
|
||||
make -j 4
|
||||
|
||||
Example codes (serial/parallel, depending on the build):
|
||||
make examples -j 4
|
||||
|
||||
@@ -170,18 +128,10 @@ Note that re-configuration is only needed to change the currently configured
|
||||
options. Several shortcut targets combining (re-)configuration and compilation
|
||||
are also defined:
|
||||
|
||||
make serial -> Builds serial optimized version of the library
|
||||
make parallel -> Builds parallel optimized version of the library
|
||||
make debug -> Builds serial debug version of the library
|
||||
make pdebug -> Builds parallel debug version of the library
|
||||
make cuda -> Builds serial cuda optimized version of the library
|
||||
make pcuda -> Builds parallel cuda optimized version of the library
|
||||
make cudebug -> Builds serial cuda debug version of the library
|
||||
make pcudebug -> Builds parallel cuda debug version of the library
|
||||
make hip -> Builds serial hip optimized version of the library
|
||||
make phip -> Builds parallel hip optimized version of the library
|
||||
make hipdebug -> Builds serial hip debug version of the library
|
||||
make phipdebug -> Builds parallel hip debug version of the library
|
||||
make serial -> Builds serial optimized version of the library
|
||||
make parallel -> Builds parallel optimized version of the library
|
||||
make debug -> Builds serial debug version of the library
|
||||
make pdebug -> Builds parallel debug version of the library
|
||||
|
||||
Note that any of the above shortcuts accept configuration options, either at the
|
||||
command line or through a user configuration file.
|
||||
@@ -243,9 +193,8 @@ Configuration options (GNU make)
|
||||
See the configuration file config/defaults.mk for the default settings.
|
||||
|
||||
Compilers:
|
||||
CXX - C++ compiler, serial build
|
||||
MPICXX - MPI C++ compiler, parallel build
|
||||
CUDA_CXX - The CUDA compiler, 'nvcc'
|
||||
CXX - C++ compiler, serial build
|
||||
MPICXX - MPI C++ compiler, parallel build
|
||||
|
||||
Compiler options:
|
||||
OPTIM_FLAGS - Options for optimized build
|
||||
@@ -281,7 +230,7 @@ MFEM_DEBUG = YES/NO
|
||||
and consistency checks that may simplify bug-hunting.
|
||||
|
||||
MFEM_USE_EXCEPTIONS = YES/NO
|
||||
Enable the use of exceptions. In particular, modifies the default behavior
|
||||
Enable the use of exceptions. In particular, modifies the default bahavior
|
||||
when errors are encountered: throw an exception, instead of aborting.
|
||||
|
||||
MFEM_USE_LIBUNWIND = YES/NO
|
||||
@@ -301,12 +250,8 @@ MFEM_THREAD_SAFE = YES/NO
|
||||
Use thread-safe implementation for some classes/methods. This comes at the
|
||||
cost of extra memory allocation and de-allocation.
|
||||
|
||||
MFEM_USE_LEGACY_OPENMP = YES/NO
|
||||
Enable (basic) experimental OpenMP support. Requires MFEM_THREAD_SAFE.
|
||||
This option is deprecated.
|
||||
|
||||
MFEM_USE_OPENMP = YES/NO
|
||||
Enable the OpenMP backend.
|
||||
Enable (basic) experimental OpenMP support. Requires MFEM_THREAD_SAFE.
|
||||
|
||||
MFEM_USE_MEMALLOC = YES/NO
|
||||
Internal MFEM option: enable batch allocation for some small objects.
|
||||
@@ -345,32 +290,12 @@ MFEM_USE_SUPERLU = YES/NO
|
||||
SuperLURowLocMatrix a distributed CSR matrix class needed by SuperLU. When
|
||||
enabled, this option uses the SUPERLU_* library options, see below.
|
||||
|
||||
MFEM_USE_SUPERLU5 = YES/NO
|
||||
If SuperLU functionality is enabled, use the older 5.1.0 version rather than
|
||||
the more recent 6+ versions.
|
||||
|
||||
MFEM_USE_MUMPS = YES/NO
|
||||
Enable MFEM functionality based on the MUMPS library. Currently, this
|
||||
option adds the class MUMPSSolver (a parallel sparse direct solver).
|
||||
When enabled, this option uses the MUMPS_* library options, see below.
|
||||
|
||||
MFEM_USE_STRUMPACK = YES/NO
|
||||
Enable MFEM functionality based on the STRUMPACK sparse direct solver and
|
||||
preconditioner through the STRUMPACKSolver and STRUMPACKRowLocMatrix
|
||||
classes. When enabled, this option uses the STRUMPACK_* library options, see
|
||||
below.
|
||||
|
||||
MFEM_USE_GINKGO = YES/NO
|
||||
Enable MFEM functionality based on the Ginkgo library, which provides
|
||||
iterative linear solvers and preconditioners with OpenMP, CUDA backends, see
|
||||
https://github.com/ginkgo-project/ginkgo. When enabled, the user can use
|
||||
Ginkgo's solvers and preconditioners as shown in examples/ginkgo/.
|
||||
|
||||
MFEM_USE_AMGX = YES/NO
|
||||
Enable MFEM functionality based on the AmgX multigrid library from NVIDIA.
|
||||
Allows the user to use SparseMatrices and HypreParMatrices to solve linear
|
||||
systems with the routines from the AmgX library.
|
||||
|
||||
MFEM_USE_GNUTLS = YES/NO
|
||||
Enable secure socket support in class socketstream, using the auxiliary
|
||||
GnuTLS_* classes, based on the GnuTLS library. This option may be useful in
|
||||
@@ -398,10 +323,6 @@ MFEM_USE_PETSC = YES/NO
|
||||
and other features based on the PETSc package. When enabled, this option uses
|
||||
the PETSC_* library options, see below.
|
||||
|
||||
MFEM_USE_SLEPC = YES/NO
|
||||
Enable MFEM eigensolvers based on the SLEPc package. When enabled, this
|
||||
option uses the SLEPC_* library options, see below.
|
||||
|
||||
MFEM_USE_MPFR = YES/NO
|
||||
MPFR is a library for multiple-precision floating-point computations. This
|
||||
option enables the use of MPFR in MFEM, e.g. for precise computation of 1D
|
||||
@@ -409,17 +330,11 @@ MFEM_USE_MPFR = YES/NO
|
||||
see below.
|
||||
|
||||
MFEM_USE_SIDRE = YES/NO
|
||||
Sidre is a component of LLNL's axom project, https://github.com/LLNL/axom,
|
||||
that provides an HDF5-based file format for visualization or restart
|
||||
capability following the Conduit (https://github.com/LLNL/conduit) mesh
|
||||
blueprint specification. When enabled, this option requires installation of
|
||||
HDF5 (see also MFEM_USE_NETCDF), Conduit and LLNL's axom project.
|
||||
|
||||
MFEM_USE_SIMD = YES/NO
|
||||
Enables the high performance templated classes to use architecture dependent
|
||||
SIMD intrinsics instead of the generic implementation of class AutoSIMD in
|
||||
linalg/simd/auto.hpp. This option should be combined with suitable
|
||||
compiler options, such as -march=native, to enable optimal vectorization.
|
||||
Sidre is a component of LLNL's axom project, http://goo.gl/cZyJdn, that
|
||||
provides an HDF5-based file format for visualization or restart capability
|
||||
following the Conduit (https://github.com/LLNL/conduit) mesh blueprint
|
||||
specification. When enabled, this option requires installation of HDF5 (see
|
||||
also MFEM_USE_NETCDF), Conduit and LLNL's axom project.
|
||||
|
||||
MFEM_USE_CONDUIT = YES/NO
|
||||
Enables support for converting MFEM Mesh and Grid Function objects to and
|
||||
@@ -428,15 +343,10 @@ MFEM_USE_CONDUIT = YES/NO
|
||||
an installation of Conduit. If Conduit was built with HDF5 support, it also
|
||||
requires an installation of HDF5 (see also MFEM_USE_NETCDF).
|
||||
|
||||
MFEM_USE_ADIOS2 = YES/NO
|
||||
Enables support for ADIOS2, version 2 of the adaptable input output system
|
||||
for scientific data management. In MFEM, ADIOS2 provides parallel I/O with
|
||||
ParaView visualization.
|
||||
|
||||
MFEM_USE_ZLIB = YES/NO
|
||||
MFEM_USE_GZSTREAM = YES/NO
|
||||
Enables use of on-the-fly gzip compressed streams. With this feature enabled
|
||||
(YES), MFEM can compress its output files on-the-fly. In addition, it can
|
||||
read back files compressed with zlib (or any compression utility capable
|
||||
read back files compressed with gzstream (or any compression utility capable
|
||||
of creating a gzip-compatible output such as gzip).
|
||||
MFEM will write compressed files if the mode argument in the constructor
|
||||
includes a 'z' character. With this feature disabled (NO), MFEM will not be
|
||||
@@ -451,70 +361,6 @@ MFEM_USE_PUMI = YES/NO
|
||||
data management system that is capable of handling general non-manifold
|
||||
models and effectively supports automated adaptive analysis. PUMI enables
|
||||
support for parallel unstructured mesh modifications in MFEM.
|
||||
The develop branch of PUMI repository (https://github.com/SCOREC/core)
|
||||
should be used for most updated features.
|
||||
|
||||
MFEM_USE_UMPIRE = YES/NO
|
||||
Enables support for Umpire, a resource management library that allows the
|
||||
discovery, provision, and management of memory on machines with multiple
|
||||
memory devices like NUMA and GPUs.
|
||||
|
||||
MFEM_USE_HIOP = YES/NO
|
||||
Enable the usage of HiOp (https://github.com/LLNL/hiop) in MFEM. HiOp is an
|
||||
HPC solver for nonlinear optimization problems.
|
||||
|
||||
MFEM_USE_CUDA = YES/NO
|
||||
Enables support for CUDA devices in MFEM. CUDA is a parallel computing
|
||||
platform and programming model for general computing on graphical processing
|
||||
units (GPUs). The variable CUDA_ARCH is used to specify the CUDA compute
|
||||
capability used during compilation (by default, CUDA_ARCH=sm_60). When
|
||||
enabled, this option uses the CUDA_* build options, see below.
|
||||
|
||||
MFEM_USE_HIP = YES/NO
|
||||
Enables support for AMD devices in MFEM. HIP is a heterogeneous-compute
|
||||
interface for portability developed by AMD that can target both AMD and
|
||||
NVIDIA GPUs. The variable HIP_ARCH is used to specify the AMD GPU processor
|
||||
used during compilation (by default, HIP_ARCH=gfx900). When enabled, this
|
||||
option uses the HIP_* build options, see below.
|
||||
|
||||
MFEM_USE_RAJA = YES/NO
|
||||
Enable support for the RAJA performance portability layer in MFEM. RAJA
|
||||
provides a portable abstraction for loops, supporting different programming
|
||||
model backends. When using RAJA built with CUDA support, CUDA support must be
|
||||
also enabled in MFEM, i.e. MFEM_USE_CUDA=YES must be set.
|
||||
|
||||
MFEM_USE_OCCA = YES/NO
|
||||
Enables support for the OCCA library in MFEM. OCCA is an open-source library
|
||||
which aims to make it easy to program different types of devices (e.g. CPU,
|
||||
GPU, FPGA) by providing an unified API for interacting with JIT-compiled
|
||||
backends. In order to use the OCCA CUDA backend, CUDA support must be enabled
|
||||
in MFEM as well, i.e. MFEM_USE_CUDA=YES must be set.
|
||||
|
||||
MFEM_USE_GSLIB = YES/NO
|
||||
Enables MFEM functionality based on the GSLIB library, and specifically its
|
||||
FindPoints component, which provides a robust algorithms to evaluate finite
|
||||
element functions in a collection of points in physical space. When enabled,
|
||||
the user can use the GSLIB-FindPoints methods as shown in miniapps/gslib.
|
||||
|
||||
MFEM_USE_CEED = YES/NO
|
||||
Enables support for the libCEED library in MFEM. libCEED is a portable
|
||||
library for performant high-order operator evaluation developed by the Center
|
||||
for Efficient Exascale Discretizations in the Exascale Computing Project.
|
||||
|
||||
MFEM_USE_MKL_CPARDISO = YES/NO
|
||||
Enables the interface to MKL CPardiso: the Intel MKL Parallel Direct Sparse
|
||||
Solver for Clusters. Make sure to set the correct values for MKL_MPI_WRAPPER
|
||||
and MKL_LIBRARY_SUBDIR as shown in defaults.mk. If you configure MFEM with
|
||||
MFEM_USE_LAPACK=YES, verify that the MKL LAPACK libraries are used. The
|
||||
OpenMP capabilities are disabled at link time.
|
||||
|
||||
MFEM_USE_CALIPER = YES/NO
|
||||
Enables the interface to Caliper. Caliper is a library to integrate
|
||||
performance profiling capabilities into applications. To use Caliper,
|
||||
developers mark code regions of interest using either Caliper's annotation
|
||||
API or their equivalent in MFEM. Applications can then enable performance
|
||||
profiling at runtime with Caliper's configuration API. Alternatively, one
|
||||
can configure Caliper through environment variables or config files.
|
||||
|
||||
MFEM_BUILD_TAG = (any value)
|
||||
An optional tag to characterize the build. Exported to config/config.mk.
|
||||
@@ -537,17 +383,13 @@ directory and use the string @MFEM_DIR@, e.g. HYPRE_OPT = -I@MFEM_DIR@/../hypre.
|
||||
The specific libraries and their options are:
|
||||
|
||||
- HYPRE, required for the parallel build, i.e. when MFEM_USE_MPI = YES.
|
||||
See also the "Specific options for hypre" section at the end of this file.
|
||||
URL: https://github.com/hypre-space/hypre and https://www.llnl.gov/casc/hypre
|
||||
URL: http://www.llnl.gov/CASC/hypre
|
||||
Options: HYPRE_OPT, HYPRE_LIB.
|
||||
Versions: HYPRE >= 2.10.0b,
|
||||
HYPRE >= 2.20.0 for '--enable-mixedint' support.
|
||||
|
||||
- METIS, used when MFEM_USE_METIS = YES. If using METIS 5, set
|
||||
MFEM_USE_METIS_5 = YES (default is to use METIS 4).
|
||||
URL: http://glaros.dtc.umn.edu/gkhome/metis/metis/overview
|
||||
Options: METIS_OPT, METIS_LIB.
|
||||
Versions: METIS 4.0.3 or 5.1.0.
|
||||
|
||||
- LAPACK (optional), used when MFEM_USE_LAPACK = YES. Alternative, optimized
|
||||
implementations can also be used, e.g. the ATLAS project.
|
||||
@@ -555,8 +397,7 @@ The specific libraries and their options are:
|
||||
http://math-atlas.sourceforge.net (ATLAS)
|
||||
Options: LAPACK_OPT (currently not used/needed), LAPACK_LIB.
|
||||
|
||||
- OpenMP (optional), usually part of compiler, used when either MFEM_USE_OPENMP
|
||||
or MFEM_USE_LEGACY_OPENMP is set to YES.
|
||||
- OpenMP (optional), usually part of compiler, used when MFEM_USE_OPENMP = YES.
|
||||
Options: OPENMP_OPT, OPENMP_LIB.
|
||||
|
||||
- High-resolution POSIX clocks: when using MFEM_TIMER_TYPE = 2, it may be
|
||||
@@ -566,68 +407,39 @@ The specific libraries and their options are:
|
||||
- SUNDIALS (optional), used when MFEM_USE_SUNDIALS = YES.
|
||||
Beginning with MFEM v3.3, SUNDIALS v2.7.0 is supported.
|
||||
Beginning with MFEM v3.3.2, SUNDIALS v3.0.0 is also supported.
|
||||
Beginning with MFEM v4.1, only SUNDIALS v5.0.0+ is supported.
|
||||
When MFEM_USE_CUDA is enabled, only SUNDIALS v5.4.0+ is supported.
|
||||
If MFEM_USE_MPI is enabled, we expect that SUNDIALS is built with support for
|
||||
both MPI and hypre.
|
||||
If MFEM_USE_CUDA is enabled, we expect that SUNDIALS is built with support
|
||||
for CUDA.
|
||||
URL: http://computation.llnl.gov/projects/sundials/sundials-software
|
||||
Options: SUNDIALS_OPT, SUNDIALS_LIB.
|
||||
Versions: SUNDIALS >= 5.0.0, SUNDIALS >= 5.4.0 for CUDA support.
|
||||
|
||||
- Mesquite (optional), used when MFEM_USE_MESQUITE = YES.
|
||||
URL: http://trilinos.org/oldsite/packages/mesquite
|
||||
Options: MESQUITE_OPT, MESQUITE_LIB.
|
||||
The Mesquite support is deprecated and will be removed in the future.
|
||||
|
||||
- SuiteSparse (optional), used when MFEM_USE_SUITESPARSE = YES.
|
||||
URL: http://faculty.cse.tamu.edu/davis/suitesparse.html
|
||||
Options: SUITESPARSE_OPT, SUITESPARSE_LIB.
|
||||
Versions: SuiteSparse >= 4.5.4, older versions may work too.
|
||||
|
||||
- SuperLU_DIST (optional), used when MFEM_USE_SUPERLU = YES. Note that
|
||||
SuperLU_DIST requires ParMETIS, which includes METIS 5 in its distribution.
|
||||
Both ParMETIS and the included METIS 5 should be built and installed in the
|
||||
same location. If using SuperLU_Dist v5, set MFEM_USE_SUPERLU5=YES.
|
||||
same location.
|
||||
URL: http://crd-legacy.lbl.gov/~xiaoye/SuperLU
|
||||
Options: SUPERLU_OPT, SUPERLU_LIB.
|
||||
Versions: SuperLU_DIST >= 5.1.0.
|
||||
|
||||
- MUMPS (optional), used when MFEM_USE_MUMPS = YES. Note that MUMPS
|
||||
requires LAPACK, SCALAPACK and a reordering package such as PORD or METIS.
|
||||
URL: http://mumps.enseeiht.fr
|
||||
Options: MUMPS_OPT, MUMPS_LIB.
|
||||
Versions: MUMPS >= 5.1.1
|
||||
|
||||
- STRUMPACK (optional), used when MFEM_USE_STRUMPACK = YES. Note that STRUMPACK
|
||||
requires the PT-Scotch and Scalapack libraries as well as ParMETIS, which
|
||||
includes METIS 5 in its distribution. Starting with STRUMPACK v2.2.0, ParMETIS
|
||||
and PT-Scotch are optional dependencies.
|
||||
includes METIS 5 in its distribution.
|
||||
The support for STRUMPACK was added in MFEM v3.3.2 and it requires STRUMPACK
|
||||
2.0.0 or later.
|
||||
URL: http://portal.nersc.gov/project/sparse/strumpack
|
||||
Options: STRUMPACK_OPT, STRUMPACK_LIB.
|
||||
Versions: STRUMPACK >= 3.0.0.
|
||||
|
||||
- Ginkgo (optional), used when MFEM_USE_GINKGO = YES. Note that Ginkgo needs a
|
||||
C++ compiler that supports the C++-14 standard. For additional requirements
|
||||
and dependencies of specific modules, see the Ginkgo webpage below.
|
||||
URL: https://ginkgo-project.github.io
|
||||
Options: GINKGO_OPT, GINKGO_LIB, GINKGO_DIR, GINKGO_BUILD_TYPE (Release or Debug).
|
||||
Versions: Ginkgo >= 1.4.0.
|
||||
|
||||
- AmgX (optional), used when MFEM_USE_AMGX = YES.
|
||||
URL: https://github.com/NVIDIA/AMGX
|
||||
Options: AMGX_OPT, AMGX_LIB.
|
||||
Versions: AmgX >= 2.1, older versions may work too.
|
||||
|
||||
- GnuTLS (optional), used when MFEM_USE_GNUTLS = YES. On most Linux systems,
|
||||
GnuTLS is available as a development package, e.g. gnutls-devel. On Mac OS X,
|
||||
one can get the library through the Homebrew package manager (http://brew.sh).
|
||||
URL: http://gnutls.org
|
||||
Options: GNUTLS_OPT, GNUTLS_LIB.
|
||||
Versions: GnuTLS >= 2.12.0, older versions may work too.
|
||||
|
||||
- NetCDF (optional), used when MFEM_USE_NETCDF = YES, required for reading Cubit
|
||||
mesh files. Also requires installation of HDF5 and ZLIB, as explained at the
|
||||
@@ -635,7 +447,6 @@ The specific libraries and their options are:
|
||||
don't need the C++ or parallel versions.
|
||||
URL: www.unidata.ucar.edu/software/netcdf
|
||||
Options: NETCDF_OPT, NETCDF_LIB.
|
||||
Versions: NetCDF >= 4.4.0.
|
||||
|
||||
- PETSc (optional), used when MFEM_USE_PETSC = YES. Version 3.8 or higher of
|
||||
the PETSC dev branch is required. The MFEM and PETSc builds can share common
|
||||
@@ -647,96 +458,22 @@ The specific libraries and their options are:
|
||||
--with-shared-libraries=0
|
||||
URL: https://www.mcs.anl.gov/petsc
|
||||
Options: PETSC_OPT, PETSC_LIB.
|
||||
Versions: PETSc >= 3.8.0 (PETSc build without CUDA)
|
||||
PETSc >= 3.15.0 (PETSc built with CUDA)
|
||||
|
||||
- SLEPc (optional), used when MFEM_USE_SLEPC = YES. SLEPc depends on PETSc and
|
||||
uses some of the PETSc options when compiled.
|
||||
URL: https://slepc.upv.es/
|
||||
Options: SLEPC_OPT, SLEPC_LIB.
|
||||
Versions: SLEPc >= 3.8.0.
|
||||
|
||||
- Sidre (optional), part of LLNL's axom project, used when MFEM_USE_SIDRE = YES.
|
||||
Starting with MFEM v4.1, Axom version 0.3.1 or later is required.
|
||||
URL: https://github.com/LLNL/axom
|
||||
URL: http://goo.gl/cZyJdn (axom, to be released)
|
||||
https://github.com/LLNL/conduit (Conduit)
|
||||
https://support.hdfgroup.org/HDF5 (HDF5)
|
||||
Options: SIDRE_OPT, SIDRE_LIB.
|
||||
Versions: Axom >= 0.3.1.
|
||||
|
||||
- Conduit (optional), used when MFEM_USE_CONDUIT = YES. Conduit Mesh Blueprint
|
||||
- Conduit, used when MFEM_USE_CONDUIT = YES. Direct Conduit Mesh Blueprint
|
||||
support requires Conduit >= v0.3.1 and VisIt >= v2.13.1 to read the output.
|
||||
URL: https://github.com/LLNL/conduit (Conduit)
|
||||
https://support.hdfgroup.org/HDF5 (HDF5)
|
||||
Options: CONDUIT_OPT, CONDUIT_LIB.
|
||||
Versions: Conduit >= 0.3.1.
|
||||
|
||||
- ADIOS2 (optional) used when MFEM_USE_ADIOS2 = YES.
|
||||
URL: https://adios2.readthedocs.io/
|
||||
Versions: ADIOS >= 2.5.0.
|
||||
|
||||
- PUMI (optional), used when MFEM_USE_PUMI = YES.
|
||||
- PUMI, used when MFEM_USE_PUMI = YES.
|
||||
URL: https://scorec.rpi.edu/pumi
|
||||
https://github.com/SCOREC/core
|
||||
Options: PUMI_OPT, PUMI_LIB.
|
||||
Versions: PUMI == 2.2.3.
|
||||
|
||||
- HiOp (optional), used when MFEM_USE_HIOP = YES.
|
||||
URL: https://github.com/LLNL/hiop
|
||||
Options: HIOP_OPT, HIOP_LIB.
|
||||
Versions: HIOP >= 0.4.
|
||||
|
||||
- GSLIB (optional), used when MFEM_USE_GSLIB = YES. The gslib library must be
|
||||
built prior to the MFEM build, as follows: download gslib-1.0.7, untar it at
|
||||
the same level as MFEM and create a symbolic link: "ln -s gslib-1.0.7 gslib".
|
||||
Build gslib in parallel or in serial based on the desired MFEM build: "make
|
||||
clean; make CC=mpicc" or "make clean; make CC=gcc MPI=0". Build MFEM with
|
||||
MFEM_USE_GSLIB=YES.
|
||||
URL: https://github.com/gslib/gslib/archive/v1.0.7.tar.gz
|
||||
Options: GSLIB_OPT, GSLIB_LIB.
|
||||
Versions: GSLIB >= 1.0.7.
|
||||
|
||||
- MKL CPardiso (optional), used when MFEM_USE_MKL_CPARDISO = YES.
|
||||
URL: https://software.intel.com/content/www/us/en/develop/tools/math-kernel-library.html
|
||||
Options: MKL_CPARDISO_OPT, MKL_CPARDISO_LIB.
|
||||
Versions: Intel MKL >= 2020.
|
||||
|
||||
- CUDA (optional), used when MFEM_USE_CUDA = YES.
|
||||
URL: https://developer.nvidia.com/cuda-toolkit
|
||||
Options: CUDA_CXX, CUDA_ARCH, CUDA_OPT, CUDA_LIB.
|
||||
Versions: CUDA >= 10.1.168.
|
||||
|
||||
- HIP (optional), used when MFEM_USE_HIP = YES.
|
||||
URL: https://rocm.github.io/ROCmInstall.html
|
||||
Options: HIP_CXX, HIP_ARCH, HIP_OPT, HIP_LIB.
|
||||
|
||||
- OCCA (optional), used when MFEM_USE_OCCA = YES.
|
||||
URL: https://libocca.org
|
||||
Options: OCCA_DIR, OCCA_OPT, OCCA_LIB.
|
||||
Versions: OCCA >= 1.1.0.
|
||||
|
||||
- libCEED (optional), used when MFEM_USE_CEED = YES.
|
||||
URL: https://github.com/CEED/libCEED
|
||||
https://ceed.exascaleproject.org/libceed
|
||||
Options: CEED_DIR, CEED_OPT, CEED_LIB.
|
||||
Versions: libCEED >= 0.8.
|
||||
|
||||
- RAJA (optional), used when MFEM_USE_RAJA = YES.
|
||||
Beginning with MFEM v4.3, only RAJA v0.13.0+ is supported.
|
||||
URL: https://github.com/LLNL/RAJA
|
||||
Options: RAJA_DIR, RAJA_OPT, RAJA_LIB.
|
||||
Versions: RAJA >= 0.13.0.
|
||||
|
||||
- Caliper (optional), used when MFEM_USE_CALIPER = YES.
|
||||
URL: https://github.com/LLNL/Caliper
|
||||
Options: CALIPER_DIR
|
||||
Versions: CALIPER >= 2.5.0, older versions may work too.
|
||||
|
||||
- Umpire, used when MFEM_USE_UMPIRE = YES.
|
||||
Umpire requires camp when the Umpire version is >= 3.0.0.
|
||||
URL: https://github.com/LLNL/Umpire
|
||||
Options: UMPIRE_DIR, UMPIRE_OPT, UMPIRE_LIB.
|
||||
Versions: Umpire >= 2.0.0.
|
||||
|
||||
- MPFR (optional), used when MFEM_USE_MPFR = YES.
|
||||
URL: http://mpfr.org, it depends on the GMP library: https://gmplib.org
|
||||
@@ -748,11 +485,12 @@ The specific libraries and their options are:
|
||||
URL: http://www.nongnu.org/libunwind
|
||||
Options: LIBUNWIND_OPT, LIBUNWIND_LIB.
|
||||
|
||||
- ZLIB (optional), used when MFEM_USE_ZLIB = YES, or when MFEM_USE_NETCDF =
|
||||
- ZLIB (optional), used when MFEM_USE_GZSTREAM = YES, or when MFEM_USE_NETCDF =
|
||||
YES (in the default settings for NETCDF_OPT and NETCDF_LIB).
|
||||
URL: https://zlib.net
|
||||
Options: ZLIB_OPT, ZLIB_LIB.
|
||||
|
||||
|
||||
Building with CMake
|
||||
===================
|
||||
The MFEM build system consists of two steps: configuration and compilation.
|
||||
@@ -841,8 +579,6 @@ Configuration variables (CMake)
|
||||
===============================
|
||||
See the configuration file config/defaults.cmake for the default settings.
|
||||
|
||||
Note: the option MFEM_USE_CUDA requires CMake version 3.8 or newer!
|
||||
|
||||
Non-standard CMake variables for compilers:
|
||||
CXX - If set, overwrite the auto-detected C++ compiler, serial build
|
||||
MPICXX - If set, overwrite the auto-detected MPI C++ compiler, parallel build
|
||||
@@ -860,30 +596,18 @@ MFEM_USE_METIS - Set to ${MFEM_USE_MPI}, can be overwritten.
|
||||
MFEM_USE_LIBUNWIND
|
||||
MFEM_USE_LAPACK
|
||||
MFEM_THREAD_SAFE
|
||||
MFEM_USE_LEGACY_OPENMP
|
||||
MFEM_USE_OPENMP
|
||||
MFEM_USE_MEMALLOC
|
||||
MFEM_TIMER_TYPE - Set automatically, can be overwritten.
|
||||
MFEM_USE_MESQUITE
|
||||
MFEM_USE_SUITESPARSE
|
||||
MFEM_USE_SUPERLU
|
||||
MFEM_USE_MUMPS
|
||||
MFEM_USE_STRUMPACK
|
||||
MFEM_USE_GINKGO
|
||||
MFEM_USE_AMGX
|
||||
MFEM_USE_GNUTLS
|
||||
MFEM_USE_NETCDF
|
||||
MFEM_USE_MPFR
|
||||
MFEM_USE_ZLIB
|
||||
MFEM_USE_GZSTREAM
|
||||
MFEM_USE_PUMI
|
||||
MFEM_USE_HIOP
|
||||
MFEM_USE_CUDA
|
||||
MFEM_USE_OCCA
|
||||
MFEM_USE_CEED
|
||||
MFEM_USE_RAJA
|
||||
MFEM_USE_UMPIRE
|
||||
MFEM_USE_SIDRE
|
||||
MFEM_USE_CALIPER
|
||||
|
||||
The following options are CMake specific:
|
||||
|
||||
@@ -920,24 +644,16 @@ The CMake build system adds auto-detection for the following packages/libraries:
|
||||
|
||||
- HYPRE
|
||||
- METIS - The option MFEM_USE_METIS_5 is auto-detected.
|
||||
- ParMETIS
|
||||
- MESQUITE
|
||||
- SuiteSparse
|
||||
- SuperLUDist, STRUMPACK
|
||||
- Ginkgo
|
||||
- AMGX
|
||||
- ParMETIS
|
||||
- GNUTLS - Extends the built-in CMake support, to search GNUTLS_DIR as well.
|
||||
- NETCDF
|
||||
- MPFR
|
||||
- LIBUNWIND
|
||||
- POSIXCLOCKS
|
||||
- PUMI
|
||||
- HIOP
|
||||
- OCCA
|
||||
- RAJA
|
||||
- UMPIRE
|
||||
- AXOM - Used when MFEM_USE_SIDRE is enabled
|
||||
- CALIPER
|
||||
|
||||
The following built-in CMake packages are also used:
|
||||
|
||||
@@ -973,19 +689,3 @@ MFEM_MPIEXEC = mpirun # default
|
||||
MFEM_MPIEXEC_NP = -np # default
|
||||
MFEM_MPIEXEC = srun # example for platforms using SLURM
|
||||
MFEM_MPIEXEC_NP = -n # example for platforms using SLURM
|
||||
|
||||
|
||||
Specific options for hypre
|
||||
==========================
|
||||
The hypre library has multiple options to define local and global index storage
|
||||
sizes. By default, all indices are stored as an architecture aware integer. For
|
||||
most platforms, this will be 32-bit. This limits the maximum number of global
|
||||
degrees of freedom in a vector or matrix to about 2 billion. In order to solve
|
||||
larger problems, there are two options:
|
||||
|
||||
1. Building hypre with '--enable-bigint' defines the local and global indices to
|
||||
be 64-bit. This is convenient, but requires more memory than necessary.
|
||||
|
||||
2. Building hypre with '--enable-mixedint' defines the local indiced to be
|
||||
32-bit, while using a 64-bit storage for global indices. This option is
|
||||
currently tested only in ex1p, and may not work in more general settings.
|
||||
|
||||
@@ -1,29 +1,504 @@
|
||||
BSD 3-Clause License
|
||||
GNU LESSER GENERAL PUBLIC LICENSE
|
||||
Version 2.1, February 1999
|
||||
|
||||
Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC
|
||||
All rights reserved.
|
||||
Copyright (C) 1991, 1999 Free Software Foundation, Inc.
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
Everyone is permitted to copy and distribute verbatim copies
|
||||
of this license document, but changing it is not allowed.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions are met:
|
||||
[This is the first released version of the Lesser GPL. It also counts
|
||||
as the successor of the GNU Library Public License, version 2, hence
|
||||
the version number 2.1.]
|
||||
|
||||
* Redistributions of source code must retain the above copyright notice, this
|
||||
list of conditions and the following disclaimer.
|
||||
Preamble
|
||||
|
||||
* Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimer in the documentation
|
||||
and/or other materials provided with the distribution.
|
||||
The licenses for most software are designed to take away your
|
||||
freedom to share and change it. By contrast, the GNU General Public
|
||||
Licenses are intended to guarantee your freedom to share and change
|
||||
free software--to make sure the software is free for all its users.
|
||||
|
||||
This license, the Lesser General Public License, applies to some
|
||||
specially designated software packages--typically libraries--of the
|
||||
Free Software Foundation and other authors who decide to use it. You
|
||||
can use it too, but we suggest you first think carefully about whether
|
||||
this license or the ordinary General Public License is the better
|
||||
strategy to use in any particular case, based on the explanations below.
|
||||
|
||||
When we speak of free software, we are referring to freedom of use,
|
||||
not price. Our General Public Licenses are designed to make sure that
|
||||
you have the freedom to distribute copies of free software (and charge
|
||||
for this service if you wish); that you receive source code or can get
|
||||
it if you want it; that you can change the software and use pieces of
|
||||
it in new free programs; and that you are informed that you can do
|
||||
these things.
|
||||
|
||||
To protect your rights, we need to make restrictions that forbid
|
||||
distributors to deny you these rights or to ask you to surrender these
|
||||
rights. These restrictions translate to certain responsibilities for
|
||||
you if you distribute copies of the library or if you modify it.
|
||||
|
||||
For example, if you distribute copies of the library, whether gratis
|
||||
or for a fee, you must give the recipients all the rights that we gave
|
||||
you. You must make sure that they, too, receive or can get the source
|
||||
code. If you link other code with the library, you must provide
|
||||
complete object files to the recipients, so that they can relink them
|
||||
with the library after making changes to the library and recompiling
|
||||
it. And you must show them these terms so they know their rights.
|
||||
|
||||
We protect your rights with a two-step method: (1) we copyright the
|
||||
library, and (2) we offer you this license, which gives you legal
|
||||
permission to copy, distribute and/or modify the library.
|
||||
|
||||
To protect each distributor, we want to make it very clear that
|
||||
there is no warranty for the free library. Also, if the library is
|
||||
modified by someone else and passed on, the recipients should know
|
||||
that what they have is not the original version, so that the original
|
||||
author's reputation will not be affected by problems that might be
|
||||
introduced by others.
|
||||
|
||||
Finally, software patents pose a constant threat to the existence of
|
||||
any free program. We wish to make sure that a company cannot
|
||||
effectively restrict the users of a free program by obtaining a
|
||||
restrictive license from a patent holder. Therefore, we insist that
|
||||
any patent license obtained for a version of the library must be
|
||||
consistent with the full freedom of use specified in this license.
|
||||
|
||||
Most GNU software, including some libraries, is covered by the
|
||||
ordinary GNU General Public License. This license, the GNU Lesser
|
||||
General Public License, applies to certain designated libraries, and
|
||||
is quite different from the ordinary General Public License. We use
|
||||
this license for certain libraries in order to permit linking those
|
||||
libraries into non-free programs.
|
||||
|
||||
When a program is linked with a library, whether statically or using
|
||||
a shared library, the combination of the two is legally speaking a
|
||||
combined work, a derivative of the original library. The ordinary
|
||||
General Public License therefore permits such linking only if the
|
||||
entire combination fits its criteria of freedom. The Lesser General
|
||||
Public License permits more lax criteria for linking other code with
|
||||
the library.
|
||||
|
||||
We call this license the "Lesser" General Public License because it
|
||||
does Less to protect the user's freedom than the ordinary General
|
||||
Public License. It also provides other free software developers Less
|
||||
of an advantage over competing non-free programs. These disadvantages
|
||||
are the reason we use the ordinary General Public License for many
|
||||
libraries. However, the Lesser license provides advantages in certain
|
||||
special circumstances.
|
||||
|
||||
For example, on rare occasions, there may be a special need to
|
||||
encourage the widest possible use of a certain library, so that it becomes
|
||||
a de-facto standard. To achieve this, non-free programs must be
|
||||
allowed to use the library. A more frequent case is that a free
|
||||
library does the same job as widely used non-free libraries. In this
|
||||
case, there is little to gain by limiting the free library to free
|
||||
software only, so we use the Lesser General Public License.
|
||||
|
||||
In other cases, permission to use a particular library in non-free
|
||||
programs enables a greater number of people to use a large body of
|
||||
free software. For example, permission to use the GNU C Library in
|
||||
non-free programs enables many more people to use the whole GNU
|
||||
operating system, as well as its variant, the GNU/Linux operating
|
||||
system.
|
||||
|
||||
Although the Lesser General Public License is Less protective of the
|
||||
users' freedom, it does ensure that the user of a program that is
|
||||
linked with the Library has the freedom and the wherewithal to run
|
||||
that program using a modified version of the Library.
|
||||
|
||||
The precise terms and conditions for copying, distribution and
|
||||
modification follow. Pay close attention to the difference between a
|
||||
"work based on the library" and a "work that uses the library". The
|
||||
former contains code derived from the library, whereas the latter must
|
||||
be combined with the library in order to run.
|
||||
|
||||
GNU LESSER GENERAL PUBLIC LICENSE
|
||||
TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION
|
||||
|
||||
0. This License Agreement applies to any software library or other
|
||||
program which contains a notice placed by the copyright holder or
|
||||
other authorized party saying it may be distributed under the terms of
|
||||
this Lesser General Public License (also called "this License").
|
||||
Each licensee is addressed as "you".
|
||||
|
||||
A "library" means a collection of software functions and/or data
|
||||
prepared so as to be conveniently linked with application programs
|
||||
(which use some of those functions and data) to form executables.
|
||||
|
||||
The "Library", below, refers to any such software library or work
|
||||
which has been distributed under these terms. A "work based on the
|
||||
Library" means either the Library or any derivative work under
|
||||
copyright law: that is to say, a work containing the Library or a
|
||||
portion of it, either verbatim or with modifications and/or translated
|
||||
straightforwardly into another language. (Hereinafter, translation is
|
||||
included without limitation in the term "modification".)
|
||||
|
||||
"Source code" for a work means the preferred form of the work for
|
||||
making modifications to it. For a library, complete source code means
|
||||
all the source code for all modules it contains, plus any associated
|
||||
interface definition files, plus the scripts used to control compilation
|
||||
and installation of the library.
|
||||
|
||||
Activities other than copying, distribution and modification are not
|
||||
covered by this License; they are outside its scope. The act of
|
||||
running a program using the Library is not restricted, and output from
|
||||
such a program is covered only if its contents constitute a work based
|
||||
on the Library (independent of the use of the Library in a tool for
|
||||
writing it). Whether that is true depends on what the Library does
|
||||
and what the program that uses the Library does.
|
||||
|
||||
1. You may copy and distribute verbatim copies of the Library's
|
||||
complete source code as you receive it, in any medium, provided that
|
||||
you conspicuously and appropriately publish on each copy an
|
||||
appropriate copyright notice and disclaimer of warranty; keep intact
|
||||
all the notices that refer to this License and to the absence of any
|
||||
warranty; and distribute a copy of this License along with the
|
||||
Library.
|
||||
|
||||
You may charge a fee for the physical act of transferring a copy,
|
||||
and you may at your option offer warranty protection in exchange for a
|
||||
fee.
|
||||
|
||||
2. You may modify your copy or copies of the Library or any portion
|
||||
of it, thus forming a work based on the Library, and copy and
|
||||
distribute such modifications or work under the terms of Section 1
|
||||
above, provided that you also meet all of these conditions:
|
||||
|
||||
a) The modified work must itself be a software library.
|
||||
|
||||
b) You must cause the files modified to carry prominent notices
|
||||
stating that you changed the files and the date of any change.
|
||||
|
||||
c) You must cause the whole of the work to be licensed at no
|
||||
charge to all third parties under the terms of this License.
|
||||
|
||||
d) If a facility in the modified Library refers to a function or a
|
||||
table of data to be supplied by an application program that uses
|
||||
the facility, other than as an argument passed when the facility
|
||||
is invoked, then you must make a good faith effort to ensure that,
|
||||
in the event an application does not supply such function or
|
||||
table, the facility still operates, and performs whatever part of
|
||||
its purpose remains meaningful.
|
||||
|
||||
(For example, a function in a library to compute square roots has
|
||||
a purpose that is entirely well-defined independent of the
|
||||
application. Therefore, Subsection 2d requires that any
|
||||
application-supplied function or table used by this function must
|
||||
be optional: if the application does not supply it, the square
|
||||
root function must still compute square roots.)
|
||||
|
||||
These requirements apply to the modified work as a whole. If
|
||||
identifiable sections of that work are not derived from the Library,
|
||||
and can be reasonably considered independent and separate works in
|
||||
themselves, then this License, and its terms, do not apply to those
|
||||
sections when you distribute them as separate works. But when you
|
||||
distribute the same sections as part of a whole which is a work based
|
||||
on the Library, the distribution of the whole must be on the terms of
|
||||
this License, whose permissions for other licensees extend to the
|
||||
entire whole, and thus to each and every part regardless of who wrote
|
||||
it.
|
||||
|
||||
Thus, it is not the intent of this section to claim rights or contest
|
||||
your rights to work written entirely by you; rather, the intent is to
|
||||
exercise the right to control the distribution of derivative or
|
||||
collective works based on the Library.
|
||||
|
||||
In addition, mere aggregation of another work not based on the Library
|
||||
with the Library (or with a work based on the Library) on a volume of
|
||||
a storage or distribution medium does not bring the other work under
|
||||
the scope of this License.
|
||||
|
||||
3. You may opt to apply the terms of the ordinary GNU General Public
|
||||
License instead of this License to a given copy of the Library. To do
|
||||
this, you must alter all the notices that refer to this License, so
|
||||
that they refer to the ordinary GNU General Public License, version 2,
|
||||
instead of to this License. (If a newer version than version 2 of the
|
||||
ordinary GNU General Public License has appeared, then you can specify
|
||||
that version instead if you wish.) Do not make any other change in
|
||||
these notices.
|
||||
|
||||
Once this change is made in a given copy, it is irreversible for
|
||||
that copy, so the ordinary GNU General Public License applies to all
|
||||
subsequent copies and derivative works made from that copy.
|
||||
|
||||
This option is useful when you wish to copy part of the code of
|
||||
the Library into a program that is not a library.
|
||||
|
||||
4. You may copy and distribute the Library (or a portion or
|
||||
derivative of it, under Section 2) in object code or executable form
|
||||
under the terms of Sections 1 and 2 above provided that you accompany
|
||||
it with the complete corresponding machine-readable source code, which
|
||||
must be distributed under the terms of Sections 1 and 2 above on a
|
||||
medium customarily used for software interchange.
|
||||
|
||||
If distribution of object code is made by offering access to copy
|
||||
from a designated place, then offering equivalent access to copy the
|
||||
source code from the same place satisfies the requirement to
|
||||
distribute the source code, even though third parties are not
|
||||
compelled to copy the source along with the object code.
|
||||
|
||||
5. A program that contains no derivative of any portion of the
|
||||
Library, but is designed to work with the Library by being compiled or
|
||||
linked with it, is called a "work that uses the Library". Such a
|
||||
work, in isolation, is not a derivative work of the Library, and
|
||||
therefore falls outside the scope of this License.
|
||||
|
||||
However, linking a "work that uses the Library" with the Library
|
||||
creates an executable that is a derivative of the Library (because it
|
||||
contains portions of the Library), rather than a "work that uses the
|
||||
library". The executable is therefore covered by this License.
|
||||
Section 6 states terms for distribution of such executables.
|
||||
|
||||
When a "work that uses the Library" uses material from a header file
|
||||
that is part of the Library, the object code for the work may be a
|
||||
derivative work of the Library even though the source code is not.
|
||||
Whether this is true is especially significant if the work can be
|
||||
linked without the Library, or if the work is itself a library. The
|
||||
threshold for this to be true is not precisely defined by law.
|
||||
|
||||
If such an object file uses only numerical parameters, data
|
||||
structure layouts and accessors, and small macros and small inline
|
||||
functions (ten lines or less in length), then the use of the object
|
||||
file is unrestricted, regardless of whether it is legally a derivative
|
||||
work. (Executables containing this object code plus portions of the
|
||||
Library will still fall under Section 6.)
|
||||
|
||||
Otherwise, if the work is a derivative of the Library, you may
|
||||
distribute the object code for the work under the terms of Section 6.
|
||||
Any executables containing that work also fall under Section 6,
|
||||
whether or not they are linked directly with the Library itself.
|
||||
|
||||
6. As an exception to the Sections above, you may also combine or
|
||||
link a "work that uses the Library" with the Library to produce a
|
||||
work containing portions of the Library, and distribute that work
|
||||
under terms of your choice, provided that the terms permit
|
||||
modification of the work for the customer's own use and reverse
|
||||
engineering for debugging such modifications.
|
||||
|
||||
You must give prominent notice with each copy of the work that the
|
||||
Library is used in it and that the Library and its use are covered by
|
||||
this License. You must supply a copy of this License. If the work
|
||||
during execution displays copyright notices, you must include the
|
||||
copyright notice for the Library among them, as well as a reference
|
||||
directing the user to the copy of this License. Also, you must do one
|
||||
of these things:
|
||||
|
||||
a) Accompany the work with the complete corresponding
|
||||
machine-readable source code for the Library including whatever
|
||||
changes were used in the work (which must be distributed under
|
||||
Sections 1 and 2 above); and, if the work is an executable linked
|
||||
with the Library, with the complete machine-readable "work that
|
||||
uses the Library", as object code and/or source code, so that the
|
||||
user can modify the Library and then relink to produce a modified
|
||||
executable containing the modified Library. (It is understood
|
||||
that the user who changes the contents of definitions files in the
|
||||
Library will not necessarily be able to recompile the application
|
||||
to use the modified definitions.)
|
||||
|
||||
b) Use a suitable shared library mechanism for linking with the
|
||||
Library. A suitable mechanism is one that (1) uses at run time a
|
||||
copy of the library already present on the user's computer system,
|
||||
rather than copying library functions into the executable, and (2)
|
||||
will operate properly with a modified version of the library, if
|
||||
the user installs one, as long as the modified version is
|
||||
interface-compatible with the version that the work was made with.
|
||||
|
||||
c) Accompany the work with a written offer, valid for at
|
||||
least three years, to give the same user the materials
|
||||
specified in Subsection 6a, above, for a charge no more
|
||||
than the cost of performing this distribution.
|
||||
|
||||
d) If distribution of the work is made by offering access to copy
|
||||
from a designated place, offer equivalent access to copy the above
|
||||
specified materials from the same place.
|
||||
|
||||
e) Verify that the user has already received a copy of these
|
||||
materials or that you have already sent this user a copy.
|
||||
|
||||
For an executable, the required form of the "work that uses the
|
||||
Library" must include any data and utility programs needed for
|
||||
reproducing the executable from it. However, as a special exception,
|
||||
the materials to be distributed need not include anything that is
|
||||
normally distributed (in either source or binary form) with the major
|
||||
components (compiler, kernel, and so on) of the operating system on
|
||||
which the executable runs, unless that component itself accompanies
|
||||
the executable.
|
||||
|
||||
It may happen that this requirement contradicts the license
|
||||
restrictions of other proprietary libraries that do not normally
|
||||
accompany the operating system. Such a contradiction means you cannot
|
||||
use both them and the Library together in an executable that you
|
||||
distribute.
|
||||
|
||||
7. You may place library facilities that are a work based on the
|
||||
Library side-by-side in a single library together with other library
|
||||
facilities not covered by this License, and distribute such a combined
|
||||
library, provided that the separate distribution of the work based on
|
||||
the Library and of the other library facilities is otherwise
|
||||
permitted, and provided that you do these two things:
|
||||
|
||||
a) Accompany the combined library with a copy of the same work
|
||||
based on the Library, uncombined with any other library
|
||||
facilities. This must be distributed under the terms of the
|
||||
Sections above.
|
||||
|
||||
b) Give prominent notice with the combined library of the fact
|
||||
that part of it is a work based on the Library, and explaining
|
||||
where to find the accompanying uncombined form of the same work.
|
||||
|
||||
8. You may not copy, modify, sublicense, link with, or distribute
|
||||
the Library except as expressly provided under this License. Any
|
||||
attempt otherwise to copy, modify, sublicense, link with, or
|
||||
distribute the Library is void, and will automatically terminate your
|
||||
rights under this License. However, parties who have received copies,
|
||||
or rights, from you under this License will not have their licenses
|
||||
terminated so long as such parties remain in full compliance.
|
||||
|
||||
9. You are not required to accept this License, since you have not
|
||||
signed it. However, nothing else grants you permission to modify or
|
||||
distribute the Library or its derivative works. These actions are
|
||||
prohibited by law if you do not accept this License. Therefore, by
|
||||
modifying or distributing the Library (or any work based on the
|
||||
Library), you indicate your acceptance of this License to do so, and
|
||||
all its terms and conditions for copying, distributing or modifying
|
||||
the Library or works based on it.
|
||||
|
||||
10. Each time you redistribute the Library (or any work based on the
|
||||
Library), the recipient automatically receives a license from the
|
||||
original licensor to copy, distribute, link with or modify the Library
|
||||
subject to these terms and conditions. You may not impose any further
|
||||
restrictions on the recipients' exercise of the rights granted herein.
|
||||
You are not responsible for enforcing compliance by third parties with
|
||||
this License.
|
||||
|
||||
11. If, as a consequence of a court judgment or allegation of patent
|
||||
infringement or for any other reason (not limited to patent issues),
|
||||
conditions are imposed on you (whether by court order, agreement or
|
||||
otherwise) that contradict the conditions of this License, they do not
|
||||
excuse you from the conditions of this License. If you cannot
|
||||
distribute so as to satisfy simultaneously your obligations under this
|
||||
License and any other pertinent obligations, then as a consequence you
|
||||
may not distribute the Library at all. For example, if a patent
|
||||
license would not permit royalty-free redistribution of the Library by
|
||||
all those who receive copies directly or indirectly through you, then
|
||||
the only way you could satisfy both it and this License would be to
|
||||
refrain entirely from distribution of the Library.
|
||||
|
||||
If any portion of this section is held invalid or unenforceable under any
|
||||
particular circumstance, the balance of the section is intended to apply,
|
||||
and the section as a whole is intended to apply in other circumstances.
|
||||
|
||||
It is not the purpose of this section to induce you to infringe any
|
||||
patents or other property right claims or to contest validity of any
|
||||
such claims; this section has the sole purpose of protecting the
|
||||
integrity of the free software distribution system which is
|
||||
implemented by public license practices. Many people have made
|
||||
generous contributions to the wide range of software distributed
|
||||
through that system in reliance on consistent application of that
|
||||
system; it is up to the author/donor to decide if he or she is willing
|
||||
to distribute software through any other system and a licensee cannot
|
||||
impose that choice.
|
||||
|
||||
This section is intended to make thoroughly clear what is believed to
|
||||
be a consequence of the rest of this License.
|
||||
|
||||
12. If the distribution and/or use of the Library is restricted in
|
||||
certain countries either by patents or by copyrighted interfaces, the
|
||||
original copyright holder who places the Library under this License may add
|
||||
an explicit geographical distribution limitation excluding those countries,
|
||||
so that distribution is permitted only in or among countries not thus
|
||||
excluded. In such case, this License incorporates the limitation as if
|
||||
written in the body of this License.
|
||||
|
||||
13. The Free Software Foundation may publish revised and/or new
|
||||
versions of the Lesser General Public License from time to time.
|
||||
Such new versions will be similar in spirit to the present version,
|
||||
but may differ in detail to address new problems or concerns.
|
||||
|
||||
Each version is given a distinguishing version number. If the Library
|
||||
specifies a version number of this License which applies to it and
|
||||
"any later version", you have the option of following the terms and
|
||||
conditions either of that version or of any later version published by
|
||||
the Free Software Foundation. If the Library does not specify a
|
||||
license version number, you may choose any version ever published by
|
||||
the Free Software Foundation.
|
||||
|
||||
14. If you wish to incorporate parts of the Library into other free
|
||||
programs whose distribution conditions are incompatible with these,
|
||||
write to the author to ask for permission. For software which is
|
||||
copyrighted by the Free Software Foundation, write to the Free
|
||||
Software Foundation; we sometimes make exceptions for this. Our
|
||||
decision will be guided by the two goals of preserving the free status
|
||||
of all derivatives of our free software and of promoting the sharing
|
||||
and reuse of software generally.
|
||||
|
||||
NO WARRANTY
|
||||
|
||||
15. BECAUSE THE LIBRARY IS LICENSED FREE OF CHARGE, THERE IS NO
|
||||
WARRANTY FOR THE LIBRARY, TO THE EXTENT PERMITTED BY APPLICABLE LAW.
|
||||
EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT HOLDERS AND/OR
|
||||
OTHER PARTIES PROVIDE THE LIBRARY "AS IS" WITHOUT WARRANTY OF ANY
|
||||
KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE
|
||||
LIBRARY IS WITH YOU. SHOULD THE LIBRARY PROVE DEFECTIVE, YOU ASSUME
|
||||
THE COST OF ALL NECESSARY SERVICING, REPAIR OR CORRECTION.
|
||||
|
||||
16. IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN
|
||||
WRITING WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MAY MODIFY
|
||||
AND/OR REDISTRIBUTE THE LIBRARY AS PERMITTED ABOVE, BE LIABLE TO YOU
|
||||
FOR DAMAGES, INCLUDING ANY GENERAL, SPECIAL, INCIDENTAL OR
|
||||
CONSEQUENTIAL DAMAGES ARISING OUT OF THE USE OR INABILITY TO USE THE
|
||||
LIBRARY (INCLUDING BUT NOT LIMITED TO LOSS OF DATA OR DATA BEING
|
||||
RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD PARTIES OR A
|
||||
FAILURE OF THE LIBRARY TO OPERATE WITH ANY OTHER SOFTWARE), EVEN IF
|
||||
SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH
|
||||
DAMAGES.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
How to Apply These Terms to Your New Libraries
|
||||
|
||||
If you develop a new library, and you want it to be of the greatest
|
||||
possible use to the public, we recommend making it free software that
|
||||
everyone can redistribute and change. You can do so by permitting
|
||||
redistribution under these terms (or, alternatively, under the terms of the
|
||||
ordinary General Public License).
|
||||
|
||||
To apply these terms, attach the following notices to the library. It is
|
||||
safest to attach them to the start of each source file to most effectively
|
||||
convey the exclusion of warranty; and each file should have at least the
|
||||
"copyright" line and a pointer to where the full notice is found.
|
||||
|
||||
<one line to give the library's name and a brief idea of what it does.>
|
||||
Copyright (C) <year> <name of author>
|
||||
|
||||
This library is free software; you can redistribute it and/or
|
||||
modify it under the terms of the GNU Lesser General Public
|
||||
License as published by the Free Software Foundation; either
|
||||
version 2.1 of the License, or (at your option) any later version.
|
||||
|
||||
This library is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
||||
Lesser General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU Lesser General Public
|
||||
License along with this library; if not, write to the Free Software
|
||||
Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
|
||||
Also add information on how to contact you by electronic and paper mail.
|
||||
|
||||
You should also get your employer (if you work as a programmer) or your
|
||||
school, if any, to sign a "copyright disclaimer" for the library, if
|
||||
necessary. Here is a sample; alter the names:
|
||||
|
||||
Yoyodyne, Inc., hereby disclaims all copyright interest in the
|
||||
library `Frob' (a library for tweaking knobs) written by James Random Hacker.
|
||||
|
||||
<signature of Ty Coon>, 1 April 1990
|
||||
Ty Coon, President of Vice
|
||||
|
||||
That's all there is to it!
|
||||
|
||||
* Neither the name of the copyright holder nor the names of its
|
||||
contributors may be used to endorse or promote products derived from
|
||||
this software without specific prior written permission.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
@@ -1,33 +0,0 @@
|
||||
This work was produced under the auspices of the U.S. Department of Energy by
|
||||
Lawrence Livermore National Laboratory under Contract DE-AC52-07NA27344.
|
||||
|
||||
This work was prepared as an account of work sponsored by an agency of
|
||||
the United States Government. Neither the United States Government nor
|
||||
Lawrence Livermore National Security, LLC, nor any of their employees
|
||||
makes any warranty, expressed or implied, or assumes any legal liability
|
||||
or responsibility for the accuracy, completeness, or usefulness of any
|
||||
information, apparatus, product, or process disclosed, or represents that
|
||||
its use would not infringe privately owned rights.
|
||||
|
||||
Reference herein to any specific commercial product, process, or service
|
||||
by trade name, trademark, manufacturer, or otherwise does not necessarily
|
||||
constitute or imply its endorsement, recommendation, or favoring by the
|
||||
United States Government or Lawrence Livermore National Security, LLC.
|
||||
|
||||
The views and opinions of authors expressed herein do not necessarily
|
||||
state or reflect those of the United States Government or Lawrence
|
||||
Livermore National Security, LLC, and shall not be used for advertising
|
||||
or product endorsement purposes.
|
||||
|
||||
Inclusion of external software:
|
||||
|
||||
This project distributes the sources of several external software products with
|
||||
their own respective licenses which can be found in their code and attached
|
||||
license files. These software products and their licenses are as follows:
|
||||
|
||||
* AmgXWrapper (linalg/amgxsolver.{hpp,cpp}) -- MIT license
|
||||
* Catch++ (tests/unit/catch.hpp) -- Boost 1.0 license
|
||||
* Gecko (general/gecko.{cpp,hpp}) -- BSD 3-clause license
|
||||
* Picojson (fem/picojson.h) -- Custom 2-clause license
|
||||
* TinyXML2 (general/tinyxml2.{cpp,h}) -- zlib license
|
||||
* Zstr (general/zstr.hpp) -- MIT license
|
||||
@@ -5,19 +5,19 @@
|
||||
| | | | | || _|| __/| | | | | |
|
||||
|_| |_| |_||_| \___||_| |_| |_|
|
||||
|
||||
https://mfem.org
|
||||
http://mfem.org
|
||||
|
||||
MFEM is a modular parallel C++ library for finite element methods. Its goal is
|
||||
to enable high-performance scalable finite element discretization research and
|
||||
application development on a wide variety of platforms, ranging from laptops to
|
||||
supercomputers.
|
||||
to enable the research and development of scalable finite element discretization
|
||||
and solver algorithms through general finite element abstractions, accurate and
|
||||
flexible visualization, and tight integration with the hypre library.
|
||||
|
||||
* For building instructions, see the file INSTALL, or type "make help".
|
||||
|
||||
* Copyright and licensing information can be found in files LICENSE and NOTICE.
|
||||
* Copyright and licensing information can be found in the file COPYRIGHT.
|
||||
|
||||
* The best starting point for new users interested in MFEM's features is to
|
||||
review the examples and miniapps at https://mfem.org/examples.
|
||||
* The best starting point for new users interested in MFEM's features is the
|
||||
interactive documentation in examples/README.html.
|
||||
|
||||
* Developers interested in contributing to the library, should read the
|
||||
instructions and documentation in the CONTRIBUTING.md file.
|
||||
@@ -39,31 +39,29 @@ conforming and non-conforming (AMR) adaptive refinement. Arbitrary element
|
||||
transformations, allowing for high-order mesh elements with curved boundaries,
|
||||
are also supported.
|
||||
|
||||
When used as a "finite element to linear algebra translator", MFEM can take a
|
||||
problem described in terms of finite element-type objects, and produce the
|
||||
corresponding linear algebra vectors and fully or partially assembled operators,
|
||||
e.g. in the form of global sparse matrices or matrix-free operators. The library
|
||||
includes simple smoothers and Krylov solvers, such as PCG, MINRES and GMRES, as
|
||||
well as support for sequential sparse direct solvers from the SuiteSparse
|
||||
MFEM is commonly used as a "finite element to linear algebra translator", since
|
||||
it can take a problem described in terms of finite element-type objects, and
|
||||
produce the corresponding linear algebra vectors and sparse matrices. In order
|
||||
to facilitate this, MFEM uses compressed sparse row (CSR) sparse matrix storage
|
||||
and includes simple smoothers and Krylov solvers, such as PCG, MINRES and GMRES,
|
||||
as well as support for sequential sparse direct solvers from the SuiteSparse
|
||||
library. Nonlinear solvers (the Newton method), eigensolvers (LOBPCG), and
|
||||
several explicit and implicit Runge-Kutta time integrators are also available.
|
||||
|
||||
MFEM supports MPI-based parallelism throughout the library, and can readily be
|
||||
used as a scalable unstructured finite element problem generator. Starting with
|
||||
version 4.0, MFEM offers support for GPU acceleration, and programming models,
|
||||
such as CUDA, HIP, OCCA, RAJA and OpenMP. MFEM-based applications require
|
||||
minimal changes to switch from a serial to a highly-performant MPI-parallel
|
||||
version of the code, where they can take advantage of the integrated linear
|
||||
solvers from the hypre library. Comprehensive support for other external
|
||||
packages, e.g. PETSc, SUNDIALS and libCEED is also included, giving access to
|
||||
additional linear and nonlinear solvers, preconditioners, time integrators, etc.
|
||||
used as a scalable unstructured finite element problem generator. MFEM-based
|
||||
applications require minimal changes to transition from a serial to a
|
||||
high-performing parallel version of the code, where they can take advantage of
|
||||
the integrated scalable linear solvers from the hypre library. Comprehensive
|
||||
support for other external packages, e.g. PETSc and SUNDIALS is also included,
|
||||
giving access to many additional linear and nonlinear solvers, preconditioners,
|
||||
time integrators, etc.
|
||||
|
||||
For examples of using MFEM, see the examples/ and miniapps/ directories, as well
|
||||
as the OpenGL visualization tool GLVis which is available at https://glvis.org.
|
||||
as the OpenGL visualization tool GLVis which is available at http://glvis.org.
|
||||
|
||||
MFEM is distributed under the terms of the BSD-3 license. All new contributions
|
||||
must be made under this license. See LICENSE and NOTICE for details.
|
||||
This project is released under the LGPL v2.1 license with static linking
|
||||
exception. See files COPYRIGHT and LICENSE file for full details.
|
||||
|
||||
SPDX-License-Identifier: BSD-3-Clause
|
||||
LLNL Release Number: LLNL-CODE-806117
|
||||
LLNL Release Number: LLNL-CODE-443211
|
||||
DOI: 10.11578/dc.20171025.1248
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_ALL_HPP
|
||||
#define MFEM_BACKENDS_ALL_HPP
|
||||
|
||||
#include "../config/config.hpp"
|
||||
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "base/backend.hpp"
|
||||
|
||||
#ifdef MFEM_USE_OCCA
|
||||
#include "occa/backend.hpp"
|
||||
#endif
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_ALL_HPP
|
||||
@@ -0,0 +1,213 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
#define MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "layout.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Extension to the template class Array<T>
|
||||
class PArray : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Layout with shared ownership (smart pointer)
|
||||
DLayout layout;
|
||||
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new array (in @a *clone) of the same dynamic
|
||||
type as this array using the same layout and ItemSize().
|
||||
|
||||
Set @a *clone to NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this array is copied to the new
|
||||
array; otherwise, the new array remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the array data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
virtual PArray *DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const = 0;
|
||||
|
||||
/// Resize the array, reallocating its data if necessary.
|
||||
/** If @a buffer is not NULL, return the array data (in @a *buffer), if it
|
||||
is stored as a contiguous array on the host; otherwise, set @a *buffer to
|
||||
NULL. Returns 0 on success and non-zero otherwise, e.g. if memory
|
||||
allocation fails.
|
||||
|
||||
If the @a new_layout is not supported, a non-zero error code will be
|
||||
returned.
|
||||
|
||||
The @a new_layout has to be valid, i.e. new_layout != NULL and
|
||||
new_layout->HasEngine() == true.
|
||||
|
||||
@note If reallocation is performed, the previous content of the array is
|
||||
NOT copied to the new location. */
|
||||
virtual int DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Get access to the contents of the array in host memory, as a
|
||||
contiguous array. */
|
||||
/** If the array data is stored as a contiguous array in host memory, return
|
||||
a pointer to it. Otherwise, copy the data to @a buffer (if @a buffer is
|
||||
not NULL) and return @a buffer.
|
||||
@note If not NULL, @a buffer is assumed to be of size greater than or
|
||||
equal to Size(). */
|
||||
virtual void *DoPullData(void *buffer, std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Set all entries of the array to the (single) value pointed to by
|
||||
@a value_ptr. */
|
||||
virtual void DoFill(const void *value_ptr, std::size_t item_size) = 0;
|
||||
|
||||
/** @brief Set all Size() entries of the array from the given contiguous
|
||||
array, @a src_buffer, on the host. */
|
||||
virtual void DoPushData(const void *src_buffer, std::size_t item_size) = 0;
|
||||
|
||||
/// Copy the data from @a src to @a *this.
|
||||
/** Both arrays must have the same dynamic type, layout, and item_size. */
|
||||
virtual void DoAssign(const PArray &src, std::size_t item_size) = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
/** @brief The @a layout parameter will be reference counted and therefore it
|
||||
should be dynamically allocated. */
|
||||
/** The @a layout must be valid in the sense that layout != NULL and
|
||||
layout->HasEngine() == true. */
|
||||
PArray(PLayout &p_layout)
|
||||
: layout(&p_layout)
|
||||
{
|
||||
MFEM_ASSERT(layout && layout->HasEngine(), "invalid layout");
|
||||
}
|
||||
|
||||
virtual ~PArray() { }
|
||||
|
||||
/// Get the current size of the array.
|
||||
std::size_t Size() const { return layout->Size(); }
|
||||
|
||||
/// Get the current layout of the array.
|
||||
PLayout &GetLayout() const { return *layout; }
|
||||
|
||||
/// TODO
|
||||
/// Note: we cannot use static_cast for class PArray.
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return dynamic_cast<derived_t&>(*this); }
|
||||
|
||||
/// TODO
|
||||
/// Note: we cannot use static_cast for class PArray.
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return dynamic_cast<const derived_t&>(*this); }
|
||||
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
// TODO: Asynchronous execution interface ...
|
||||
|
||||
|
||||
/**
|
||||
@name Public virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new array (in @a *clone) of the same dynamic
|
||||
type as this array using the same layout and ItemSize().
|
||||
|
||||
Set @a *clone to NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this array is copied to the new
|
||||
array; otherwise, the new array remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the array data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
template <typename T>
|
||||
DArray Clone(bool copy_data, T **buffer) const
|
||||
{ return DArray(DoClone(copy_data, (void**)buffer, sizeof(T))); }
|
||||
|
||||
/// Resize the array, reallocating its data if necessary.
|
||||
/** If @a buffer is not NULL, return the array data (in @a *buffer), if it
|
||||
is stored as a contiguous array on the host; otherwise, set @a *buffer to
|
||||
NULL. Returns 0 on success and non-zero otherwise, e.g. if memory
|
||||
allocation fails.
|
||||
|
||||
If the @a new_layout is not supported, a non-zero error code will be
|
||||
returned.
|
||||
|
||||
The @a new_layout has to be valid, i.e. new_layout != NULL and
|
||||
new_layout->HasEngine() == true.
|
||||
|
||||
@note If reallocation is performed, the previous content of the array is
|
||||
NOT copied to the new location. */
|
||||
template <typename T>
|
||||
int Resize(PLayout &new_layout, T **buffer)
|
||||
{ return DoResize(new_layout, (void**)buffer, sizeof(T)); }
|
||||
|
||||
/// Shortcut for Resize(*layout, buffer).
|
||||
/** This method is useful for updating the array after its layout is changed
|
||||
externally. */
|
||||
template <typename T>
|
||||
int Update(T **buffer)
|
||||
{ return DoResize(*layout, (void**)buffer, sizeof(T)); }
|
||||
|
||||
/// Shortcut for layout->Resize(new_size) followed by Update()
|
||||
template <typename T>
|
||||
int Resize(std::size_t new_size, T **buffer)
|
||||
{ layout->Resize(new_size); return Update(buffer); }
|
||||
|
||||
/** @brief Get access to the contents of the array in host memory, as a
|
||||
contiguous array. */
|
||||
/** If the array data is stored as a contiguous array in host memory, return
|
||||
a pointer to it. Otherwise, copy the data to @a buffer (if @a buffer is
|
||||
not NULL) and return @a buffer.
|
||||
@note If not NULL, @a buffer is assumed to be of size greater than or
|
||||
equal to Size(). */
|
||||
template <typename T>
|
||||
T *PullData(T *buffer)
|
||||
{ return Size() ? (T*)DoPullData((void*)buffer, sizeof(T)) : NULL; }
|
||||
|
||||
/** @brief Set all entries of the array to the (single) value pointed to by
|
||||
@a value_ptr. */
|
||||
template <typename T>
|
||||
void Fill(const T &value) { if (Size()) { DoFill(&value, sizeof(T)); } }
|
||||
|
||||
/** @brief Set all Size() entries of the array from the given contiguous
|
||||
array, @a src_buffer, on the host. */
|
||||
template <typename T>
|
||||
void PushData(const T *src_buffer)
|
||||
{ if (Size()) { DoPushData(src_buffer, sizeof(T)); } }
|
||||
|
||||
/// Copy the data from @a src to @a *this.
|
||||
/** Both arrays must have the same dynamic type, layout, and entry type. */
|
||||
template <typename T>
|
||||
void Assign(const PArray &src) { if (Size()) { DoAssign(src, sizeof(T)); } }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_ARRAY_HPP
|
||||
@@ -0,0 +1,57 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
#define MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "memory_resource.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "array.hpp"
|
||||
#include "vector.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
|
||||
#include <string>
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include <mpi.h>
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// TODO
|
||||
class Backend
|
||||
{
|
||||
public:
|
||||
/// TODO
|
||||
virtual ~Backend() { }
|
||||
|
||||
/// TODO
|
||||
virtual bool Supports(const std::string &engine_spec) const = 0;
|
||||
|
||||
/// TODO
|
||||
virtual Engine *Create(const std::string &engine_spec) = 0;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// TODO
|
||||
virtual Engine *Create(MPI_Comm comm, const std::string &engine_spec) = 0;
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_BACKEND_HPP
|
||||
@@ -0,0 +1,72 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
#define MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class Vector;
|
||||
class OperatorHandle;
|
||||
class BilinearForm;
|
||||
|
||||
/// TODO: doxygen
|
||||
class PBilinearForm : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
/// Not owned.
|
||||
BilinearForm *bform;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
PBilinearForm(const Engine &e, BilinearForm &bf)
|
||||
: engine(&e), bform(&bf) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~PBilinearForm() { }
|
||||
|
||||
/// Get the associated Engine
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
/// Assemble the PBilinearForm.
|
||||
/** This method is called from the method BilinearForm::Assemble() of the
|
||||
associated BilinearForm #bform.
|
||||
@returns True, if the host assembly should be skipped. */
|
||||
virtual bool Assemble() = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void FormSystemMatrix(const Array<int> &ess_tdof_list,
|
||||
OperatorHandle &A) = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void FormLinearSystem(const Array<int> &ess_tdof_list,
|
||||
Vector &x, Vector &b,
|
||||
OperatorHandle &A, Vector &X, Vector &B,
|
||||
int copy_interior) = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual void RecoverFEMSolution(const Vector &X, const Vector &b,
|
||||
Vector &x) = 0;
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_BILINEARFORM_HPP
|
||||
@@ -0,0 +1,49 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
Engine::Engine(Backend *b, int n_mem, int n_workers)
|
||||
: backend(b),
|
||||
#ifdef MFEM_USE_MPI
|
||||
comm(MPI_COMM_NULL),
|
||||
#endif
|
||||
num_mem_res(n_mem),
|
||||
num_workers(n_workers),
|
||||
memory_resources(new MemoryResource*[num_mem_res]()),
|
||||
workers_weights(new double[num_workers]()),
|
||||
workers_mem_res(new int[num_workers]())
|
||||
{
|
||||
// Note: all arrays are value-initialized with zeros.
|
||||
}
|
||||
|
||||
Engine::~Engine()
|
||||
{
|
||||
delete [] workers_mem_res;
|
||||
delete [] workers_weights;
|
||||
for (int i = 0; i < num_mem_res; i++)
|
||||
{
|
||||
delete memory_resources[i];
|
||||
}
|
||||
delete [] memory_resources;
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
@@ -0,0 +1,190 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
#define MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/scalars.hpp"
|
||||
#include "memory_resource.hpp"
|
||||
#include "smart_pointers.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#include <mpi.h>
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
// Forward declarations.
|
||||
class Backend;
|
||||
template <typename T> class Array;
|
||||
class Operator;
|
||||
class FiniteElementSpace;
|
||||
class LinearForm;
|
||||
class BilinearForm;
|
||||
class MixedBilinearForm;
|
||||
class NonlinearForm;
|
||||
|
||||
|
||||
/// In parallel, each MPI rank will usually create a single engine.
|
||||
class Engine : public RefCounted
|
||||
{
|
||||
protected:
|
||||
Backend *backend; ///< Backend that created the engine. Not owned.
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
MPI_Comm comm; ///< Associated MPI communicator (may be MPI_COMM_NULL).
|
||||
#endif
|
||||
|
||||
/// Number of memory resources used by the Engine.
|
||||
int num_mem_res;
|
||||
/// Number of workers used by the Engine.
|
||||
int num_workers;
|
||||
|
||||
/// Memory resources used by the engine - array of pointers.
|
||||
/** Both the array and the entries are owned. */
|
||||
MemoryResource **memory_resources;
|
||||
|
||||
/// Relative computational speed of the workers. Owned.
|
||||
double *workers_weights;
|
||||
|
||||
/// For each worker, which memory resource it uses.
|
||||
int *workers_mem_res;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
Engine(Backend *b, int n_mem, int n_workers);
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual ~Engine();
|
||||
|
||||
|
||||
/**
|
||||
@name Machine resources interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// Get the associated MPI_Comm
|
||||
MPI_Comm GetComm() const { return comm; }
|
||||
#endif
|
||||
|
||||
/// TODO
|
||||
int GetNumMemRes() const { return num_mem_res; }
|
||||
|
||||
/// TODO
|
||||
MemoryResource &GetMemRes(int idx) const { return *memory_resources[idx]; }
|
||||
|
||||
/// TODO
|
||||
int GetNumWorkers() const { return num_workers; }
|
||||
|
||||
/// TODO
|
||||
const double *GetWorkersWeights() const { return workers_weights; }
|
||||
|
||||
/// TODO
|
||||
const int *GetWorkersMemRes() const { return workers_mem_res; }
|
||||
|
||||
///@}
|
||||
// End: Machine resources interface
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element data structures and algorithms
|
||||
*/
|
||||
///@{
|
||||
|
||||
// TODO: Asynchronous execution in this class ...
|
||||
|
||||
/// Allocate and return a new layout for the given @a size.
|
||||
/** The layout decomposition (in the case of multiple workers) is determined
|
||||
automatically by the Engine using a deterministic algorithm: calls to
|
||||
this method with the same @a size will produce the same result, as long
|
||||
as the Engine remains unmodified between the calls.
|
||||
|
||||
The returned object is allocated with operator new and must be
|
||||
deallocated by the caller.
|
||||
|
||||
TODO: Returns NULL if memory allocation fails?
|
||||
*/
|
||||
virtual DLayout MakeLayout(std::size_t size) const = 0;
|
||||
|
||||
/// Allocate and return a new layout for the given worker decomposition.
|
||||
/** The returned object is allocated with operator new and must be
|
||||
deallocated by the caller.
|
||||
|
||||
TODO: Returns NULL if memory allocation fails?
|
||||
|
||||
The @a offsets should satisfy: offsets.Size() == number of workers + 1,
|
||||
offsets[0] == 0, and offsets[i] <= offsets[i+1], for i: 0 <= i < number
|
||||
of workers. */
|
||||
virtual DLayout MakeLayout(const Array<std::size_t> &offsets) const = 0;
|
||||
|
||||
// Note: There may be other ways to construct layouts in the future, e.g.
|
||||
// block-vector layouts, or multi-vector layouts.
|
||||
|
||||
/// TODO
|
||||
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const = 0;
|
||||
|
||||
/// Allocate and return a new vector using the given @a layout.
|
||||
/** The returned object is a smart pointer that will automatically deallocate
|
||||
the vector.
|
||||
|
||||
TODO: Produce an error if memory allocation fails?
|
||||
|
||||
Only layouts returned by this Engine are guaranteed to be supported.
|
||||
Using a type that is not supported will produce an error. */
|
||||
virtual DVector MakeVector(PLayout &layout,
|
||||
int type_id = ScalarId<double>::value) const = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual DFiniteElementSpace MakeFESpace(FiniteElementSpace &fes) const = 0;
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual DBilinearForm MakeBilinearForm(BilinearForm &bf) const = 0;
|
||||
|
||||
|
||||
// Question: How do we construct coefficients?
|
||||
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual void AssembleLinearForm(LinearForm &l_form) const = 0;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual Operator *MakeOperator(const MixedBilinearForm &mbl_form) const = 0;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual Operator *MakeOperator(const NonlinearForm &nl_form) const = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_ENGINE_HPP
|
||||
@@ -0,0 +1,98 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
#define MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "utils.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class FiniteElementSpace;
|
||||
class QuadratureSpace;
|
||||
|
||||
/// TODO: doxygen
|
||||
class PFiniteElementSpace : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
/// Not owned.
|
||||
mfem::FiniteElementSpace *fes;
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
PFiniteElementSpace(const Engine &e, mfem::FiniteElementSpace &fespace)
|
||||
: engine(&e), fes(&fespace) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~PFiniteElementSpace() { }
|
||||
|
||||
/// Get the associated engine
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
/// Return the associated mfem::FiniteElementSpace
|
||||
mfem::FiniteElementSpace *GetFESpace() const { return fes; }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element space functionality
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// TODO
|
||||
/** Return the operator mapping T-vectors to L-vectors. If a NULL pointer is
|
||||
returned then the mapping is the idenity. */
|
||||
virtual const mfem::Operator *GetProlongationOperator() const = 0;
|
||||
|
||||
/// TODO
|
||||
/** Return the operator mapping L-vectors to T-vectors that extracts the
|
||||
subset of all true dofs, i.e. no assembly is performed. If a NULL pointer
|
||||
is returned then the mapping is the idenity. */
|
||||
virtual const mfem::Operator *GetRestrictionOperator() const = 0;
|
||||
|
||||
/// TODO
|
||||
/** Return the operator mapping L-vectors to Q-vectors that evaluates the
|
||||
values of a GridFunction as a QuadratureFunction on the given
|
||||
QuadratureSpace. If the returned pointer is NULL, then the mapping is the
|
||||
identity. */
|
||||
virtual const mfem::Operator *GetInterpolationOperator(
|
||||
const mfem::QuadratureSpace &qspace) const = 0;
|
||||
|
||||
/// TODO
|
||||
/** Return the operator mapping L-vectors to Q-vectors that evaluates the
|
||||
_reference element_ gradients of a GridFunction as a QuadratureFunction
|
||||
on the given QuadratureSpace. */
|
||||
virtual const mfem::Operator *GetGradientOperator(
|
||||
const mfem::QuadratureSpace &qspace) const = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_FE_SPACE_HPP
|
||||
@@ -0,0 +1,110 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
#define MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "smart_pointers.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic layout (array/vector layout descriptor)
|
||||
class PLayout : public RefCounted
|
||||
{
|
||||
protected:
|
||||
/// Engine with shared ownership
|
||||
SharedPtr<const Engine> engine;
|
||||
std::size_t size;
|
||||
|
||||
template <typename DObject>
|
||||
struct Maker
|
||||
{
|
||||
template <typename entry_t>
|
||||
static DObject MakeNew(PLayout &layout);
|
||||
};
|
||||
|
||||
public:
|
||||
explicit PLayout(std::size_t s = 0) : engine(NULL), size(s) { }
|
||||
|
||||
explicit PLayout(const Engine &e, std::size_t s = 0)
|
||||
: engine(&e), size(s) { }
|
||||
|
||||
virtual ~PLayout() { }
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Resize the layout
|
||||
virtual void Resize(std::size_t new_size) { size = new_size; }
|
||||
|
||||
/// Resize the layout based on the given worker offsets
|
||||
virtual void Resize(const Array<std::size_t> &offsets)
|
||||
{ MFEM_ABORT("method not supported"); }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
/// Layouts without engine cannot create DArray, DVector, etc.
|
||||
bool HasEngine() const { return engine != NULL; }
|
||||
|
||||
/// TODO: doxygen
|
||||
const Engine &GetEngine() const { return *engine; }
|
||||
|
||||
/// TODO: doxygen
|
||||
std::size_t Size() const { return size; }
|
||||
|
||||
/// TODO
|
||||
/** Useful in backends for down-casting to a backend-specific layout type.
|
||||
|
||||
When MFEM_DEBUG=YES, performs a type check using dynamic_cast. */
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
/// TODO
|
||||
/** Useful in backends for down-casting to a backend-specific layout type.
|
||||
|
||||
When MFEM_DEBUG=YES, performs a type check using dynamic_cast. */
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename DObject, typename entry_t>
|
||||
DObject Make()
|
||||
{
|
||||
MFEM_ASSERT(HasEngine(), "this method requires an Engine");
|
||||
return Maker<DObject>::template MakeNew<entry_t>(*this);
|
||||
}
|
||||
};
|
||||
|
||||
template <> struct PLayout::Maker<DArray>
|
||||
{
|
||||
template <typename entry_t> static DArray MakeNew(PLayout &layout)
|
||||
{ return layout.GetEngine().MakeArray(layout, sizeof(entry_t)); }
|
||||
};
|
||||
|
||||
template <> struct PLayout::Maker<DVector>
|
||||
{
|
||||
template <typename entry_t> static DVector MakeNew(PLayout &layout)
|
||||
{ return layout.GetEngine().MakeVector(layout, ScalarId<entry_t>::value); }
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_LAYOUT_HPP
|
||||
@@ -0,0 +1,59 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "memory_resource.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <cerrno>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
void *NewDeleteMemoryResource::DoAllocate(std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
void *p = ::operator new[](bytes);
|
||||
MFEM_VERIFY(!alignment || (std::size_t)(p) % alignment == 0,
|
||||
"invalid alignment");
|
||||
return p;
|
||||
}
|
||||
|
||||
void NewDeleteMemoryResource::DoDeallocate(void *p, std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
::operator delete[](p);
|
||||
}
|
||||
|
||||
|
||||
void *AlignedMemoryResource::DoAllocate(std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
void *p;
|
||||
if (!alignment) { alignment = sizeof(long double); }
|
||||
MFEM_VERIFY(posix_memalign(&p, alignment, bytes) == 0,
|
||||
"error in posix_memalign(): " << strerror(errno));
|
||||
return p;
|
||||
}
|
||||
|
||||
void AlignedMemoryResource::DoDeallocate(void *p, std::size_t bytes,
|
||||
std::size_t alignment)
|
||||
{
|
||||
free(p);
|
||||
}
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
@@ -0,0 +1,70 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
#define MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include <cstddef>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic memory resource. Similar to C++17's std::pmr::memory_resource.
|
||||
class MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment) = 0;
|
||||
virtual void DoDeallocate(void* p, std::size_t bytes,
|
||||
std::size_t alignment) = 0;
|
||||
|
||||
public:
|
||||
// Implicitly defined default & copy constructors
|
||||
|
||||
/// Virtual destructor.
|
||||
virtual ~MemoryResource() { }
|
||||
|
||||
/// If alignment == 0, use default alignment.
|
||||
void *Allocate(std::size_t bytes, std::size_t alignment = 0)
|
||||
{ return DoAllocate(bytes, alignment); }
|
||||
|
||||
/// If alignment == 0, use default alignment.
|
||||
void Deallocate(void *p, std::size_t bytes, std::size_t alignment = 0)
|
||||
{ DoDeallocate(p, bytes, alignment); }
|
||||
};
|
||||
|
||||
|
||||
/** @brief Dynamic host memory resource using operator new[](std::size_t) for
|
||||
allocation and operator delete[](void*) for deallocation. */
|
||||
class NewDeleteMemoryResource : public MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
|
||||
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
|
||||
};
|
||||
|
||||
|
||||
/** @brief Dynamic host memory resource using posix_memalign() for aligned
|
||||
allocation and free() for deallocation. */
|
||||
class AlignedMemoryResource : public MemoryResource
|
||||
{
|
||||
protected:
|
||||
virtual void *DoAllocate(std::size_t bytes, std::size_t alignment);
|
||||
virtual void DoDeallocate(void *p, std::size_t bytes, std::size_t alignment);
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_MEMORY_RESOURCE_HPP
|
||||
@@ -0,0 +1,234 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
#define MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "utils.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
#include <cstddef>
|
||||
|
||||
// #define MFEM_TRACE_SHARED_PTR
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#include "../../general/globals.hpp"
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Base class for classes with simple reference counting.
|
||||
/** Reference counting is performed by the class SharedPtr. */
|
||||
class RefCounted
|
||||
{
|
||||
private:
|
||||
mutable unsigned ref_count;
|
||||
|
||||
/// Only class SharedPtr can access ref_count.
|
||||
template <typename T> friend class SharedPtr;
|
||||
|
||||
public:
|
||||
RefCounted() : ref_count(0) { }
|
||||
|
||||
/** @brief Prevent SharedPtr objects from deleting this object by
|
||||
incrementing the reference counter by one. */
|
||||
void DontDelete() const { ++ref_count; }
|
||||
};
|
||||
|
||||
|
||||
/** @brief Smart pointer class that manages objects of type T derived from class
|
||||
RefCounted. */
|
||||
/** This class is generally meant to work with dynamically allocated object,
|
||||
specifically objects allocated with operator new(). It will invoke operator
|
||||
delete() to destroy the managed object when its reference counter reaches
|
||||
zero. This behavior can be overriden by calling RefCounted::DontDelete() to
|
||||
ensure that an object will not be deleted by a SharedPtr that holds a
|
||||
pointer to it.
|
||||
@note This class is NOT thread-safe and does not support circular ownership.
|
||||
*/
|
||||
template <typename T>
|
||||
class SharedPtr
|
||||
{
|
||||
public:
|
||||
typedef T stored_type;
|
||||
|
||||
private:
|
||||
T *ptr;
|
||||
|
||||
void Init(T *new_ptr)
|
||||
{
|
||||
ptr = new_ptr;
|
||||
if (ptr) { ++ptr->RefCounted::ref_count; }
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#elif 0
|
||||
mfem::out << " [" << _MFEM_FUNC_NAME << "]: ptr = " << ptr;
|
||||
if (ptr)
|
||||
{
|
||||
mfem::out << ", new ref_count = " << ptr->RefCounted::ref_count;
|
||||
}
|
||||
mfem::out << '\n';
|
||||
#endif
|
||||
}
|
||||
void Destroy()
|
||||
{
|
||||
MFEM_ASSERT(!ptr || ptr->RefCounted::ref_count >= 1, "invalid use");
|
||||
if (ptr && --ptr->RefCounted::ref_count == 0) { delete ptr; }
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
#elif 0
|
||||
mfem::out << " [" << _MFEM_FUNC_NAME << "]: ptr = " << ptr;
|
||||
if (ptr)
|
||||
{
|
||||
mfem::out << ", new ref_count = " << ptr->RefCounted::ref_count;
|
||||
}
|
||||
mfem::out << '\n';
|
||||
#endif
|
||||
}
|
||||
|
||||
public:
|
||||
SharedPtr() : ptr(NULL)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]: ptr = " << ptr << '\n';
|
||||
#endif
|
||||
}
|
||||
|
||||
SharedPtr(const SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(other.ptr);
|
||||
}
|
||||
|
||||
template <typename U>
|
||||
SharedPtr(const SharedPtr<U> &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(other.Get());
|
||||
}
|
||||
|
||||
explicit SharedPtr(T *p)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Init(p);
|
||||
}
|
||||
|
||||
~SharedPtr()
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Destroy();
|
||||
}
|
||||
|
||||
SharedPtr &operator=(const SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Reset(other.ptr); return *this;
|
||||
}
|
||||
|
||||
template <typename U>
|
||||
SharedPtr &operator=(const SharedPtr<U> &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Reset(other.Get()); return *this;
|
||||
}
|
||||
|
||||
T &operator*() const { return *ptr; }
|
||||
T *operator->() const { return ptr; }
|
||||
|
||||
operator bool() const { return ptr; }
|
||||
bool operator!() const { return !ptr; }
|
||||
|
||||
template <typename U>
|
||||
bool operator==(const SharedPtr<U> &other) const
|
||||
{ return ptr == other.Get(); }
|
||||
template <typename U>
|
||||
bool operator!=(const SharedPtr<U> &other) const
|
||||
{ return ptr != other.Get(); }
|
||||
|
||||
// Comparison to any type convertible to void *, e.g. the type of NULL.
|
||||
template <typename U>
|
||||
bool operator==(const U &p) const { return ptr == (void*) p; }
|
||||
template <typename U>
|
||||
bool operator!=(const U &p) const { return ptr != (void*) p; }
|
||||
|
||||
T *Get() const { return ptr; }
|
||||
|
||||
/// TODO
|
||||
template <typename derived_t>
|
||||
derived_t *As() const { return util::As<derived_t>(ptr); }
|
||||
|
||||
unsigned UseCount() const { return ptr ? ptr->RefCounted::ref_count : 0; }
|
||||
|
||||
void Reset()
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
Destroy();
|
||||
ptr = NULL;
|
||||
}
|
||||
|
||||
/// The type U* needs to be implicitly convertible to T*
|
||||
template <typename U>
|
||||
void Reset(U *new_ptr)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
if (ptr != new_ptr) { Destroy(); Init(new_ptr); }
|
||||
}
|
||||
|
||||
void Swap(SharedPtr &other)
|
||||
{
|
||||
#ifdef MFEM_TRACE_SHARED_PTR
|
||||
mfem::out << '[' << _MFEM_FUNC_NAME << "]\n";
|
||||
#endif
|
||||
std::swap(ptr, other.ptr);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template <class T>
|
||||
inline void Swap(SharedPtr<T> &a, SharedPtr<T> &b) { a.Swap(b); }
|
||||
|
||||
|
||||
class PLayout;
|
||||
typedef SharedPtr<PLayout> DLayout;
|
||||
|
||||
class PArray;
|
||||
typedef SharedPtr<PArray> DArray;
|
||||
|
||||
class PVector;
|
||||
typedef SharedPtr<PVector> DVector;
|
||||
|
||||
class PFiniteElementSpace;
|
||||
typedef SharedPtr<PFiniteElementSpace> DFiniteElementSpace;
|
||||
|
||||
class PBilinearForm;
|
||||
typedef SharedPtr<PBilinearForm> DBilinearForm;
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_SMART_POINTERS_HPP
|
||||
@@ -0,0 +1,52 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
#define MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/error.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace util
|
||||
{
|
||||
|
||||
//
|
||||
// Inline methods
|
||||
//
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename derived_t, typename base_t>
|
||||
inline derived_t *As(base_t *base_obj)
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<derived_t*>(base_obj) != NULL,
|
||||
"invalid object type");
|
||||
return static_cast<derived_t*>(base_obj);
|
||||
}
|
||||
|
||||
/// TODO: doxygen
|
||||
template <typename derived_t, typename base_t>
|
||||
inline derived_t *Is(base_t *base_obj)
|
||||
{
|
||||
return dynamic_cast<derived_t*>(base_obj);
|
||||
}
|
||||
|
||||
} // namespace mfem::util
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_UTILS_HPP
|
||||
@@ -0,0 +1,153 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
#define MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#ifdef MFEM_USE_BACKENDS
|
||||
|
||||
#include "../../general/scalars.hpp"
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
/// Polymorphic vector - array of scalars.
|
||||
class PVector : virtual public PArray
|
||||
{
|
||||
protected:
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new vector of the same dynamic type as this
|
||||
vector using the same layout with entries specified by @a buffer_type_id
|
||||
which should be a constant defined by the `value` field in a
|
||||
specialization of the template class mfem::ScalarId.
|
||||
|
||||
Returns NULL if allocation fails.
|
||||
|
||||
If @a copy_data is true, the contents of this vector is copied to the new
|
||||
vector; otherwise, the new vector remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the vector data of the newly created
|
||||
object (in @a *buffer), if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const = 0;
|
||||
|
||||
/** @brief Compute and return the dot product of @a *this and @a x. In the
|
||||
case of an MPI-parallel vector, the result must be the MPI-global dot
|
||||
product. */
|
||||
/** Both vectors must have the same dynamic type and layout. */
|
||||
virtual void DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const = 0;
|
||||
|
||||
// TODO: add reduction operations: min, max, sum
|
||||
|
||||
/// Perform the operation @a *this = @a a @a x + @a b @a y.
|
||||
/** Rules:
|
||||
- the dynamic type of both @a x and @a y is the same as that of @a *this
|
||||
- if @a a == 0, neither @a x nor its data are accessed
|
||||
- if @a b == 0, neither @a y nor its data are accessed
|
||||
- @a x's data is never the same as @a y's data, unless @a a == 0, or
|
||||
@a b == 0
|
||||
- @a x's data or @a y's data may be the same as the data of @a *this
|
||||
- all accessed vectors, @a x, @a y, and @a *this have the same layout. */
|
||||
virtual void DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id) = 0;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
/** @brief Create a PVector. */
|
||||
/** The @a layout must be valid in the sense that layout != NULL and
|
||||
layout->HasEngine() == true. */
|
||||
PVector(PLayout &p_layout)
|
||||
: PArray(p_layout) { }
|
||||
|
||||
template <typename derived_t>
|
||||
derived_t &As() { return *util::As<derived_t>(this); }
|
||||
|
||||
template <typename derived_t>
|
||||
const derived_t &As() const { return *util::As<const derived_t>(this); }
|
||||
|
||||
|
||||
// TODO: Error handling ... handle errors at the Engine level, at the class
|
||||
// level, or at the method level?
|
||||
|
||||
// TODO: Asynchronous execution interface ...
|
||||
|
||||
// TODO: Multi-vector interface ...
|
||||
|
||||
|
||||
/**
|
||||
@name Public virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/** @brief Create and return a new vector of the same dynamic type as this
|
||||
vector using the same layout with entries of type @a scalar_t.
|
||||
|
||||
If @a copy_data is true, the contents of this vector is copied to the new
|
||||
vector; otherwise, the new vector remains uninitialized.
|
||||
|
||||
If @a buffer is not NULL, return the vector data of the newly created
|
||||
object (in @a *buffer) , if it is stored as a contiguous array on the
|
||||
host; otherwise, set @a *buffer to NULL. */
|
||||
template <typename scalar_t>
|
||||
DVector Clone(bool copy_data, scalar_t **buffer) const
|
||||
{
|
||||
return DVector(DoVectorClone(copy_data, (void**)buffer,
|
||||
ScalarId<scalar_t>::value));
|
||||
}
|
||||
|
||||
/** @brief Compute and return the dot product of @a *this and @a x. In the
|
||||
case of an MPI-parallel vector, the result must be the MPI-global dot
|
||||
product. */
|
||||
/** Both vectors must have the same dynamic type and layout. */
|
||||
template <typename scalar_t>
|
||||
scalar_t DotProduct(const PVector &x) const
|
||||
{
|
||||
scalar_t result;
|
||||
DoDotProduct(x, &result, ScalarId<scalar_t>::value);
|
||||
return result;
|
||||
}
|
||||
|
||||
// TODO: add reduction operations: min, max, sum
|
||||
|
||||
/// Perform the operation @a *this = @a a @a x + @a b @a y.
|
||||
/** Rules:
|
||||
- the dynamic type of both @a x and @a y is the same as that of @a *this
|
||||
- if @a a == 0, neither @a x nor its data are accessed
|
||||
- if @a b == 0, neither @a y nor its data are accessed
|
||||
- @a x's data is never the same as @a y's data, unless @a a == 0, or
|
||||
@a b == 0
|
||||
- @a x's data or @a y's data may be the same as the data of @a *this
|
||||
- all accessed vectors, @a x, @a y, and @a *this have the same layout. */
|
||||
template <typename scalar_t>
|
||||
void Axpby(const scalar_t &a, const PVector &x,
|
||||
const scalar_t &b, const PVector &y)
|
||||
{ if (Size()) { DoAxpby(&a, x, &b, y, ScalarId<scalar_t>::value); } }
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // MFEM_USE_BACKENDS
|
||||
|
||||
#endif // MFEM_BACKENDS_BASE_VECTOR_HPP
|
||||
@@ -0,0 +1,66 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
ELEMENT_BATCH : How many elements are in each
|
||||
. computation batch
|
||||
NUM_DOFS_1D : Dofs in the 1D segments
|
||||
NUM_DOFS_2D : Dofs in the 2D faces
|
||||
NUM_DOFS_3D : Dofs in the 3D domain
|
||||
NUM_QUAD_1D : Dofs in the 1D segments
|
||||
NUM_QUAD_2D : Dofs in the 2D faces
|
||||
NUM_QUAD_3D : Dofs in the 3D domain
|
||||
NUM_MAX_1D : max(NUM_QUAD_1D, NUM_DOFS_1D)
|
||||
NUM_QUAD_DOFS_1D: NUM_QUAD_1D * NUM_DOFS_1D
|
||||
COEFF_ARGS : Code that passes required arguments to the kernel
|
||||
COEFF : Code that computes the coefficient
|
||||
================================================
|
||||
|
||||
[MISSING]
|
||||
- Add support to auto-pick @dim and use @idxOrder on stack arrays
|
||||
| double a[2][2];
|
||||
| a[0][1]; <-- regular index
|
||||
| a(0,1); <-- uses @idxOrder a[0][1] or a[1][0]
|
||||
- Add support for @idxOrder to change indexing order after allocation
|
||||
| double a[2][2] @idxOrder(0,1);
|
||||
| a(0,1) -> a[1][0]
|
||||
| @set(a, idxOrder(1,0));
|
||||
| a(0,1) -> a[0][1]
|
||||
- Add support to iterate over loop depending on mode
|
||||
| for(i; @inner) {
|
||||
| for(0 < j < N) {} <-- ++j or j += block?
|
||||
| }
|
||||
*/
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://diffusion/tensor/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://diffusion/tensor/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://diffusion/tensor/cpu.okl"
|
||||
# endif
|
||||
#else
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://diffusion/simplex/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://diffusion/simplex/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://diffusion/simplex/cpu.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -0,0 +1,66 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
ELEMENT_BATCH : How many elements are in each
|
||||
. computation batch
|
||||
NUM_DOFS_1D : Dofs in the 1D segments
|
||||
NUM_DOFS_2D : Dofs in the 2D faces
|
||||
NUM_DOFS_3D : Dofs in the 3D domain
|
||||
NUM_QUAD_1D : Dofs in the 1D segments
|
||||
NUM_QUAD_2D : Dofs in the 2D faces
|
||||
NUM_QUAD_3D : Dofs in the 3D domain
|
||||
NUM_MAX_1D : max(NUM_QUAD_1D, NUM_DOFS_1D)
|
||||
NUM_QUAD_DOFS_1D: NUM_QUAD_1D * NUM_DOFS_1D
|
||||
COEFF_ARGS : Code that passes required arguments to the kernel
|
||||
COEFF : Code that computes the coefficient
|
||||
================================================
|
||||
|
||||
[MISSING]
|
||||
- Add support to auto-pick @dim and use @idxOrder on stack arrays
|
||||
| double a[2][2];
|
||||
| a[0][1]; <-- regular index
|
||||
| a(0,1); <-- uses @idxOrder a[0][1] or a[1][0]
|
||||
- Add support for @idxOrder to change indexing order after allocation
|
||||
| double a[2][2] @idxOrder(0,1);
|
||||
| a(0,1) -> a[1][0]
|
||||
| @set(a, idxOrder(1,0));
|
||||
| a(0,1) -> a[0][1]
|
||||
- Add support to iterate over loop depending on mode
|
||||
| for(i; @inner) {
|
||||
| for(0 < j < N) {} <-- ++j or j += block?
|
||||
| }
|
||||
*/
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://mass/tensor/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://mass/tensor/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://mass/tensor/cpu.okl"
|
||||
# endif
|
||||
#else
|
||||
# ifdef OCCA_USING_GPU
|
||||
# if USING_LOW_ORDER
|
||||
# include "mfem-occa://mass/simplex/gpuHighOrder.okl"
|
||||
# else
|
||||
# include "mfem-occa://mass/simplex/gpuHighOrder.okl"
|
||||
# endif
|
||||
# else
|
||||
# include "mfem-occa://mass/simplex/cpu.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -0,0 +1,38 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
ELEMENT_BATCH : How many elements are in each
|
||||
. computation batch
|
||||
NUM_DOFS_1D : Dofs in the 1D segments
|
||||
NUM_DOFS_2D : Dofs in the 2D faces
|
||||
NUM_DOFS_3D : Dofs in the 3D domain
|
||||
NUM_QUAD_1D : Dofs in the 1D segments
|
||||
NUM_QUAD_2D : Dofs in the 2D faces
|
||||
NUM_QUAD_3D : Dofs in the 3D domain
|
||||
NUM_MAX_1D : max(NUM_QUAD_1D, NUM_DOFS_1D)
|
||||
NUM_QUAD_DOFS_1D: NUM_QUAD_1D * NUM_DOFS_1D
|
||||
CONST_COEFF : If the coefficient is constant, pass it
|
||||
. as a define
|
||||
================================================
|
||||
|
||||
[MISSING]
|
||||
See kernels/DiffusionIntegrator.okl
|
||||
*/
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# ifndef OCCA_USING_GPU
|
||||
# include "mfem-occa://vmass/tensor/cpu.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -0,0 +1,121 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
PArray *Array::DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
Array *new_array = new Array(OccaLayout(), item_size);
|
||||
if (copy_data)
|
||||
{
|
||||
new_array->slice.copyFrom(slice);
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
*buffer = new_array->GetBuffer();
|
||||
}
|
||||
return new_array;
|
||||
}
|
||||
|
||||
int Array::DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size)
|
||||
{
|
||||
MFEM_ASSERT(dynamic_cast<Layout *>(&new_layout) != NULL,
|
||||
"new_layout is not an OCCA Layout");
|
||||
Layout *lt = static_cast<Layout *>(&new_layout);
|
||||
int err = OccaResize(lt, item_size);
|
||||
if (!err && buffer)
|
||||
{
|
||||
*buffer = GetBuffer();
|
||||
}
|
||||
return err;
|
||||
}
|
||||
|
||||
void *Array::DoPullData(void *buffer, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
if (!slice.getDevice().hasSeparateMemorySpace())
|
||||
{
|
||||
return slice.ptr();
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
slice.copyTo(buffer);
|
||||
}
|
||||
return buffer;
|
||||
}
|
||||
|
||||
void Array::DoFill(const void *value_ptr, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
switch (item_size)
|
||||
{
|
||||
case sizeof(int8_t):
|
||||
OccaFill(*(const int8_t *)value_ptr);
|
||||
break;
|
||||
case sizeof(int16_t):
|
||||
OccaFill(*(const int16_t *)value_ptr);
|
||||
break;
|
||||
case sizeof(int32_t):
|
||||
OccaFill(*(const int32_t *)value_ptr);
|
||||
break;
|
||||
// case sizeof(int64_t):
|
||||
// OccaFill(*(const int64_t *)value_ptr);
|
||||
// break;
|
||||
case sizeof(double):
|
||||
OccaFill(*(const double *)value_ptr);
|
||||
break;
|
||||
// case sizeof(::occa::double2):
|
||||
// OccaFill(*(const ::occa::double2 *)value_ptr);
|
||||
// break;
|
||||
default:
|
||||
MFEM_ABORT("item_size = " << item_size << " is not supported");
|
||||
}
|
||||
}
|
||||
|
||||
void Array::DoPushData(const void *src_buffer, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
if (slice.getDevice().hasSeparateMemorySpace() || slice.ptr() != src_buffer)
|
||||
{
|
||||
slice.copyFrom(src_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
void Array::DoAssign(const PArray &src, std::size_t item_size)
|
||||
{
|
||||
// called only when Size() != 0
|
||||
|
||||
// Note: static_cast can not be used here since PArray is a virtual base
|
||||
// class.
|
||||
const Array *source = dynamic_cast<const Array *>(&src);
|
||||
MFEM_ASSERT(source != NULL, "invalid source Array type");
|
||||
OccaAssign(*source);
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,175 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
#define MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "layout.hpp"
|
||||
#include "../base/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Array : public virtual PArray
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// DLayout layout;
|
||||
|
||||
// Always true: Size()*item_size == slice.size() <= data.size()
|
||||
mutable ::occa::memory data, slice;
|
||||
|
||||
//
|
||||
// Virtual interface
|
||||
//
|
||||
|
||||
virtual PArray *DoClone(bool copy_data, void **buffer,
|
||||
std::size_t item_size) const;
|
||||
|
||||
virtual int DoResize(PLayout &new_layout, void **buffer,
|
||||
std::size_t item_size);
|
||||
|
||||
virtual void *DoPullData(void *buffer, std::size_t item_size);
|
||||
|
||||
virtual void DoFill(const void *value_ptr, std::size_t item_size);
|
||||
|
||||
virtual void DoPushData(const void *src_buffer, std::size_t item_size);
|
||||
|
||||
virtual void DoAssign(const PArray &src, std::size_t item_size);
|
||||
|
||||
//
|
||||
// Auxiliary methods
|
||||
//
|
||||
|
||||
inline void *GetBuffer() const;
|
||||
|
||||
public:
|
||||
Array(const Engine &e)
|
||||
: PArray(*(new Layout(e, 0))),
|
||||
data(e.Alloc(0)),
|
||||
slice(data)
|
||||
{ }
|
||||
|
||||
Array(Layout <, std::size_t item_size)
|
||||
: PArray(lt),
|
||||
data(lt.OccaEngine().Alloc(lt.Size()*item_size)),
|
||||
slice(data)
|
||||
{ }
|
||||
|
||||
virtual ~Array() { }
|
||||
|
||||
inline void MakeRef(Array &master);
|
||||
|
||||
Layout &OccaLayout() const { return layout->As<Layout>(); }
|
||||
|
||||
const Engine &OccaEngine() const { return OccaLayout().OccaEngine(); }
|
||||
|
||||
::occa::memory &OccaMem() { return slice; }
|
||||
const ::occa::memory &OccaMem() const { return slice; }
|
||||
|
||||
inline int OccaResize(Layout *lt, std::size_t item_size);
|
||||
|
||||
inline int OccaResize(std::size_t new_size, std::size_t item_size);
|
||||
|
||||
template <typename T>
|
||||
inline void OccaFill(const T val);
|
||||
|
||||
inline void OccaAssign(const Array &src);
|
||||
|
||||
inline void OccaPush(const void *src);
|
||||
};
|
||||
|
||||
|
||||
//
|
||||
// Inline methods
|
||||
//
|
||||
|
||||
inline void *Array::GetBuffer() const
|
||||
{
|
||||
if (!slice.getDevice().hasSeparateMemorySpace())
|
||||
{
|
||||
return slice.ptr();
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
inline int Array::OccaResize(Layout *lt, std::size_t item_size)
|
||||
{
|
||||
layout.Reset(lt); // Reset() checks if the pointer is the same
|
||||
const std::size_t new_bytes = lt->Size()*item_size;
|
||||
if (data.size() < new_bytes ||
|
||||
data.getDevice() != lt->OccaEngine().GetDevice())
|
||||
{
|
||||
data = lt->OccaEngine().Alloc(new_bytes);
|
||||
slice = data;
|
||||
// If memory allocation fails - an exception is thrown.
|
||||
}
|
||||
else if (slice.size() != new_bytes)
|
||||
{
|
||||
slice = data.slice(0, new_bytes);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
inline void Array::MakeRef(Array &master)
|
||||
{
|
||||
layout = master.layout;
|
||||
data = master.data;
|
||||
slice = master.slice;
|
||||
}
|
||||
|
||||
inline int Array::OccaResize(std::size_t new_size, std::size_t item_size)
|
||||
{
|
||||
Layout &ol = OccaLayout();
|
||||
ol.OccaResize(new_size);
|
||||
return OccaResize(&ol, item_size);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
inline void Array::OccaFill(const T val)
|
||||
{
|
||||
::occa::linalg::operator_eq<T>(slice, val);
|
||||
}
|
||||
|
||||
inline void Array::OccaAssign(const Array &src)
|
||||
{
|
||||
if (slice != src.slice && slice.size() != 0)
|
||||
{
|
||||
MFEM_ASSERT(slice.size() == src.slice.size(), "");
|
||||
slice.copyFrom(src.slice);
|
||||
}
|
||||
}
|
||||
|
||||
inline void Array::OccaPush(const void *src)
|
||||
{
|
||||
if (slice.size() != 0)
|
||||
{
|
||||
slice.copyFrom(src);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_ARRAY_HPP
|
||||
@@ -0,0 +1,47 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
bool Backend::Supports(const std::string &engine_spec) const
|
||||
{
|
||||
// TODO: check if 'engine_spec' is valid OCCA string.
|
||||
return true;
|
||||
}
|
||||
|
||||
mfem::Engine *Create(const std::string &engine_spec)
|
||||
{
|
||||
return new Engine(engine_spec);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec)
|
||||
{
|
||||
return new Engine(comm, engine_spec);
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,49 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
// Only the Backend and Engine classes should be exposed through "backend.hpp"
|
||||
#include "../base/backend.hpp"
|
||||
#include "engine.hpp"
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Backend : public mfem::Backend
|
||||
{
|
||||
public:
|
||||
virtual ~Backend();
|
||||
|
||||
virtual bool Supports(const std::string &engine_spec) const;
|
||||
|
||||
virtual mfem::Engine *Create(const std::string &engine_spec);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
virtual mfem::Engine *Create(MPI_Comm comm, const std::string &engine_spec);
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BACKEND_HPP
|
||||
@@ -0,0 +1,538 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/bilinearform.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
OccaBilinearForm::OccaBilinearForm(FiniteElementSpace *ofespace_) :
|
||||
Operator(ofespace_->OccaVLayout()),
|
||||
localX(ofespace_->OccaEVLayout()),
|
||||
localY(ofespace_->OccaEVLayout())
|
||||
{
|
||||
Init(ofespace_->OccaEngine(), ofespace_, ofespace_);
|
||||
}
|
||||
|
||||
OccaBilinearForm::OccaBilinearForm(FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_) :
|
||||
Operator(otrialFESpace_->OccaVLayout(),
|
||||
otestFESpace_->OccaVLayout()),
|
||||
localX(otrialFESpace_->OccaEVLayout()),
|
||||
localY(otestFESpace_->OccaEVLayout())
|
||||
{
|
||||
Init(otrialFESpace_->OccaEngine(), otrialFESpace_, otestFESpace_);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::Init(const Engine &e,
|
||||
FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_)
|
||||
{
|
||||
engine.Reset(&e);
|
||||
|
||||
otrialFESpace = otrialFESpace_;
|
||||
trialFESpace = otrialFESpace_->GetFESpace();
|
||||
|
||||
otestFESpace = otestFESpace_;
|
||||
testFESpace = otestFESpace_->GetFESpace();
|
||||
|
||||
mesh = trialFESpace->GetMesh();
|
||||
|
||||
const int elements = GetNE();
|
||||
|
||||
const int trialVDim = trialFESpace->GetVDim();
|
||||
|
||||
const int trialLocalDofs = otrialFESpace->GetLocalDofs();
|
||||
const int testLocalDofs = otestFESpace->GetLocalDofs();
|
||||
|
||||
// First-touch policy when running with OpenMP
|
||||
if (GetDevice().mode() == "OpenMP")
|
||||
{
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
::occa::kernel initLocalKernel =
|
||||
GetDevice().buildKernel(okl_path + "utils.okl",
|
||||
"InitLocalVector");
|
||||
|
||||
const std::size_t sd = sizeof(double);
|
||||
const uint64_t trialEntries = sd * (elements * trialLocalDofs);
|
||||
const uint64_t testEntries = sd * (elements * testLocalDofs);
|
||||
for (int v = 0; v < trialVDim; ++v)
|
||||
{
|
||||
const uint64_t trialOffset = v * trialEntries;
|
||||
const uint64_t testOffset = v * testEntries;
|
||||
|
||||
initLocalKernel(elements, trialLocalDofs,
|
||||
localX.OccaMem().slice(trialOffset, trialEntries));
|
||||
initLocalKernel(elements, testLocalDofs,
|
||||
localY.OccaMem().slice(testOffset, testEntries));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int OccaBilinearForm::BaseGeom() const
|
||||
{
|
||||
return mesh->GetElementBaseGeometry();
|
||||
}
|
||||
|
||||
int OccaBilinearForm::GetDim() const
|
||||
{
|
||||
return mesh->Dimension();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetNE() const
|
||||
{
|
||||
return mesh->GetNE();
|
||||
}
|
||||
|
||||
Mesh& OccaBilinearForm::GetMesh() const
|
||||
{
|
||||
return *mesh;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaBilinearForm::GetTrialOccaFESpace() const
|
||||
{
|
||||
return *otrialFESpace;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaBilinearForm::GetTestOccaFESpace() const
|
||||
{
|
||||
return *otestFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaBilinearForm::GetTrialFESpace() const
|
||||
{
|
||||
return *trialFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaBilinearForm::GetTestFESpace() const
|
||||
{
|
||||
return *testFESpace;
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTrialNDofs() const
|
||||
{
|
||||
return trialFESpace->GetNDofs();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTestNDofs() const
|
||||
{
|
||||
return testFESpace->GetNDofs();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTrialVDim() const
|
||||
{
|
||||
return trialFESpace->GetVDim();
|
||||
}
|
||||
|
||||
int64_t OccaBilinearForm::GetTestVDim() const
|
||||
{
|
||||
return testFESpace->GetVDim();
|
||||
}
|
||||
|
||||
const FiniteElement& OccaBilinearForm::GetTrialFE(const int i) const
|
||||
{
|
||||
return *(trialFESpace->GetFE(i));
|
||||
}
|
||||
|
||||
const FiniteElement& OccaBilinearForm::GetTestFE(const int i) const
|
||||
{
|
||||
return *(testFESpace->GetFE(i));
|
||||
}
|
||||
|
||||
// Adds new Domain Integrator.
|
||||
void OccaBilinearForm::AddDomainIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, DomainIntegrator);
|
||||
}
|
||||
|
||||
// Adds new Boundary Integrator.
|
||||
void OccaBilinearForm::AddBoundaryIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, BoundaryIntegrator);
|
||||
}
|
||||
|
||||
// Adds new interior Face Integrator.
|
||||
void OccaBilinearForm::AddInteriorFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, InteriorFaceIntegrator);
|
||||
}
|
||||
|
||||
// Adds new boundary Face Integrator.
|
||||
void OccaBilinearForm::AddBoundaryFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
AddIntegrator(integrator, props, BoundaryFaceIntegrator);
|
||||
}
|
||||
|
||||
// Adds Integrator based on OccaIntegratorType
|
||||
void OccaBilinearForm::AddIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props,
|
||||
const OccaIntegratorType itype)
|
||||
{
|
||||
if (integrator == NULL)
|
||||
{
|
||||
std::stringstream error_ss;
|
||||
error_ss << "OccaBilinearForm::";
|
||||
switch (itype)
|
||||
{
|
||||
case DomainIntegrator : error_ss << "AddDomainIntegrator"; break;
|
||||
case BoundaryIntegrator : error_ss << "AddBoundaryIntegrator"; break;
|
||||
case InteriorFaceIntegrator: error_ss << "AddInteriorFaceIntegrator"; break;
|
||||
case BoundaryFaceIntegrator: error_ss << "AddBoundaryFaceIntegrator"; break;
|
||||
}
|
||||
error_ss << " (...):\n"
|
||||
<< " Integrator is NULL";
|
||||
const std::string error = error_ss.str();
|
||||
mfem_error(error.c_str());
|
||||
}
|
||||
integrator->SetupIntegrator(*this, baseKernelProps + props, itype);
|
||||
integrators.push_back(integrator);
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTrialProlongation() const
|
||||
{
|
||||
return otrialFESpace->GetProlongationOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTestProlongation() const
|
||||
{
|
||||
return otestFESpace->GetProlongationOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTrialRestriction() const
|
||||
{
|
||||
return otrialFESpace->GetRestrictionOperator();
|
||||
}
|
||||
|
||||
const mfem::Operator* OccaBilinearForm::GetTestRestriction() const
|
||||
{
|
||||
return otestFESpace->GetRestrictionOperator();
|
||||
}
|
||||
|
||||
void OccaBilinearForm::Assemble()
|
||||
{
|
||||
// [MISSING] Find geometric information that is needed by intergrators
|
||||
// to share between integrators.
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->Assemble();
|
||||
}
|
||||
}
|
||||
|
||||
void OccaBilinearForm::FormLinearSystem(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *&Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
FormOperator(constraintList, Aout);
|
||||
InitRHS(constraintList, x, b, Aout, X, B, copy_interior);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::FormOperator(const mfem::Array<int> &constraintList,
|
||||
mfem::Operator *&Aout)
|
||||
{
|
||||
const mfem::Operator *trialP = GetTrialProlongation();
|
||||
const mfem::Operator *testP = GetTestProlongation();
|
||||
mfem::Operator *rap = this;
|
||||
|
||||
if (trialP)
|
||||
{
|
||||
rap = new RAPOperator(*testP, *this, *trialP);
|
||||
}
|
||||
|
||||
Aout = new OccaConstrainedOperator(rap, constraintList,
|
||||
rap != this);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::InitRHS(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
// FIXME: move these kernels to the Backend?
|
||||
static ::occa::kernelBuilder get_subvector_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_get_subvector",
|
||||
|
||||
"const int dof_i = v2[i];"
|
||||
"v0[i] = dof_i >= 0 ? v1[dof_i] : -v1[-dof_i - 1];",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
static ::occa::kernelBuilder set_subvector_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_set_subvector",
|
||||
"const int dof_i = v2[i];"
|
||||
"if (dof_i >= 0) { v0[dof_i] = v1[i]; }"
|
||||
"else { v0[-dof_i - 1] = -v1[i]; }",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
const mfem::Operator *P = GetTrialProlongation();
|
||||
const mfem::Operator *R = GetTrialRestriction();
|
||||
|
||||
if (P)
|
||||
{
|
||||
// Variational restriction with P
|
||||
B.Resize(P->InLayout());
|
||||
P->MultTranspose(b, B);
|
||||
X.Resize(R->OutLayout());
|
||||
R->Mult(x, X);
|
||||
}
|
||||
else
|
||||
{
|
||||
// rap, X and B point to the same data as this, x and b
|
||||
X.MakeRef(x);
|
||||
B.MakeRef(b);
|
||||
}
|
||||
|
||||
if (!copy_interior && constraintList.Size() > 0)
|
||||
{
|
||||
::occa::kernel get_subvector_kernel =
|
||||
get_subvector_builder.build(GetDevice());
|
||||
::occa::kernel set_subvector_kernel =
|
||||
set_subvector_builder.build(GetDevice());
|
||||
|
||||
const Array &constrList = constraintList.Get_PArray()->As<Array>();
|
||||
Vector subvec(constrList.OccaLayout());
|
||||
|
||||
get_subvector_kernel(constraintList.Size(),
|
||||
subvec.OccaMem(),
|
||||
X.Get_PVector()->As<Vector>().OccaMem(),
|
||||
constrList.OccaMem());
|
||||
|
||||
X.Fill(0.0);
|
||||
|
||||
set_subvector_kernel(constraintList.Size(),
|
||||
X.Get_PVector()->As<Vector>().OccaMem(),
|
||||
subvec.OccaMem(),
|
||||
constrList.OccaMem());
|
||||
}
|
||||
|
||||
// FIXME: add case for HypreParMatrix here
|
||||
OccaConstrainedOperator *cA = dynamic_cast<OccaConstrainedOperator*>(A);
|
||||
if (cA)
|
||||
{
|
||||
cA->EliminateRHS(X.Get_PVector()->As<Vector>(),
|
||||
B.Get_PVector()->As<Vector>());
|
||||
}
|
||||
else
|
||||
{
|
||||
mfem_error("OccaBilinearForm::InitRHS expects an OccaConstrainedOperator");
|
||||
}
|
||||
}
|
||||
|
||||
// Matrix vector multiplication.
|
||||
void OccaBilinearForm::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
otrialFESpace->GlobalToLocal(x, localX);
|
||||
localY.OccaFill<double>(0.0);
|
||||
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->MultAdd(localX, localY);
|
||||
}
|
||||
|
||||
otestFESpace->LocalToGlobal(localY, y);
|
||||
}
|
||||
|
||||
// Matrix transpose vector multiplication.
|
||||
void OccaBilinearForm::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
otestFESpace->GlobalToLocal(x, localX);
|
||||
localY.OccaFill<double>(0.0);
|
||||
|
||||
const int integratorCount = (int) integrators.size();
|
||||
for (int i = 0; i < integratorCount; ++i)
|
||||
{
|
||||
integrators[i]->MultTransposeAdd(localX, localY);
|
||||
}
|
||||
|
||||
otrialFESpace->LocalToGlobal(localY, y);
|
||||
}
|
||||
|
||||
void OccaBilinearForm::OccaRecoverFEMSolution(const mfem::Vector &X,
|
||||
const mfem::Vector &b,
|
||||
mfem::Vector &x)
|
||||
{
|
||||
const mfem::Operator *P = this->GetTrialProlongation();
|
||||
if (P)
|
||||
{
|
||||
// Apply conforming prolongation
|
||||
x.Resize(P->OutLayout());
|
||||
P->Mult(X, x);
|
||||
}
|
||||
// Otherwise X and x point to the same data
|
||||
}
|
||||
|
||||
// Frees memory bilinear form.
|
||||
OccaBilinearForm::~OccaBilinearForm()
|
||||
{
|
||||
// Make sure all integrators free their data
|
||||
IntegratorVector::iterator it = integrators.begin();
|
||||
while (it != integrators.end())
|
||||
{
|
||||
delete *it;
|
||||
++it;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void BilinearForm::InitOccaBilinearForm()
|
||||
{
|
||||
// Init 'obform' using 'bform'
|
||||
MFEM_ASSERT(bform != NULL, "");
|
||||
MFEM_ASSERT(obform == NULL, "");
|
||||
|
||||
FiniteElementSpace &ofes =
|
||||
bform->FESpace()->Get_PFESpace()->As<FiniteElementSpace>();
|
||||
obform = new OccaBilinearForm(&ofes);
|
||||
|
||||
// Transfer domain integrators
|
||||
mfem::Array<mfem::BilinearFormIntegrator*> &dbfi = *bform->GetDBFI();
|
||||
for (int i = 0; i < dbfi.Size(); i++)
|
||||
{
|
||||
std::string integ_name(dbfi[i]->Name());
|
||||
Coefficient *scal_coeff = dbfi[i]->GetScalarCoefficient();
|
||||
ConstantCoefficient *const_coeff =
|
||||
dynamic_cast<ConstantCoefficient*>(scal_coeff);
|
||||
GridFunctionCoefficient *gridfunc_coeff =
|
||||
dynamic_cast<GridFunctionCoefficient*>(scal_coeff);
|
||||
// TODO: other types of coefficients ...
|
||||
|
||||
OccaCoefficient *ocoeff = NULL;
|
||||
if (const_coeff)
|
||||
{
|
||||
ocoeff = new OccaCoefficient(obform->OccaEngine(),
|
||||
const_coeff->constant);
|
||||
}
|
||||
else if (gridfunc_coeff)
|
||||
{
|
||||
ocoeff = new OccaCoefficient(obform->OccaEngine(),
|
||||
*gridfunc_coeff->GetGridFunction(), true);
|
||||
}
|
||||
else if (!scal_coeff)
|
||||
{
|
||||
ocoeff = new OccaCoefficient(obform->OccaEngine(), 1.0);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Coefficient type not supported");
|
||||
}
|
||||
|
||||
OccaIntegrator *ointeg = NULL;
|
||||
if (integ_name == "(undefined)")
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator does not define Name()");
|
||||
}
|
||||
else if (integ_name == "mass")
|
||||
{
|
||||
ointeg = new OccaMassIntegrator(*ocoeff);
|
||||
}
|
||||
else if (integ_name == "diffusion")
|
||||
{
|
||||
ointeg = new OccaDiffusionIntegrator(*ocoeff);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("BilinearFormIntegrator [Name() = " << integ_name
|
||||
<< "] is not supported");
|
||||
}
|
||||
|
||||
// NOTE: The integrators copy ocoeff, so it can be deleted here so there
|
||||
// is no memory leak.
|
||||
delete ocoeff;
|
||||
|
||||
const mfem::IntegrationRule *ir = dbfi[i]->GetIntRule();
|
||||
if (ir) { ointeg->SetIntegrationRule(*ir); }
|
||||
|
||||
obform->AddDomainIntegrator(ointeg);
|
||||
}
|
||||
|
||||
// TODO: other types of integrators ...
|
||||
}
|
||||
|
||||
bool BilinearForm::Assemble()
|
||||
{
|
||||
if (obform == NULL) { InitOccaBilinearForm(); }
|
||||
|
||||
obform->Assemble();
|
||||
|
||||
return true; // --> host assembly is not needed
|
||||
}
|
||||
|
||||
void BilinearForm::FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::OperatorHandle &A)
|
||||
{
|
||||
if (A.Type() == mfem::Operator::ANY_TYPE)
|
||||
{
|
||||
mfem::Operator *Aout = NULL;
|
||||
obform->FormOperator(ess_tdof_list, Aout);
|
||||
A.Reset(Aout);
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ABORT("Operator::Type is not supported, type = " << A.Type());
|
||||
}
|
||||
}
|
||||
|
||||
void BilinearForm::FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::OperatorHandle &A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior)
|
||||
{
|
||||
FormSystemMatrix(ess_tdof_list, A);
|
||||
obform->InitRHS(ess_tdof_list, x, b, A.Ptr(), X, B, copy_interior);
|
||||
}
|
||||
|
||||
void BilinearForm::RecoverFEMSolution(const mfem::Vector &X,
|
||||
const mfem::Vector &b,
|
||||
mfem::Vector &x)
|
||||
{
|
||||
obform->OccaRecoverFEMSolution(X, b, x);
|
||||
}
|
||||
|
||||
BilinearForm::~BilinearForm()
|
||||
{
|
||||
delete obform;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,213 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "fespace.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
enum OccaIntegratorType
|
||||
{
|
||||
DomainIntegrator = 0,
|
||||
BoundaryIntegrator = 1,
|
||||
InteriorFaceIntegrator = 2,
|
||||
BoundaryFaceIntegrator = 3
|
||||
};
|
||||
|
||||
class OccaIntegrator;
|
||||
|
||||
|
||||
/** Class for bilinear form - "Matrix" with associated FE space and
|
||||
BLFIntegrators. */
|
||||
class OccaBilinearForm : public Operator
|
||||
{
|
||||
friend class OccaIntegrator;
|
||||
|
||||
protected:
|
||||
typedef std::vector<OccaIntegrator*> IntegratorVector;
|
||||
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
// State information
|
||||
mutable mfem::Mesh *mesh;
|
||||
|
||||
mutable FiniteElementSpace *otrialFESpace;
|
||||
mutable mfem::FiniteElementSpace *trialFESpace;
|
||||
|
||||
mutable FiniteElementSpace *otestFESpace;
|
||||
mutable mfem::FiniteElementSpace *testFESpace;
|
||||
|
||||
IntegratorVector integrators;
|
||||
|
||||
// Device data
|
||||
::occa::properties baseKernelProps;
|
||||
|
||||
// The input and output vectors are mapped to local nodes for efficient
|
||||
// operations. In other words, they are E-vectors.
|
||||
// The size is: (number of elements) * (nodes in element) * (vector dim)
|
||||
mutable Vector localX, localY;
|
||||
|
||||
public:
|
||||
OccaBilinearForm(FiniteElementSpace *ofespace_);
|
||||
|
||||
OccaBilinearForm(FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_);
|
||||
|
||||
void Init(const Engine &e,
|
||||
FiniteElementSpace *otrialFESpace_,
|
||||
FiniteElementSpace *otestFESpace_);
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
// Useful mesh Information
|
||||
int BaseGeom() const;
|
||||
int GetDim() const;
|
||||
int64_t GetNE() const;
|
||||
|
||||
mfem::Mesh& GetMesh() const;
|
||||
|
||||
FiniteElementSpace& GetTrialOccaFESpace() const;
|
||||
FiniteElementSpace& GetTestOccaFESpace() const;
|
||||
|
||||
mfem::FiniteElementSpace& GetTrialFESpace() const;
|
||||
mfem::FiniteElementSpace& GetTestFESpace() const;
|
||||
|
||||
// Useful FE information
|
||||
int64_t GetTrialNDofs() const;
|
||||
int64_t GetTestNDofs() const;
|
||||
|
||||
int64_t GetTrialVDim() const;
|
||||
int64_t GetTestVDim() const;
|
||||
|
||||
const mfem::FiniteElement& GetTrialFE(const int i) const;
|
||||
const mfem::FiniteElement& GetTestFE(const int i) const;
|
||||
|
||||
// Adds new Domain Integrator.
|
||||
void AddDomainIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new Boundary Integrator.
|
||||
void AddBoundaryIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new interior Face Integrator.
|
||||
void AddInteriorFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds new boundary Face Integrator.
|
||||
void AddBoundaryFaceIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props =
|
||||
::occa::properties());
|
||||
|
||||
// Adds Integrator based on OccaIntegratorType
|
||||
void AddIntegrator(OccaIntegrator *integrator,
|
||||
const ::occa::properties &props,
|
||||
const OccaIntegratorType itype);
|
||||
|
||||
virtual const mfem::Operator *GetTrialProlongation() const;
|
||||
virtual const mfem::Operator *GetTestProlongation() const;
|
||||
|
||||
virtual const mfem::Operator *GetTrialRestriction() const;
|
||||
virtual const mfem::Operator *GetTestRestriction() const;
|
||||
|
||||
// Assembles the form i.e. sums over all domain/bdr integrators.
|
||||
virtual void Assemble();
|
||||
|
||||
void FormLinearSystem(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *&Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior = 0);
|
||||
|
||||
void FormOperator(const mfem::Array<int> &constraintList,
|
||||
mfem::Operator *&Aout);
|
||||
|
||||
void InitRHS(const mfem::Array<int> &constraintList,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::Operator *Aout,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior = 0);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
|
||||
void OccaRecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
|
||||
mfem::Vector &x);
|
||||
|
||||
// Destroys bilinear form.
|
||||
~OccaBilinearForm();
|
||||
};
|
||||
|
||||
|
||||
/// TODO: doxygen
|
||||
class BilinearForm : public mfem::PBilinearForm
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// mfem::BilinearForm *bform;
|
||||
OccaBilinearForm *obform;
|
||||
|
||||
// Called from Assemble() if obform is NULL to initialize obform.
|
||||
void InitOccaBilinearForm();
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
BilinearForm(const Engine &e, mfem::BilinearForm &bf)
|
||||
: mfem::PBilinearForm(e, bf), obform(NULL) { }
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~BilinearForm();
|
||||
|
||||
/// Assemble the PBilinearForm.
|
||||
/** This method is called from the method mfem::BilinearForm::Assemble() of
|
||||
the associated mfem::BilinearForm, #bform.
|
||||
@returns True, if the host assembly should NOT be performed. */
|
||||
virtual bool Assemble();
|
||||
|
||||
virtual void FormSystemMatrix(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::OperatorHandle &A);
|
||||
|
||||
virtual void FormLinearSystem(const mfem::Array<int> &ess_tdof_list,
|
||||
mfem::Vector &x, mfem::Vector &b,
|
||||
mfem::OperatorHandle &A,
|
||||
mfem::Vector &X, mfem::Vector &B,
|
||||
int copy_interior);
|
||||
|
||||
virtual void RecoverFEMSolution(const mfem::Vector &X, const mfem::Vector &b,
|
||||
mfem::Vector &x);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BILINEAR_FORM_HPP
|
||||
@@ -0,0 +1,954 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
std::map<std::string, OccaDofQuadMaps> OccaDofQuadMaps::AllDofQuadMaps;
|
||||
|
||||
OccaGeometry OccaGeometry::Get(::occa::device device,
|
||||
FiniteElementSpace &ofespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const int flags)
|
||||
{
|
||||
OccaGeometry geom;
|
||||
|
||||
mfem::Mesh &mesh = *(ofespace.GetMesh());
|
||||
if (!mesh.GetNodes())
|
||||
{
|
||||
mesh.SetCurvature(1, false, -1, mfem::Ordering::byVDIM);
|
||||
}
|
||||
mfem::GridFunction &nodes = *(mesh.GetNodes());
|
||||
const mfem::FiniteElementSpace &fespace = *(nodes.FESpace());
|
||||
const mfem::FiniteElement &fe = *(fespace.GetFE(0));
|
||||
|
||||
const int dims = fe.GetDim();
|
||||
const int elements = fespace.GetNE();
|
||||
const int numDofs = fe.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
MFEM_ASSERT(dims == mesh.SpaceDimension(), "");
|
||||
|
||||
geom.meshNodes.allocate(device,
|
||||
dims, numDofs, elements);
|
||||
|
||||
const mfem::Table &e2dTable = fespace.GetElementToDofTable();
|
||||
const int *elementMap = e2dTable.GetJ();
|
||||
nodes.Pull();
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
for (int dof = 0; dof < numDofs; ++dof)
|
||||
{
|
||||
const int gid = elementMap[dof + numDofs*e];
|
||||
for (int dim = 0; dim < dims; ++dim)
|
||||
{
|
||||
geom.meshNodes(dim, dof, e) = nodes[fespace.DofToVDof(gid,dim)];
|
||||
}
|
||||
}
|
||||
}
|
||||
geom.meshNodes.keepInDevice();
|
||||
|
||||
if (flags & Jacobian)
|
||||
{
|
||||
geom.J.allocate(device,
|
||||
dims, dims, numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.J.allocate(device, 1);
|
||||
}
|
||||
if (flags & JacobianInv)
|
||||
{
|
||||
geom.invJ.allocate(device,
|
||||
dims, dims, numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.invJ.allocate(device, 1);
|
||||
}
|
||||
if (flags & JacobianDet)
|
||||
{
|
||||
geom.detJ.allocate(device,
|
||||
numQuad, elements);
|
||||
}
|
||||
else
|
||||
{
|
||||
geom.detJ.allocate(device, 1);
|
||||
}
|
||||
|
||||
geom.J.stopManaging();
|
||||
geom.invJ.stopManaging();
|
||||
geom.detJ.stopManaging();
|
||||
|
||||
OccaDofQuadMaps &maps = OccaDofQuadMaps::GetSimplexMaps(device, fe, ir);
|
||||
|
||||
::occa::properties props;
|
||||
props["defines/NUM_DOFS"] = numDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
props["defines/STORE_JACOBIAN"] = (flags & Jacobian);
|
||||
props["defines/STORE_JACOBIAN_INV"] = (flags & JacobianInv);
|
||||
props["defines/STORE_JACOBIAN_DET"] = (flags & JacobianDet);
|
||||
|
||||
const std::string &okl_path = ofespace.OccaEngine().GetOklPath();
|
||||
::occa::kernel init = device.buildKernel(okl_path + "geometry.okl",
|
||||
stringWithDim("InitGeometryInfo",
|
||||
fe.GetDim()),
|
||||
props);
|
||||
init(elements,
|
||||
maps.dofToQuadD,
|
||||
geom.meshNodes,
|
||||
geom.J, geom.invJ, geom.detJ);
|
||||
|
||||
return geom;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps::OccaDofQuadMaps() :
|
||||
hash() {}
|
||||
|
||||
OccaDofQuadMaps::OccaDofQuadMaps(const OccaDofQuadMaps &maps)
|
||||
{
|
||||
*this = maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::operator = (const OccaDofQuadMaps &maps)
|
||||
{
|
||||
hash = maps.hash;
|
||||
dofToQuad = maps.dofToQuad;
|
||||
dofToQuadD = maps.dofToQuadD;
|
||||
quadToDof = maps.quadToDof;
|
||||
quadToDofD = maps.quadToDofD;
|
||||
quadWeights = maps.quadWeights;
|
||||
return *this;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device,
|
||||
*fespace.GetFE(0),
|
||||
*fespace.GetFE(0),
|
||||
ir,
|
||||
transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device, fe, fe, ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const FiniteElementSpace &trialFESpace,
|
||||
const FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return Get(device,
|
||||
*trialFESpace.GetFE(0),
|
||||
*testFESpace.GetFE(0),
|
||||
ir,
|
||||
transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::Get(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return (dynamic_cast<const mfem::TensorBasisElement*>(&trialFE)
|
||||
? GetTensorMaps(device, trialFE, testFE, ir, transpose)
|
||||
: GetSimplexMaps(device, trialFE, testFE, ir, transpose));
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return GetTensorMaps(device,
|
||||
fe, fe,
|
||||
ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const mfem::TensorBasisElement &trialTFE =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(trialFE);
|
||||
const mfem::TensorBasisElement &testTFE =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(testFE);
|
||||
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "Tensor"
|
||||
<< "O1:" << trialFE.GetOrder()
|
||||
<< "O2:" << testFE.GetOrder()
|
||||
<< "BT1:" << trialTFE.GetBasisType()
|
||||
<< "BT2:" << testTFE.GetBasisType()
|
||||
<< "Q:" << ir.GetNPoints();
|
||||
std::string hash = ss.str();
|
||||
|
||||
// If we've already made the dof-quad maps, reuse them
|
||||
OccaDofQuadMaps &maps = AllDofQuadMaps[hash];
|
||||
if (!maps.hash.size())
|
||||
{
|
||||
// Create the dof-quad maps
|
||||
maps.hash = hash;
|
||||
|
||||
OccaDofQuadMaps trialMaps = GetD2QTensorMaps(device, trialFE, ir);
|
||||
OccaDofQuadMaps testMaps = GetD2QTensorMaps(device, testFE , ir, true);
|
||||
|
||||
maps.dofToQuad = trialMaps.dofToQuad;
|
||||
maps.dofToQuadD = trialMaps.dofToQuadD;
|
||||
maps.quadToDof = testMaps.dofToQuad;
|
||||
maps.quadToDofD = testMaps.dofToQuadD;
|
||||
maps.quadWeights = testMaps.quadWeights;
|
||||
}
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps OccaDofQuadMaps::GetD2QTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const mfem::TensorBasisElement &tfe =
|
||||
dynamic_cast<const mfem::TensorBasisElement&>(fe);
|
||||
|
||||
const mfem::Poly_1D::Basis &basis = tfe.GetBasis1D();
|
||||
const int order = fe.GetOrder();
|
||||
// [MISSING] Get 1D dofs
|
||||
const int dofs = order + 1;
|
||||
const int dims = fe.GetDim();
|
||||
|
||||
// Create the dof -> quadrature point map
|
||||
const mfem::IntegrationRule &ir1D =
|
||||
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir.GetOrder());
|
||||
const int quadPoints = ir1D.GetNPoints();
|
||||
const int quadPoints2D = quadPoints*quadPoints;
|
||||
const int quadPoints3D = quadPoints2D*quadPoints;
|
||||
const int quadPointsND = ((dims == 1) ? quadPoints :
|
||||
((dims == 2) ? quadPoints2D : quadPoints3D));
|
||||
|
||||
OccaDofQuadMaps maps;
|
||||
// Initialize the dof -> quad mapping
|
||||
maps.dofToQuad.allocate(device,
|
||||
quadPoints, dofs);
|
||||
maps.dofToQuadD.allocate(device,
|
||||
quadPoints, dofs);
|
||||
|
||||
double *quadWeights1DData = NULL;
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
maps.dofToQuad.reindex(1,0);
|
||||
maps.dofToQuadD.reindex(1,0);
|
||||
// Initialize quad weights only for transpose
|
||||
maps.quadWeights.allocate(device,
|
||||
quadPointsND);
|
||||
quadWeights1DData = new double[quadPoints];
|
||||
}
|
||||
|
||||
mfem::Vector d2q(dofs);
|
||||
mfem::Vector d2qD(dofs);
|
||||
for (int q = 0; q < quadPoints; ++q)
|
||||
{
|
||||
const mfem::IntegrationPoint &ip = ir1D.IntPoint(q);
|
||||
basis.Eval(ip.x, d2q, d2qD);
|
||||
if (transpose)
|
||||
{
|
||||
quadWeights1DData[q] = ip.weight;
|
||||
}
|
||||
for (int d = 0; d < dofs; ++d)
|
||||
{
|
||||
maps.dofToQuad(q, d) = d2q[d];
|
||||
maps.dofToQuadD(q, d) = d2qD[d];
|
||||
}
|
||||
}
|
||||
|
||||
maps.dofToQuad.keepInDevice();
|
||||
maps.dofToQuadD.keepInDevice();
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
for (int q = 0; q < quadPointsND; ++q)
|
||||
{
|
||||
const int qx = q % quadPoints;
|
||||
const int qz = q / quadPoints2D;
|
||||
const int qy = (q - qz*quadPoints2D) / quadPoints;
|
||||
double w = quadWeights1DData[qx];
|
||||
if (dims > 1)
|
||||
{
|
||||
w *= quadWeights1DData[qy];
|
||||
}
|
||||
if (dims > 2)
|
||||
{
|
||||
w *= quadWeights1DData[qz];
|
||||
}
|
||||
maps.quadWeights[q] = w;
|
||||
}
|
||||
maps.quadWeights.keepInDevice();
|
||||
delete [] quadWeights1DData;
|
||||
}
|
||||
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
return GetSimplexMaps(device,
|
||||
fe, fe,
|
||||
ir, transpose);
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaDofQuadMaps::GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "Simplex"
|
||||
<< "O1:" << trialFE.GetOrder()
|
||||
<< "O2:" << testFE.GetOrder()
|
||||
<< "Q:" << ir.GetNPoints();
|
||||
std::string hash = ss.str();
|
||||
|
||||
// If we've already made the dof-quad maps, reuse them
|
||||
OccaDofQuadMaps &maps = AllDofQuadMaps[hash];
|
||||
if (!maps.hash.size())
|
||||
{
|
||||
// Create the dof-quad maps
|
||||
maps.hash = hash;
|
||||
|
||||
OccaDofQuadMaps trialMaps = GetD2QSimplexMaps(device, trialFE, ir);
|
||||
OccaDofQuadMaps testMaps = GetD2QSimplexMaps(device, testFE , ir, true);
|
||||
|
||||
maps.dofToQuad = trialMaps.dofToQuad;
|
||||
maps.dofToQuadD = trialMaps.dofToQuadD;
|
||||
maps.quadToDof = testMaps.dofToQuad;
|
||||
maps.quadToDofD = testMaps.dofToQuadD;
|
||||
maps.quadWeights = testMaps.quadWeights;
|
||||
}
|
||||
return maps;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps OccaDofQuadMaps::GetD2QSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose)
|
||||
{
|
||||
const int dims = fe.GetDim();
|
||||
const int numDofs = fe.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
OccaDofQuadMaps maps;
|
||||
// Initialize the dof -> quad mapping
|
||||
maps.dofToQuad.allocate(device,
|
||||
numQuad, numDofs);
|
||||
maps.dofToQuadD.allocate(device,
|
||||
dims, numQuad, numDofs);
|
||||
|
||||
if (transpose)
|
||||
{
|
||||
maps.dofToQuad.reindex(1,0);
|
||||
maps.dofToQuadD.reindex(1,0);
|
||||
// Initialize quad weights only for transpose
|
||||
maps.quadWeights.allocate(device,
|
||||
numQuad);
|
||||
}
|
||||
|
||||
mfem::Vector d2q(numDofs);
|
||||
mfem::DenseMatrix d2qD(numDofs, dims);
|
||||
for (int q = 0; q < numQuad; ++q)
|
||||
{
|
||||
const mfem::IntegrationPoint &ip = ir.IntPoint(q);
|
||||
if (transpose)
|
||||
{
|
||||
maps.quadWeights[q] = ip.weight;
|
||||
}
|
||||
fe.CalcShape(ip, d2q);
|
||||
fe.CalcDShape(ip, d2qD);
|
||||
for (int d = 0; d < numDofs; ++d)
|
||||
{
|
||||
const double w = d2q[d];
|
||||
maps.dofToQuad(q, d) = w;
|
||||
for (int dim = 0; dim < dims; ++dim)
|
||||
{
|
||||
const double wD = d2qD(d, dim);
|
||||
maps.dofToQuadD(dim, q, d) = wD;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
maps.dofToQuad.keepInDevice();
|
||||
maps.dofToQuadD.keepInDevice();
|
||||
if (transpose)
|
||||
{
|
||||
maps.quadWeights.keepInDevice();
|
||||
}
|
||||
|
||||
return maps;
|
||||
}
|
||||
|
||||
//---[ Integrator Defines ]-----------
|
||||
std::string stringWithDim(const std::string &s, const int dim)
|
||||
{
|
||||
std::string ret = s;
|
||||
ret += ('0' + (char) dim);
|
||||
ret += 'D';
|
||||
return ret;
|
||||
}
|
||||
|
||||
int closestWarpBatchTo(const int value)
|
||||
{
|
||||
return ((value + 31) / 32) * 32;
|
||||
}
|
||||
|
||||
int closestMultipleWarpBatch(const int multiple, const int maxSize)
|
||||
{
|
||||
if (multiple > maxSize)
|
||||
{
|
||||
return maxSize;
|
||||
}
|
||||
int batch = (32 / multiple);
|
||||
int minDiff = 32 - (multiple * batch);
|
||||
for (int i = 64; i <= maxSize; i += 32)
|
||||
{
|
||||
const int newDiff = i - (multiple * (i / multiple));
|
||||
if (newDiff < minDiff)
|
||||
{
|
||||
batch = (i / multiple);
|
||||
minDiff = newDiff;
|
||||
}
|
||||
}
|
||||
return batch;
|
||||
}
|
||||
|
||||
void SetProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["defines/TRIAL_VDIM"] = trialFESpace.GetVDim();
|
||||
props["defines/TEST_VDIM"] = testFESpace.GetVDim();
|
||||
props["defines/NUM_DIM"] = trialFESpace.GetDim();
|
||||
|
||||
if (trialFESpace.hasTensorBasis())
|
||||
{
|
||||
SetTensorProperties(trialFESpace, testFESpace, ir, props);
|
||||
}
|
||||
else
|
||||
{
|
||||
SetSimplexProperties(trialFESpace, testFESpace, ir, props);
|
||||
}
|
||||
}
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetTensorProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace.GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace.GetFE(0));
|
||||
|
||||
const mfem::IntegrationRule &ir1D =
|
||||
mfem::IntRules.Get(mfem::Geometry::SEGMENT, ir.GetOrder());
|
||||
|
||||
const int trialDofs = trialFE.GetDof();
|
||||
const int testDofs = testFE.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
const int trialDofs1D = trialFE.GetOrder() + 1;
|
||||
const int testDofs1D = testFE.GetOrder() + 1;
|
||||
const int quad1D = ir1D.GetNPoints();
|
||||
int trialDofsND = trialDofs1D;
|
||||
int testDofsND = testDofs1D;
|
||||
int quadND = quad1D;
|
||||
|
||||
const bool trialByVDIM = (trialFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
const bool testByVDIM = (testFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TRIAL_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TEST_ORDERING"] = (int) testByVDIM;
|
||||
|
||||
props["defines/USING_TENSOR_OPS"] = 1;
|
||||
props["defines/NUM_DOFS"] = trialDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
|
||||
props["defines/TRIAL_DOFS"] = trialDofs;
|
||||
props["defines/TEST_DOFS"] = testDofs;
|
||||
|
||||
for (int d = 1; d <= 3; ++d)
|
||||
{
|
||||
if (d > 1)
|
||||
{
|
||||
trialDofsND *= trialDofs1D;
|
||||
testDofsND *= testDofs1D;
|
||||
quadND *= quad1D;
|
||||
}
|
||||
props["defines"][stringWithDim("NUM_DOFS_", d)] = trialDofsND;
|
||||
props["defines"][stringWithDim("NUM_QUAD_", d)] = quadND;
|
||||
|
||||
props["defines"][stringWithDim("TRIAL_DOFS_", d)] = trialDofsND;
|
||||
props["defines"][stringWithDim("TEST_DOFS_" , d)] = testDofsND;
|
||||
}
|
||||
|
||||
// 1D Defines
|
||||
const int m1InnerBatch = 32 * ((quad1D + 31) / 32);
|
||||
props["defines/A1_ELEMENT_BATCH"] = closestMultipleWarpBatch(quad1D, 512);
|
||||
props["defines/M1_OUTER_ELEMENT_BATCH"] = closestMultipleWarpBatch(m1InnerBatch,
|
||||
512);
|
||||
props["defines/M1_INNER_ELEMENT_BATCH"] = m1InnerBatch;
|
||||
|
||||
// 2D Defines
|
||||
props["defines/A2_ELEMENT_BATCH"] = 1;
|
||||
props["defines/A2_QUAD_BATCH"] = 1;
|
||||
props["defines/M2_ELEMENT_BATCH"] = 32;
|
||||
|
||||
// 3D Defines
|
||||
const int a3QuadBatch = closestMultipleWarpBatch(quadND, 512);
|
||||
props["defines/A3_ELEMENT_BATCH"] = closestMultipleWarpBatch(a3QuadBatch, 512);
|
||||
props["defines/A3_QUAD_BATCH"] = a3QuadBatch;
|
||||
}
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
SetSimplexProperties(fespace, fespace, ir, props);
|
||||
}
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace.GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace.GetFE(0));
|
||||
|
||||
const int trialDofs = trialFE.GetDof();
|
||||
const int testDofs = testFE.GetDof();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
const int maxDQ = std::max(std::max(trialDofs, testDofs), numQuad);
|
||||
|
||||
const bool trialByVDIM = (trialFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
const bool testByVDIM = (testFESpace.GetOrdering() == mfem::Ordering::byVDIM);
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TRIAL_ORDERING"] = (int) trialByVDIM;
|
||||
props["defines/TEST_ORDERING"] = (int) testByVDIM;
|
||||
|
||||
props["defines/USING_TENSOR_OPS"] = 0;
|
||||
props["defines/NUM_DOFS"] = trialDofs;
|
||||
props["defines/NUM_QUAD"] = numQuad;
|
||||
|
||||
props["defines/TRIAL_DOFS"] = trialDofs;
|
||||
props["defines/TEST_DOFS"] = testDofs;
|
||||
|
||||
// 2D Defines
|
||||
const int quadBatch = closestWarpBatchTo(numQuad);
|
||||
props["defines/A2_ELEMENT_BATCH"] = closestMultipleWarpBatch(quadBatch, 2048);
|
||||
props["defines/A2_QUAD_BATCH"] = quadBatch;
|
||||
props["defines/M2_INNER_BATCH"] = closestWarpBatchTo(maxDQ);
|
||||
|
||||
// 3D Defines
|
||||
props["defines/A3_ELEMENT_BATCH"] = closestMultipleWarpBatch(quadBatch, 2048);
|
||||
props["defines/A3_QUAD_BATCH"] = quadBatch;
|
||||
props["defines/M3_INNER_BATCH"] = closestWarpBatchTo(maxDQ);
|
||||
}
|
||||
|
||||
|
||||
//---[ Base Integrator ]--------------
|
||||
OccaIntegrator::OccaIntegrator(const Engine &e)
|
||||
: engine(&e),
|
||||
bform(),
|
||||
mesh(),
|
||||
otrialFESpace(),
|
||||
otestFESpace(),
|
||||
trialFESpace(),
|
||||
testFESpace(),
|
||||
itype(DomainIntegrator),
|
||||
ir(NULL),
|
||||
hasTensorBasis(false) { }
|
||||
|
||||
OccaIntegrator::~OccaIntegrator() {}
|
||||
|
||||
void OccaIntegrator::SetupMaps()
|
||||
{
|
||||
maps = OccaDofQuadMaps::Get(GetDevice(),
|
||||
*otrialFESpace,
|
||||
*otestFESpace,
|
||||
*ir);
|
||||
|
||||
mapsTranspose = OccaDofQuadMaps::Get(GetDevice(),
|
||||
*otestFESpace,
|
||||
*otrialFESpace,
|
||||
*ir);
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaIntegrator::GetTrialOccaFESpace() const
|
||||
{
|
||||
return *otrialFESpace;
|
||||
}
|
||||
|
||||
FiniteElementSpace& OccaIntegrator::GetTestOccaFESpace() const
|
||||
{
|
||||
return *otestFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaIntegrator::GetTrialFESpace() const
|
||||
{
|
||||
return *trialFESpace;
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace& OccaIntegrator::GetTestFESpace() const
|
||||
{
|
||||
return *testFESpace;
|
||||
}
|
||||
|
||||
void OccaIntegrator::SetIntegrationRule(const mfem::IntegrationRule &ir_)
|
||||
{
|
||||
ir = &ir_;
|
||||
}
|
||||
|
||||
const mfem::IntegrationRule& OccaIntegrator::GetIntegrationRule() const
|
||||
{
|
||||
return *ir;
|
||||
}
|
||||
|
||||
OccaDofQuadMaps& OccaIntegrator::GetDofQuadMaps()
|
||||
{
|
||||
return maps;
|
||||
}
|
||||
|
||||
void OccaIntegrator::SetupIntegrator(OccaBilinearForm &bform_,
|
||||
const ::occa::properties &props_,
|
||||
const OccaIntegratorType itype_)
|
||||
{
|
||||
MFEM_ASSERT(engine == &bform_.OccaEngine(), "");
|
||||
bform = &bform_;
|
||||
mesh = &(bform_.GetMesh());
|
||||
|
||||
otrialFESpace = &(bform_.GetTrialOccaFESpace());
|
||||
otestFESpace = &(bform_.GetTestOccaFESpace());
|
||||
|
||||
trialFESpace = &(bform_.GetTrialFESpace());
|
||||
testFESpace = &(bform_.GetTestFESpace());
|
||||
|
||||
hasTensorBasis = otrialFESpace->hasTensorBasis();
|
||||
|
||||
props = props_;
|
||||
itype = itype_;
|
||||
|
||||
if (ir == NULL)
|
||||
{
|
||||
SetupIntegrationRule();
|
||||
}
|
||||
|
||||
SetupMaps();
|
||||
|
||||
SetProperties(*otrialFESpace,
|
||||
*otestFESpace,
|
||||
*ir,
|
||||
props);
|
||||
|
||||
Setup();
|
||||
}
|
||||
|
||||
OccaGeometry OccaIntegrator::GetGeometry(const int flags)
|
||||
{
|
||||
return OccaGeometry::Get(GetDevice(), *otrialFESpace, *ir, flags);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetAssembleKernel(const ::occa::properties
|
||||
&props)
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
return GetKernel(stringWithDim("Assemble", fe.GetDim()),
|
||||
props);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetMultAddKernel(const ::occa::properties &props)
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
return GetKernel(stringWithDim("MultAdd", fe.GetDim()),
|
||||
props);
|
||||
}
|
||||
|
||||
::occa::kernel OccaIntegrator::GetKernel(const std::string &kernelName,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
const std::string filename = GetName() + ".okl";
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
return GetDevice().buildKernel(okl_path + filename,
|
||||
kernelName,
|
||||
props);
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Diffusion Integrator ]---------
|
||||
OccaDiffusionIntegrator::OccaDiffusionIntegrator(const OccaCoefficient &coeff_)
|
||||
:
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaDiffusionIntegrator::~OccaDiffusionIntegrator() {}
|
||||
|
||||
|
||||
std::string OccaDiffusionIntegrator::GetName()
|
||||
{
|
||||
return "DiffusionIntegrator";
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
ir = &mfem::DiffusionIntegrator::GetRule(trialFE, testFE);
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::Assemble()
|
||||
{
|
||||
const mfem::FiniteElement &fe = *(trialFESpace->GetFE(0));
|
||||
|
||||
const int dims = fe.GetDim();
|
||||
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
|
||||
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.OccaResize(symmDims * quadraturePoints * elements,
|
||||
sizeof(double));
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaDiffusionIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
// Note: x and y are E-vectors
|
||||
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Mass Integrator ]--------------
|
||||
OccaMassIntegrator::OccaMassIntegrator(const OccaCoefficient &coeff_) :
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaMassIntegrator::~OccaMassIntegrator() {}
|
||||
|
||||
std::string OccaMassIntegrator::GetName()
|
||||
{
|
||||
return "MassIntegrator";
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
mfem::ElementTransformation &T = *trialFESpace->GetElementTransformation(0);
|
||||
ir = &mfem::MassIntegrator::GetRule(trialFE, testFE, T);
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::Assemble()
|
||||
{
|
||||
if (assembledOperator.Size())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.Resize<double>(quadraturePoints * elements, NULL);
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::SetOperator(Vector &v)
|
||||
{
|
||||
assembledOperator = v;
|
||||
}
|
||||
|
||||
void OccaMassIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Vector Mass Integrator ]--------------
|
||||
OccaVectorMassIntegrator::OccaVectorMassIntegrator(const OccaCoefficient &
|
||||
coeff_)
|
||||
:
|
||||
OccaIntegrator(coeff_.OccaEngine()),
|
||||
coeff(coeff_),
|
||||
assembledOperator(*(new Layout(coeff_.OccaEngine(), 0)))
|
||||
{
|
||||
coeff.SetName("COEFF");
|
||||
}
|
||||
|
||||
OccaVectorMassIntegrator::~OccaVectorMassIntegrator() {}
|
||||
|
||||
std::string OccaVectorMassIntegrator::GetName()
|
||||
{
|
||||
return "VectorMassIntegrator";
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::SetupIntegrationRule()
|
||||
{
|
||||
const mfem::FiniteElement &trialFE = *(trialFESpace->GetFE(0));
|
||||
const mfem::FiniteElement &testFE = *(testFESpace->GetFE(0));
|
||||
mfem::ElementTransformation &T = *trialFESpace->GetElementTransformation(0);
|
||||
ir = &mfem::MassIntegrator::GetRule(trialFE, testFE, T);
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::Setup()
|
||||
{
|
||||
::occa::properties kernelProps = props;
|
||||
|
||||
coeff.Setup(*this, kernelProps);
|
||||
|
||||
// Setup assemble and mult kernels
|
||||
assembleKernel = GetAssembleKernel(kernelProps);
|
||||
multKernel = GetMultAddKernel(kernelProps);
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::Assemble()
|
||||
{
|
||||
const int elements = trialFESpace->GetNE();
|
||||
const int quadraturePoints = ir->GetNPoints();
|
||||
|
||||
OccaGeometry geom = GetGeometry(OccaGeometry::Jacobian);
|
||||
|
||||
assembledOperator.Resize<double>(quadraturePoints * elements, NULL);
|
||||
|
||||
assembleKernel((int) mesh->GetNE(),
|
||||
maps.quadWeights,
|
||||
geom.J,
|
||||
coeff,
|
||||
assembledOperator.OccaMem());
|
||||
}
|
||||
|
||||
void OccaVectorMassIntegrator::MultAdd(Vector &x, Vector &y)
|
||||
{
|
||||
multKernel((int) mesh->GetNE(),
|
||||
maps.dofToQuad,
|
||||
maps.dofToQuadD,
|
||||
maps.quadToDof,
|
||||
maps.quadToDofD,
|
||||
assembledOperator.OccaMem(),
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,323 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
#define MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "fespace.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "coefficient.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class OccaGeometry
|
||||
{
|
||||
public:
|
||||
::occa::array<double> meshNodes;
|
||||
::occa::array<double> J, invJ, detJ;
|
||||
|
||||
// byVDIM -> [x y z x y z x y z]
|
||||
// byNodes -> [x x x y y y z z z]
|
||||
static const int Jacobian = (1 << 0);
|
||||
static const int JacobianInv = (1 << 1);
|
||||
static const int JacobianDet = (1 << 2);
|
||||
|
||||
static OccaGeometry Get(::occa::device device,
|
||||
FiniteElementSpace &ofespace,
|
||||
const IntegrationRule &ir,
|
||||
const int flags = (Jacobian |
|
||||
JacobianInv |
|
||||
JacobianDet));
|
||||
};
|
||||
|
||||
class OccaDofQuadMaps
|
||||
{
|
||||
private:
|
||||
// Reuse dof-quad maps
|
||||
static std::map<std::string, OccaDofQuadMaps> AllDofQuadMaps;
|
||||
std::string hash;
|
||||
|
||||
public:
|
||||
// Local stiffness matrices (B and B^T operators)
|
||||
::occa::array<double, ::occa::dynamic> dofToQuad, dofToQuadD; // B
|
||||
::occa::array<double, ::occa::dynamic> quadToDof, quadToDofD; // B^T
|
||||
::occa::array<double> quadWeights;
|
||||
|
||||
OccaDofQuadMaps();
|
||||
OccaDofQuadMaps(const OccaDofQuadMaps &maps);
|
||||
OccaDofQuadMaps& operator = (const OccaDofQuadMaps &maps);
|
||||
|
||||
// [[x y] [x y] [x y]]
|
||||
// [[x y z] [x y z] [x y z]]
|
||||
// mfem::GridFunction* mfem::Mesh::GetNodes() { return Nodes; }
|
||||
|
||||
// mfem::FiniteElementSpace *Nodes->FESpace()
|
||||
// 25
|
||||
// 1D [x x x x x x]
|
||||
// 2D [x y x y x y]
|
||||
// GetVdim()
|
||||
// 3D ordering == byVDIM -> [x y z x y z x y z x y z x y z x y z]
|
||||
// ordering == byNODES -> [x x x x x x y y y y y y z z z z z z]
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const FiniteElementSpace &trialFESpace,
|
||||
const FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& Get(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps GetD2QTensorMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps& GetSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &trialFE,
|
||||
const mfem::FiniteElement &testFE,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
|
||||
static OccaDofQuadMaps GetD2QSimplexMaps(::occa::device device,
|
||||
const mfem::FiniteElement &fe,
|
||||
const mfem::IntegrationRule &ir,
|
||||
const bool transpose = false);
|
||||
};
|
||||
|
||||
//---[ Define Methods ]---------------
|
||||
std::string stringWithDim(const std::string &s, const int dim);
|
||||
int closestWarpBatch(const int multiple, const int maxSize);
|
||||
|
||||
void SetProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetTensorProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &fespace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
void SetSimplexProperties(FiniteElementSpace &trialFESpace,
|
||||
FiniteElementSpace &testFESpace,
|
||||
const IntegrationRule &ir,
|
||||
::occa::properties &props);
|
||||
|
||||
//---[ Base Integrator ]--------------
|
||||
class OccaIntegrator
|
||||
{
|
||||
protected:
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
OccaBilinearForm *bform;
|
||||
mfem::Mesh *mesh;
|
||||
|
||||
FiniteElementSpace *otrialFESpace;
|
||||
FiniteElementSpace *otestFESpace;
|
||||
|
||||
mfem::FiniteElementSpace *trialFESpace;
|
||||
mfem::FiniteElementSpace *testFESpace;
|
||||
|
||||
::occa::properties props;
|
||||
OccaIntegratorType itype;
|
||||
|
||||
const IntegrationRule *ir;
|
||||
bool hasTensorBasis;
|
||||
OccaDofQuadMaps maps;
|
||||
OccaDofQuadMaps mapsTranspose;
|
||||
|
||||
public:
|
||||
OccaIntegrator(const Engine &e);
|
||||
virtual ~OccaIntegrator();
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
virtual std::string GetName() = 0;
|
||||
|
||||
FiniteElementSpace& GetTrialOccaFESpace() const;
|
||||
FiniteElementSpace& GetTestOccaFESpace() const;
|
||||
|
||||
mfem::FiniteElementSpace& GetTrialFESpace() const;
|
||||
mfem::FiniteElementSpace& GetTestFESpace() const;
|
||||
|
||||
void SetIntegrationRule(const mfem::IntegrationRule &ir_);
|
||||
const mfem::IntegrationRule& GetIntegrationRule() const;
|
||||
|
||||
OccaDofQuadMaps& GetDofQuadMaps();
|
||||
|
||||
void SetupMaps();
|
||||
|
||||
virtual void SetupIntegrationRule() = 0;
|
||||
|
||||
virtual void SetupIntegrator(OccaBilinearForm &bform_,
|
||||
const ::occa::properties &props_,
|
||||
const OccaIntegratorType itype_);
|
||||
|
||||
virtual void Setup() = 0;
|
||||
|
||||
virtual void Assemble() = 0;
|
||||
/// This method works on E-vectors!
|
||||
virtual void MultAdd(Vector &x, Vector &y) = 0;
|
||||
|
||||
virtual void MultTransposeAdd(Vector &x, Vector &y)
|
||||
{
|
||||
mfem_error("OccaIntegrator::MultTransposeAdd() is not overloaded!");
|
||||
}
|
||||
|
||||
OccaGeometry GetGeometry(const int flags = (OccaGeometry::Jacobian |
|
||||
OccaGeometry::JacobianInv |
|
||||
OccaGeometry::JacobianDet));
|
||||
|
||||
::occa::kernel GetAssembleKernel(const ::occa::properties &props);
|
||||
::occa::kernel GetMultAddKernel(const ::occa::properties &props);
|
||||
|
||||
::occa::kernel GetKernel(const std::string &kernelName,
|
||||
const ::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Diffusion Integrator ]---------
|
||||
class OccaDiffusionIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaDiffusionIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaDiffusionIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Mass Integrator ]--------------
|
||||
class OccaMassIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaMassIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaMassIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
void SetOperator(Vector &v);
|
||||
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
//====================================
|
||||
|
||||
//---[ Vector Mass Integrator ]--------------
|
||||
class OccaVectorMassIntegrator : public OccaIntegrator
|
||||
{
|
||||
private:
|
||||
OccaCoefficient coeff;
|
||||
|
||||
::occa::kernel assembleKernel, multKernel;
|
||||
|
||||
Vector assembledOperator;
|
||||
|
||||
public:
|
||||
OccaVectorMassIntegrator(const OccaCoefficient &coeff_);
|
||||
virtual ~OccaVectorMassIntegrator();
|
||||
|
||||
virtual std::string GetName();
|
||||
|
||||
virtual void SetupIntegrationRule();
|
||||
|
||||
virtual void Setup();
|
||||
|
||||
virtual void Assemble();
|
||||
|
||||
virtual void MultAdd(Vector &x, Vector &y);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_BILIN_INTEG_HPP
|
||||
@@ -0,0 +1,357 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "coefficient.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
//---[ Parameter ]------------
|
||||
OccaParameter::~OccaParameter() {}
|
||||
|
||||
void OccaParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props) {}
|
||||
|
||||
::occa::kernelArg OccaParameter::KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg();
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Include Parameter ]------------
|
||||
OccaIncludeParameter::OccaIncludeParameter(const std::string &filename_) :
|
||||
filename(filename_) {}
|
||||
|
||||
OccaParameter* OccaIncludeParameter::Clone()
|
||||
{
|
||||
return new OccaIncludeParameter(filename);
|
||||
}
|
||||
|
||||
void OccaIncludeParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["headers"].asArray() += "#include " + filename;
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Source Parameter ]------------
|
||||
OccaSourceParameter::OccaSourceParameter(const std::string &source_) :
|
||||
source(source_) {}
|
||||
|
||||
OccaParameter* OccaSourceParameter::Clone()
|
||||
{
|
||||
return new OccaSourceParameter(source);
|
||||
}
|
||||
|
||||
void OccaSourceParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["headers"].asArray() += source;
|
||||
}
|
||||
//====================================
|
||||
|
||||
//---[ Vector Parameter ]-------
|
||||
OccaVectorParameter::OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const bool useRestrict_) :
|
||||
name(name_),
|
||||
v(v_),
|
||||
useRestrict(useRestrict_),
|
||||
attr("") {}
|
||||
|
||||
OccaVectorParameter::OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const std::string &attr_,
|
||||
const bool useRestrict_) :
|
||||
name(name_),
|
||||
v(v_),
|
||||
useRestrict(useRestrict_),
|
||||
attr(attr_) {}
|
||||
|
||||
OccaParameter* OccaVectorParameter::Clone()
|
||||
{
|
||||
return new OccaVectorParameter(name, v, attr, useRestrict);
|
||||
}
|
||||
|
||||
void OccaVectorParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
args += "const double *";
|
||||
if (useRestrict)
|
||||
{
|
||||
args += " restrict ";
|
||||
}
|
||||
args += name;
|
||||
if (attr.size())
|
||||
{
|
||||
args += ' ';
|
||||
args += attr;
|
||||
}
|
||||
args += ",\n";
|
||||
}
|
||||
|
||||
::occa::kernelArg OccaVectorParameter::KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg(v.OccaMem());
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ GridFunction Parameter ]-------
|
||||
OccaGridFunctionParameter::OccaGridFunctionParameter(const std::string &name_,
|
||||
const Engine &e,
|
||||
mfem::GridFunction &gf_,
|
||||
const bool useRestrict_)
|
||||
: name(name_),
|
||||
gf(gf_),
|
||||
gfQuad(e),
|
||||
useRestrict(useRestrict_) {}
|
||||
|
||||
OccaParameter* OccaGridFunctionParameter::Clone()
|
||||
{
|
||||
OccaGridFunctionParameter *param =
|
||||
new OccaGridFunctionParameter(name, gfQuad.OccaEngine(), gf, useRestrict);
|
||||
param->gfQuad.MakeRef(gfQuad);
|
||||
return param;
|
||||
}
|
||||
|
||||
void OccaGridFunctionParameter::Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
if (useRestrict)
|
||||
{
|
||||
args += "@restrict ";
|
||||
}
|
||||
args += "const double *";
|
||||
args += name;
|
||||
args += " @dim(NUM_QUAD, numElements),\n";
|
||||
|
||||
FiniteElementSpace &f = gf.FESpace()->Get_PFESpace()->As<FiniteElementSpace>();
|
||||
ToQuad(integ.GetIntegrationRule(), f, gf.Get_PVector()->As<Vector>(), gfQuad);
|
||||
}
|
||||
|
||||
::occa::kernelArg OccaGridFunctionParameter::KernelArgs()
|
||||
{
|
||||
return gfQuad.OccaMem();
|
||||
}
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Coefficient ]------------------
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const double value) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = value;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, mfem::GridFunction &gf,
|
||||
const bool useRestrict) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = "(u(q, e))";
|
||||
AddGridFunction("u", gf, useRestrict);
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const std::string &source) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = source;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const Engine &e, const char *source) :
|
||||
engine(&e),
|
||||
integ(NULL),
|
||||
name("COEFF")
|
||||
{
|
||||
coeffValue = source;
|
||||
}
|
||||
|
||||
OccaCoefficient::OccaCoefficient(const OccaCoefficient &coeff) :
|
||||
engine(coeff.engine),
|
||||
integ(NULL),
|
||||
name(coeff.name),
|
||||
coeffValue(coeff.coeffValue)
|
||||
{
|
||||
|
||||
const int paramCount = (int) coeff.params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
params.push_back(coeff.params[i]->Clone());
|
||||
}
|
||||
}
|
||||
|
||||
OccaCoefficient::~OccaCoefficient()
|
||||
{
|
||||
const int paramCount = (int) params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
delete params[i];
|
||||
}
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::SetName(const std::string &name_)
|
||||
{
|
||||
name = name_;
|
||||
return *this;
|
||||
}
|
||||
|
||||
void OccaCoefficient::Setup(OccaIntegrator &integ_,
|
||||
::occa::properties &props_)
|
||||
{
|
||||
integ = &integ_;
|
||||
|
||||
const int paramCount = (int) params.size();
|
||||
props_["defines"][name + "_ARGS"] = "";
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
params[i]->Setup(integ_, props_);
|
||||
}
|
||||
props_["defines"][name] = coeffValue;
|
||||
|
||||
props = props_;
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::Add(OccaParameter *param)
|
||||
{
|
||||
params.push_back(param);
|
||||
return *this;
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::IncludeHeader(const std::string &filename)
|
||||
{
|
||||
return Add(new OccaIncludeParameter(filename));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::IncludeSource(const std::string &source)
|
||||
{
|
||||
return Add(new OccaSourceParameter(source));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const bool useRestrict)
|
||||
{
|
||||
return Add(new OccaVectorParameter(name_, v, useRestrict));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const std::string &attr,
|
||||
const bool useRestrict)
|
||||
{
|
||||
return Add(new OccaVectorParameter(name_, v, attr, useRestrict));
|
||||
}
|
||||
|
||||
OccaCoefficient& OccaCoefficient::AddGridFunction(const std::string &name_,
|
||||
mfem::GridFunction &gf,
|
||||
const bool useRestrict)
|
||||
{
|
||||
MFEM_ASSERT(engine->CheckVector(gf.Get_PVector()) &&
|
||||
engine->CheckFESpace(gf.FESpace()->Get_PFESpace()),
|
||||
"invalid device GridFunction");
|
||||
return Add(new OccaGridFunctionParameter(name_, *engine, gf, useRestrict));
|
||||
}
|
||||
|
||||
bool OccaCoefficient::IsConstant()
|
||||
{
|
||||
return coeffValue.isNumber();
|
||||
}
|
||||
|
||||
double OccaCoefficient::GetConstantValue()
|
||||
{
|
||||
if (!IsConstant())
|
||||
{
|
||||
mfem_error("OccaCoefficient is not constant");
|
||||
}
|
||||
return coeffValue.number();
|
||||
}
|
||||
|
||||
Vector OccaCoefficient::Eval()
|
||||
{
|
||||
if (integ == NULL)
|
||||
{
|
||||
mfem_error("OccaCoefficient requires a Setup() call before Eval()");
|
||||
}
|
||||
|
||||
mfem::FiniteElementSpace &fespace = integ->GetTrialFESpace();
|
||||
const mfem::IntegrationRule &ir = integ->GetIntegrationRule();
|
||||
|
||||
const int elements = fespace.GetNE();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
Vector quadCoeff(*(new Layout(OccaEngine(), numQuad * elements)));
|
||||
Eval(quadCoeff);
|
||||
return quadCoeff;
|
||||
}
|
||||
|
||||
void OccaCoefficient::Eval(Vector &quadCoeff)
|
||||
{
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
static ::occa::kernelBuilder builder =
|
||||
::occa::kernelBuilder::fromFile(okl_path + "coefficient.okl",
|
||||
"CoefficientEval");
|
||||
|
||||
if (integ == NULL)
|
||||
{
|
||||
mfem_error("OccaCoefficient requires a Setup() call before Eval()");
|
||||
}
|
||||
|
||||
const int elements = integ->GetTrialFESpace().GetNE();
|
||||
|
||||
::occa::properties kernelProps = props;
|
||||
if (name != "COEFF")
|
||||
{
|
||||
kernelProps["defines/COEFF"] = name;
|
||||
kernelProps["defines/COEFF_ARGS"] = name + "_ARGS";
|
||||
}
|
||||
|
||||
::occa::kernel evalKernel = builder.build(GetDevice(), kernelProps);
|
||||
evalKernel(elements, *this, quadCoeff.OccaMem());
|
||||
}
|
||||
|
||||
OccaCoefficient::operator ::occa::kernelArg ()
|
||||
{
|
||||
::occa::kernelArg kArg;
|
||||
const int paramCount = (int) params.size();
|
||||
for (int i = 0; i < paramCount; ++i)
|
||||
{
|
||||
kArg.add(params[i]->KernelArgs());
|
||||
}
|
||||
return kArg;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,287 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class OccaIntegrator;
|
||||
|
||||
|
||||
class OccaParameter
|
||||
{
|
||||
public:
|
||||
virtual ~OccaParameter();
|
||||
|
||||
virtual OccaParameter* Clone() = 0;
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
|
||||
|
||||
//---[ Include Parameter ]------------
|
||||
class OccaIncludeParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
std::string filename;
|
||||
|
||||
public:
|
||||
OccaIncludeParameter(const std::string &filename_);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Source Parameter ]------------
|
||||
class OccaSourceParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
std::string source;
|
||||
|
||||
public:
|
||||
OccaSourceParameter(const std::string &filename_);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Define Parameter ]------------
|
||||
template <class TM>
|
||||
class OccaDefineParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
TM value;
|
||||
|
||||
public:
|
||||
OccaDefineParameter(const std::string &name_,
|
||||
const TM &value_) :
|
||||
name(name_),
|
||||
value(value_) {}
|
||||
|
||||
virtual OccaParameter* Clone()
|
||||
{
|
||||
return new OccaDefineParameter(name, value);
|
||||
}
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
props["defines"][name] = value;
|
||||
}
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Variable Parameter ]-----------
|
||||
template <class TM>
|
||||
class OccaVariableParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
const TM &value;
|
||||
|
||||
public:
|
||||
OccaVariableParameter(const std::string &name_,
|
||||
const TM &value_) :
|
||||
name(name_),
|
||||
value(value_) {}
|
||||
|
||||
virtual OccaParameter* Clone()
|
||||
{
|
||||
return new OccaVariableParameter(name, value);
|
||||
}
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props)
|
||||
{
|
||||
std::string &args = (props["defines/COEFF_ARGS"]
|
||||
.asString()
|
||||
.string());
|
||||
// const TM name,\n"
|
||||
args += "const ";
|
||||
args += ::occa::primitiveinfo<TM>::name;
|
||||
args += ' ';
|
||||
args += name;
|
||||
args += ",\n";
|
||||
}
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs()
|
||||
{
|
||||
return ::occa::kernelArg(value);
|
||||
}
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Vector Parameter ]-------
|
||||
class OccaVectorParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
Vector v;
|
||||
bool useRestrict;
|
||||
std::string attr;
|
||||
|
||||
public:
|
||||
OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
OccaVectorParameter(const std::string &name_,
|
||||
Vector &v_,
|
||||
const std::string &attr_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ GridFunction Parameter ]-------
|
||||
class OccaGridFunctionParameter : public OccaParameter
|
||||
{
|
||||
private:
|
||||
const std::string name;
|
||||
mfem::GridFunction &gf;
|
||||
Vector gfQuad;
|
||||
bool useRestrict;
|
||||
|
||||
public:
|
||||
OccaGridFunctionParameter(const std::string &name_,
|
||||
const Engine &e,
|
||||
mfem::GridFunction &gf_,
|
||||
const bool useRestrict_ = false);
|
||||
|
||||
virtual OccaParameter* Clone();
|
||||
|
||||
virtual void Setup(OccaIntegrator &integ,
|
||||
::occa::properties &props);
|
||||
|
||||
virtual ::occa::kernelArg KernelArgs();
|
||||
};
|
||||
//====================================
|
||||
|
||||
|
||||
//---[ Coefficient ]------------------
|
||||
// [MISSING]
|
||||
// Needs to know about the integrator's
|
||||
// - fespace
|
||||
// - ir
|
||||
// Step where parameters that need the ir get called for setup
|
||||
// For example, GridFunction (d, e) -> (q, e)
|
||||
class OccaCoefficient
|
||||
{
|
||||
private:
|
||||
SharedPtr<const Engine> engine;
|
||||
|
||||
OccaIntegrator *integ;
|
||||
|
||||
std::string name;
|
||||
::occa::json coeffValue;
|
||||
|
||||
::occa::properties props;
|
||||
std::vector<OccaParameter*> params;
|
||||
|
||||
public:
|
||||
OccaCoefficient(const Engine &e, const double value = 1.0);
|
||||
OccaCoefficient(const Engine &e, mfem::GridFunction &gf,
|
||||
const bool useRestrict = false);
|
||||
OccaCoefficient(const Engine &e, const std::string &source);
|
||||
OccaCoefficient(const Engine &e, const char *source);
|
||||
~OccaCoefficient();
|
||||
|
||||
OccaCoefficient(const OccaCoefficient &coeff);
|
||||
|
||||
const Engine &OccaEngine() const { return *engine; }
|
||||
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return engine->GetDevice(idx); }
|
||||
|
||||
OccaCoefficient& SetName(const std::string &name_);
|
||||
|
||||
void Setup(OccaIntegrator &integ_,
|
||||
::occa::properties &props_);
|
||||
|
||||
OccaCoefficient& Add(OccaParameter *param);
|
||||
|
||||
OccaCoefficient& IncludeHeader(const std::string &filename);
|
||||
OccaCoefficient& IncludeSource(const std::string &source);
|
||||
|
||||
template <class TM>
|
||||
OccaCoefficient& AddDefine(const std::string &name_, const TM &value)
|
||||
{
|
||||
return Add(new OccaDefineParameter<TM>(name_, value));
|
||||
}
|
||||
|
||||
template <class TM>
|
||||
OccaCoefficient& AddVariable(const std::string &name_, const TM &value)
|
||||
{
|
||||
return Add(new OccaVariableParameter<TM>(name_, value));
|
||||
}
|
||||
|
||||
OccaCoefficient& AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const bool useRestrict = false);
|
||||
|
||||
|
||||
OccaCoefficient& AddVector(const std::string &name_,
|
||||
Vector &v,
|
||||
const std::string &attr,
|
||||
const bool useRestrict = false);
|
||||
|
||||
OccaCoefficient& AddGridFunction(const std::string &name_,
|
||||
mfem::GridFunction &gf,
|
||||
const bool useRestrict = false);
|
||||
|
||||
bool IsConstant();
|
||||
double GetConstantValue();
|
||||
|
||||
Vector Eval();
|
||||
void Eval(Vector &quadCoeff);
|
||||
|
||||
operator ::occa::kernelArg ();
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_COEFFICIENT_HPP
|
||||
@@ -0,0 +1,40 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_OCCA_DEFINES
|
||||
#define MFEM_OCCA_DEFINES
|
||||
|
||||
#ifndef USING_TENSOR_OPS
|
||||
# define USING_TENSOR_OPS 0
|
||||
#endif
|
||||
|
||||
#ifdef OCCA_USING_GPU
|
||||
# define GPU_ORDER_2(I0, I1) @dimOrder(I0, I1)
|
||||
# define GPU_ORDER_3(I0, I1, I2) @dimOrder(I0, I1, I2)
|
||||
# define GPU_ORDER_4(I0, I1, I2, I3) @dimOrder(I0, I1, I2, I3)
|
||||
#else
|
||||
# define GPU_ORDER_2(I0, I1) @dimOrder(0, 1)
|
||||
# define GPU_ORDER_3(I0, I1, I2) @dimOrder(0, 1, 2)
|
||||
# define GPU_ORDER_4(I0, I1, I2, I3) @dimOrder(0, 1, 2, 3)
|
||||
#endif
|
||||
|
||||
#ifndef COEFF
|
||||
# define COEFF 1.0
|
||||
# define COEFF_ARGS
|
||||
#endif
|
||||
|
||||
#if USING_TENSOR_OPS
|
||||
# include "mfem-occa://defines/tensor.okl"
|
||||
#else
|
||||
# include "mfem-occa://defines/simplex.okl"
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,40 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#define USING_LOW_ORDER 1
|
||||
#define USING_HI_ORDER 0
|
||||
|
||||
typedef double* DofToQuad_t @dim(NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD2D_t @dim(2, NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD3D_t @dim(3, NUM_QUAD, NUM_DOFS);
|
||||
|
||||
typedef double* QuadToDof_t @dim(NUM_DOFS, NUM_QUAD);
|
||||
typedef double* QuadToDofD2D_t @dim(2, NUM_DOFS, NUM_QUAD);
|
||||
typedef double* QuadToDofD3D_t @dim(3, NUM_DOFS, NUM_QUAD);
|
||||
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD, numElements);
|
||||
|
||||
typedef double* SymmOperator2D_t @dim(3, NUM_QUAD, numElements);
|
||||
typedef double* SymmOperator3D_t @dim(6, NUM_QUAD, numElements);
|
||||
|
||||
typedef double* DLocal_t @dim(NUM_DOFS, numElements);
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements);
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements);
|
||||
#else
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements) @dimOrder(2,0,1);
|
||||
#endif
|
||||
|
||||
typedef int* DLocalMap_t @dim(NUM_DOFS, numElements);
|
||||
@@ -0,0 +1,85 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#if NUM_QUAD_1D < NUM_DOFS_1D
|
||||
# define NUM_MAX_1D NUM_DOFS_1D
|
||||
#else
|
||||
# define NUM_MAX_1D NUM_QUAD_1D
|
||||
#endif
|
||||
|
||||
#define NUM_MAX_2D (NUM_MAX_1D * NUM_MAX_1D)
|
||||
|
||||
#define NUM_QUAD_DOFS_1D (NUM_QUAD_1D * NUM_DOFS_1D)
|
||||
|
||||
#define QUAD_2D_ID(X, Y) (X + ((Y) * NUM_QUAD_1D))
|
||||
#define DOFS_2D_ID(X, Y) (X + ((Y) * NUM_DOFS_1D))
|
||||
|
||||
#define QUAD_3D_ID(X, Y, Z) (X + ((Y) * NUM_QUAD_1D) + ((Z) * NUM_QUAD_2D))
|
||||
#define DOFS_3D_ID(X, Y, Z) (X + ((Y) * NUM_DOFS_1D) + ((Z) * NUM_DOFS_2D))
|
||||
|
||||
#if NUM_MAX_1D < 8
|
||||
# define USING_LOW_ORDER 1
|
||||
# define USING_HI_ORDER 0
|
||||
#else
|
||||
# define USING_LOW_ORDER 0
|
||||
# define USING_HI_ORDER 1
|
||||
#endif
|
||||
|
||||
#define M1_ELEMENT_BATCHES (M1_OUTER_ELEMENT_BATCH * M1_INNER_ELEMENT_BATCH)
|
||||
|
||||
typedef double* DofToQuad_t @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
typedef double* QuadToDof_t @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
typedef double* Jacobian_t @dim(NUM_DIM, NUM_DIM, numElements);
|
||||
typedef double* Jacobian1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD_2D, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD_3D, numElements);
|
||||
|
||||
typedef double* SymmOperator1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* SymmOperator2D_t @dim(3, NUM_QUAD_2D, numElements);
|
||||
typedef double* SymmOperator3D_t @dim(6, NUM_QUAD_3D, numElements);
|
||||
|
||||
typedef double* DLocal_t @dim(NUM_DOFS, numElements);
|
||||
typedef double* DLocal1D_t @dim(NUM_DOFS_1D, numElements);
|
||||
typedef double* DLocal2D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef double* DLocal3D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
typedef double* QLocal1D_t @dim(NUM_QUAD_1D, numElements);
|
||||
typedef double* QLocal2D_t @dim(NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
typedef double* QLocal3D_t @dim(NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements);
|
||||
typedef double* DVLocal1D_t @dim(NUM_VDIM, NUM_DOFS_1D, numElements);
|
||||
typedef double* DVLocal2D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef double* DVLocal3D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements);
|
||||
typedef double* QVLocal1D_t @dim(NUM_VDIM, NUM_QUAD_1D, numElements);
|
||||
typedef double* QVLocal2D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
typedef double* QVLocal3D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements);
|
||||
#else
|
||||
typedef double* DVLocal_t @dim(NUM_VDIM, NUM_DOFS, numElements) @dimOrder(2,0,1);
|
||||
typedef double* DVLocal1D_t @dim(NUM_VDIM, NUM_DOFS_1D, numElements) @dimOrder(2,0,1);
|
||||
typedef double* DVLocal2D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, numElements) @dimOrder(3,0,1,2);
|
||||
typedef double* DVLocal3D_t @dim(NUM_VDIM, NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements) @dimOrder(4,0,1,2,3);
|
||||
|
||||
typedef double* QVLocal_t @dim(NUM_VDIM, NUM_QUAD, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal1D_t @dim(NUM_VDIM, NUM_QUAD_1D, numElements) @dimOrder(2,0,1);
|
||||
typedef double* QVLocal2D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, numElements) @dimOrder(3,0,1,2);
|
||||
typedef double* QVLocal3D_t @dim(NUM_VDIM, NUM_QUAD_1D, NUM_QUAD_1D, NUM_QUAD_1D, numElements) @dimOrder(4,0,1,2,3);
|
||||
#endif
|
||||
|
||||
typedef int* DLocalMap_t @dim(NUM_DOFS, numElements);
|
||||
typedef int* DLocalMap1D_t @dim(NUM_DOFS_1D, numElements);
|
||||
typedef int* DLocalMap2D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
typedef int* DLocalMap3D_t @dim(NUM_DOFS_1D, NUM_DOFS_1D, NUM_DOFS_1D, numElements);
|
||||
@@ -0,0 +1,168 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double *quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator2D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J12*J12 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J12*J11 + J22*J21); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J21*J21); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD2D_t quadToDofD,
|
||||
@restrict const SymmOperator2D_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
const double gradX2 = (O11 * gradX) + (O12 * gradY);
|
||||
const double gradY2 = (O12 * gradX) + (O22 * gradY);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += ((gradX2 * quadToDofD(0, d, q)) +
|
||||
(gradY2 * quadToDofD(1, d, q)));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator3D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3) + (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3) + (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD3D_t quadToDofD,
|
||||
@restrict const SymmOperator3D_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double gradX = 0, gradY = 0, gradZ = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
gradZ += s * quadToDofD(2, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double gradX2 = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
const double gradY2 = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
const double gradZ2 = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += ((gradX2 * quadToDofD(0, d, q)) +
|
||||
(gradY2 * quadToDofD(1, d, q)) +
|
||||
(gradZ2 * quadToDofD(2, d, q)));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,182 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator2D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J12*J12 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J12*J11 + J22*J21); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J21*J21); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD2D_t quadToDofD,
|
||||
@restrict const SymmOperator2D_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gradX[NUM_QUAD];
|
||||
@shared double s_gradY[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
s_gradX[q] = (O11 * gradX) + (O12 * gradY);
|
||||
s_gradY[q] = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
// FIXME: s_gradX and s_gradY are @shared used outside of @inner
|
||||
r_sol += ((s_gradX[q] * quadToDofD(0, d, q)) +
|
||||
(s_gradY[q] * quadToDofD(1, d, q)));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator3D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2) + (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3) + (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3) + (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD3D_t quadToDofD,
|
||||
@restrict const SymmOperator3D_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gradX[NUM_QUAD];
|
||||
@shared double s_gradY[NUM_QUAD];
|
||||
@shared double s_gradZ[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
|
||||
double gradX = 0, gradY = 0, gradZ = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double s = solIn(d, e);
|
||||
gradX += s * quadToDofD(0, d, q);
|
||||
gradY += s * quadToDofD(1, d, q);
|
||||
gradZ += s * quadToDofD(2, d, q);
|
||||
}
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
s_gradX[q] = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
s_gradY[q] = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
s_gradZ[q] = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_sol += ((s_gradX[q] * quadToDofD(0, d, q)) +
|
||||
(s_gradY[q] * quadToDofD(1, d, q)) +
|
||||
(s_gradZ[q] * quadToDofD(2, d, q)));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,370 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator1D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF / J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator1D_t oper,
|
||||
@restrict const DLocal1D_t solIn,
|
||||
@restrict DLocal1D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] *= oper(qx, e);
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gradX = grad[qx];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, e) += gradX * quadToDofD(dx, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator2D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_2D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J21*J21 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J21*J11 + J22*J12); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J12*J12); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator2D_t oper,
|
||||
@restrict const DLocal2D_t solIn,
|
||||
@restrict DLocal2D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D][NUM_QUAD_1D][2];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qy][qx][0] = 0;
|
||||
grad[qy][qx][1] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double gradX[NUM_QUAD_1D][2];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] = 0;
|
||||
gradX[qx][1] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] += s * dofToQuad(qx, dx);
|
||||
gradX[qx][1] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
const double wDy = dofToQuadD(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qy][qx][0] += gradX[qx][1] * wy;
|
||||
grad[qy][qx][1] += gradX[qx][0] * wDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate Dxy, xDy in plane
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_2D_ID(qx, qy);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
const double gradX = grad[qy][qx][0];
|
||||
const double gradY = grad[qy][qx][1];
|
||||
|
||||
grad[qy][qx][0] = (O11 * gradX) + (O12 * gradY);
|
||||
grad[qy][qx][1] = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double gradX[NUM_DOFS_1D][2];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX[dx][0] = 0;
|
||||
gradX[dx][1] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gX = grad[qy][qx][0];
|
||||
const double gY = grad[qy][qx][1];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = quadToDof(dx, qx);
|
||||
const double wDx = quadToDofD(dx, qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
const double wDy = quadToDofD(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, e) += ((gradX[dx][0] * wy) +
|
||||
(gradX[dx][1] * wDy));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict SymmOperator3D_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_3D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3), (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3), (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator3D_t oper,
|
||||
@restrict const DLocal3D_t solIn,
|
||||
@restrict DLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double grad[NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D][4];
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qz][qy][qx][0] = 0;
|
||||
grad[qz][qy][qx][1] = 0;
|
||||
grad[qz][qy][qx][2] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double gradXY[NUM_QUAD_1D][NUM_QUAD_1D][4];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradXY[qy][qx][0] = 0;
|
||||
gradXY[qy][qx][1] = 0;
|
||||
gradXY[qy][qx][2] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double gradX[NUM_QUAD_1D][2];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] = 0;
|
||||
gradX[qx][1] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
gradX[qx][0] += s * dofToQuad(qx, dx);
|
||||
gradX[qx][1] += s * dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
const double wDy = dofToQuadD(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double wx = gradX[qx][0];
|
||||
const double wDx = gradX[qx][1];
|
||||
gradXY[qy][qx][0] += wDx * wy;
|
||||
gradXY[qy][qx][1] += wx * wDy;
|
||||
gradXY[qy][qx][2] += wx * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
const double wDz = dofToQuadD(qz, dz);
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qz][qy][qx][0] += gradXY[qy][qx][0] * wz;
|
||||
grad[qz][qy][qx][1] += gradXY[qy][qx][1] * wz;
|
||||
grad[qz][qy][qx][2] += gradXY[qy][qx][2] * wDz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_3D_ID(qx, qy, qz);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double gradX = grad[qz][qy][qx][0];
|
||||
const double gradY = grad[qz][qy][qx][1];
|
||||
const double gradZ = grad[qz][qy][qx][2];
|
||||
|
||||
grad[qz][qy][qx][0] = (O11 * gradX) + (O12 * gradY) + (O13 * gradZ);
|
||||
grad[qz][qy][qx][1] = (O12 * gradX) + (O22 * gradY) + (O23 * gradZ);
|
||||
grad[qz][qy][qx][2] = (O13 * gradX) + (O23 * gradY) + (O33 * gradZ);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
double gradXY[NUM_DOFS_1D][NUM_DOFS_1D][4];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradXY[dy][dx][0] = 0;
|
||||
gradXY[dy][dx][1] = 0;
|
||||
gradXY[dy][dx][2] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double gradX[NUM_DOFS_1D][4];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX[dx][0] = 0;
|
||||
gradX[dx][1] = 0;
|
||||
gradX[dx][2] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double gX = grad[qz][qy][qx][0];
|
||||
const double gY = grad[qz][qy][qx][1];
|
||||
const double gZ = grad[qz][qy][qx][2];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = quadToDof(dx, qx);
|
||||
const double wDx = quadToDofD(dx, qx);
|
||||
gradX[dx][0] += gX * wDx;
|
||||
gradX[dx][1] += gY * wx;
|
||||
gradX[dx][2] += gZ * wx;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
const double wDy = quadToDofD(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradXY[dy][dx][0] += gradX[dx][0] * wy;
|
||||
gradXY[dy][dx][1] += gradX[dx][1] * wDy;
|
||||
gradXY[dy][dx][2] += gradX[dx][2] * wy;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = quadToDof(dz, qz);
|
||||
const double wDz = quadToDofD(dz, qz);
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, dz, e) += ((gradXY[dy][dx][0] * wz) +
|
||||
(gradXY[dy][dx][1] * wz) +
|
||||
(gradXY[dy][dx][2] * wDz));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,435 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator1D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A1_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A1_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF / J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator1D_t oper,
|
||||
@restrict const DLocal1D_t solIn,
|
||||
@restrict DLocal1D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double grad[NUM_QUAD_1D];
|
||||
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
|
||||
s_dofToQuadD[i] = dofToQuadD[i];
|
||||
s_quadToDofD[i] = quadToDofD[i];
|
||||
}
|
||||
}
|
||||
|
||||
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
|
||||
if (e < numElements) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] += s * s_dofToQuadD(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
grad[qx] *= oper(qx, e);
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += grad[qx] * s_quadToDofD(dx, qx);
|
||||
}
|
||||
solOut(dx, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator2D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_2D; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(0, q, e) = c_detJ * (J21*J21 + J22*J22); // (1,1)
|
||||
oper(1, q, e) = -c_detJ * (J21*J11 + J22*J12); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (J11*J11 + J12*J12); // (2,2)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator2D_t oper,
|
||||
@restrict const DLocal2D_t solIn,
|
||||
@restrict DLocal2D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_xDy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_grad[2 * NUM_QUAD_2D] @dim(2, NUM_QUAD_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double r_x[NUM_MAX_1D];
|
||||
@exclusive double r_y[NUM_QUAD_1D];
|
||||
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_dofToQuadD[id] = dofToQuadD[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
s_quadToDofD[id] = quadToDofD[id];
|
||||
}
|
||||
}
|
||||
|
||||
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
|
||||
if (e < numElements) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
s_xy(dx, qy) = 0;
|
||||
s_xDy(dx, qy) = 0;
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
r_x[dy] = solIn(dx, dy, e);
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double xy = 0;
|
||||
double xDy = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
xy += r_x[dy] * s_dofToQuad(qy, dy);
|
||||
xDy += r_x[dy] * s_dofToQuadD(qy, dy);
|
||||
}
|
||||
s_xy(dx, qy) = xy;
|
||||
s_xDy(dx, qy) = xDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
if (qy < NUM_QUAD_1D) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
double gradX = 0, gradY = 0;
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
gradX += s_xy(dx, qy) * s_dofToQuadD(qx, dx);
|
||||
gradY += s_xDy(dx, qy) * s_dofToQuad(qx, dx);
|
||||
}
|
||||
|
||||
const int q = QUAD_2D_ID(qx, qy);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O22 = oper(2, q, e);
|
||||
|
||||
s_grad(0, qx, qy) = (O11 * gradX) + (O12 * gradY);
|
||||
s_grad(1, qx, qy) = (O12 * gradX) + (O22 * gradY);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx; @inner) {
|
||||
if (qx < NUM_QUAD_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
s_xy(dy, qx) = 0;
|
||||
s_xDy(dy, qx) = 0;
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
r_x[qy] = s_grad(0, qx, qy);
|
||||
r_y[qy] = s_grad(1, qx, qy);
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double xy = 0;
|
||||
double xDy = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
xy += r_x[qy] * s_quadToDof(dy, qy);
|
||||
xDy += r_y[qy] * s_quadToDofD(dy, qy);
|
||||
}
|
||||
s_xy(dy, qx) = xy;
|
||||
s_xDy(dy, qx) = xDy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += ((s_xy(dy, qx) * s_quadToDofD(dx, qx)) +
|
||||
(s_xDy(dy, qx) * s_quadToDof(dx, qx)));
|
||||
}
|
||||
solOut(dx, dy, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
SymmOperator3D_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_3D; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
const double c_detJ = quadWeights[q] * COEFF / detJ;
|
||||
|
||||
// adj(J)
|
||||
const double A11 = (J22 * J33) - (J23 * J32);
|
||||
const double A12 = (J23 * J31) - (J21 * J33);
|
||||
const double A13 = (J21 * J32) - (J22 * J31);
|
||||
|
||||
const double A21 = (J13 * J32) - (J12 * J33);
|
||||
const double A22 = (J11 * J33) - (J13 * J31);
|
||||
const double A23 = (J12 * J31) - (J11 * J32);
|
||||
|
||||
const double A31 = (J12 * J23) - (J13 * J22);
|
||||
const double A32 = (J13 * J21) - (J11 * J23);
|
||||
const double A33 = (J11 * J22) - (J12 * J21);
|
||||
|
||||
// adj(J)^Tadj(J)
|
||||
oper(0, q, e) = c_detJ * (A11*A11 + A21*A21 + A31*A31); // (1,1)
|
||||
oper(1, q, e) = c_detJ * (A11*A12 + A21*A22 + A31*A32); // (1,2), (2,1)
|
||||
oper(2, q, e) = c_detJ * (A11*A13 + A21*A23 + A31*A33); // (1,3), (3,1)
|
||||
oper(3, q, e) = c_detJ * (A12*A12 + A22*A22 + A32*A32); // (2,2)
|
||||
oper(4, q, e) = c_detJ * (A12*A13 + A22*A23 + A32*A33); // (2,3), (3,2)
|
||||
oper(5, q, e) = c_detJ * (A13*A13 + A23*A23 + A33*A33); // (3,3)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const SymmOperator3D_t oper,
|
||||
@restrict const DLocal3D_t solIn,
|
||||
@restrict DLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_dofToQuadD[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_quadToDofD[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_z[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
@shared double s_Dz[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
@shared double s_xyDz[NUM_QUAD_2D] @dim(NUM_QUAD_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store z axis as registers
|
||||
@exclusive double r_qz[NUM_QUAD_1D];
|
||||
@exclusive double r_qDz[NUM_QUAD_1D];
|
||||
@exclusive double r_dDxyz[NUM_DOFS_1D];
|
||||
@exclusive double r_dxDyz[NUM_DOFS_1D];
|
||||
@exclusive double r_dxyDz[NUM_DOFS_1D];
|
||||
|
||||
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
const int id = (y * NUM_MAX_1D) + x;
|
||||
// Fetch Q <--> D maps
|
||||
if (id < NUM_QUAD_DOFS_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_dofToQuadD[id] = dofToQuadD[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
s_quadToDofD[id] = quadToDofD[id];
|
||||
}
|
||||
// Initialize our Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] = 0;
|
||||
r_qDz[qz] = 0;
|
||||
}
|
||||
// Initialize our solution updates in the Z axis
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
r_dDxyz[dz] = 0;
|
||||
r_dxDyz[dz] = 0;
|
||||
r_dxyDz[dz] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
// Calculate D -> Q in the Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] += s * s_dofToQuad(qz, dz);
|
||||
r_qDz[qz] += s * s_dofToQuadD(qz, dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// For each xy plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
// Fill xy plane at given z position
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
s_z(dx, dy) = r_qz[qz];
|
||||
s_Dz(dx, dy) = r_qDz[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
double Dxyz = 0;
|
||||
double xDyz = 0;
|
||||
double xyDz = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = s_dofToQuad(qy, dy);
|
||||
const double wDy = s_dofToQuadD(qy, dy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = s_dofToQuad(qx, dx);
|
||||
const double wDx = s_dofToQuadD(qx, dx);
|
||||
const double z = s_z(dx, dy);
|
||||
const double Dz = s_Dz(dx, dy);
|
||||
Dxyz += wDx * wy * z;
|
||||
xDyz += wx * wDy * z;
|
||||
xyDz += wx * wy * Dz;
|
||||
}
|
||||
}
|
||||
|
||||
const int q = QUAD_3D_ID(qx, qy, qz);
|
||||
const double O11 = oper(0, q, e);
|
||||
const double O12 = oper(1, q, e);
|
||||
const double O13 = oper(2, q, e);
|
||||
const double O22 = oper(3, q, e);
|
||||
const double O23 = oper(4, q, e);
|
||||
const double O33 = oper(5, q, e);
|
||||
|
||||
const double qDxyz = (O11 * Dxyz) + (O12 * xDyz) + (O13 * xyDz);
|
||||
const double qxDyz = (O12 * Dxyz) + (O22 * xDyz) + (O23 * xyDz);
|
||||
const double qxyDz = (O13 * Dxyz) + (O23 * xDyz) + (O33 * xyDz);
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = s_quadToDof(dz, qz);
|
||||
const double wDz = s_quadToDofD(dz, qz);
|
||||
r_dDxyz[dz] += wz * qDxyz;
|
||||
r_dxDyz[dz] += wz * qxDyz;
|
||||
r_dxyDz[dz] += wDz * qxyDz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@barrier("s_z_s_Dz_sync_1");
|
||||
}
|
||||
// Iterate over xy planes to compute solution
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
// Place xy plane in shared memory
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
s_z(qx, qy) = r_dDxyz[dz];
|
||||
s_Dz(qx, qy) = r_dxDyz[dz];
|
||||
s_xyDz(qx, qy) = r_dxyDz[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Finalize solution in xy plane
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
double solZ = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = s_quadToDof(dy, qy);
|
||||
const double wDy = s_quadToDofD(dy, qy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double wx = s_quadToDof(dx, qx);
|
||||
const double wDx = s_quadToDofD(dx, qx);
|
||||
const double Dxyz = s_z(qx, qy);
|
||||
const double xDyz = s_Dz(qx, qy);
|
||||
const double xyDz = s_xyDz(qx, qy);
|
||||
solZ += ((wDx * wy * Dxyz) +
|
||||
(wx * wDy * xDyz) +
|
||||
(wx * wy * xyDz));
|
||||
}
|
||||
}
|
||||
solOut(dx, dy, dz, e) += solZ;
|
||||
}
|
||||
}
|
||||
}
|
||||
@barrier("s_z_s_Dz_s_xyDz_sync_1");
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,167 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "url_handler.hpp"
|
||||
#include "bilinearform.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
bool Engine::fileOpenerRegistered = false;
|
||||
|
||||
void Engine::Init(const std::string &engine_spec)
|
||||
{
|
||||
//
|
||||
// Initialize inherited fields
|
||||
//
|
||||
memory_resources[0] = NULL;
|
||||
workers_weights[0]= 1.0;
|
||||
workers_mem_res[0] = 0;
|
||||
|
||||
//
|
||||
// Initialize the OCCA engine
|
||||
//
|
||||
::occa::properties props(engine_spec);
|
||||
device = new ::occa::device[1];
|
||||
device[0].setup(props);
|
||||
|
||||
okl_path = "mfem-occa://";
|
||||
if (!fileOpenerRegistered)
|
||||
{
|
||||
// The directories from "MFEM_OCCA_OKL_PATH", if any, have the highest
|
||||
// priority.
|
||||
FileOpener *fo = new FileOpener("mfem-occa://", "MFEM_OCCA_OKL_PATH");
|
||||
// Next in priority is the source path, if it exists.
|
||||
std::string mfem_src_prefix = mfem::GetSourcePath();
|
||||
fo->AddDir(mfem_src_prefix + "/backends/occa");
|
||||
// And last in priority is the install path, if it exists.
|
||||
std::string mfem_install_prefix = mfem::GetInstallPath();
|
||||
fo->AddDir(mfem_install_prefix + "/lib/mfem/occa");
|
||||
::occa::io::fileOpener::add(fo);
|
||||
fileOpenerRegistered = true;
|
||||
}
|
||||
// std::cout << "OCCA device properties:\n" << device[0].properties();
|
||||
|
||||
force_cuda_aware_mpi = false;
|
||||
}
|
||||
|
||||
Engine::Engine(const std::string &engine_spec)
|
||||
: mfem::Engine(NULL, 1, 1)
|
||||
{
|
||||
Init(engine_spec);
|
||||
}
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
Engine::Engine(MPI_Comm _comm, const std::string &engine_spec)
|
||||
: mfem::Engine(NULL, 1, 1)
|
||||
{
|
||||
comm = _comm;
|
||||
Init(engine_spec);
|
||||
}
|
||||
#endif
|
||||
|
||||
bool Engine::CheckEngine(const mfem::Engine *engine) const
|
||||
{
|
||||
return (engine != NULL && util::Is<const Engine>(engine) != NULL &&
|
||||
*util::As<const Engine>(engine) == *this);
|
||||
}
|
||||
|
||||
bool Engine::CheckLayout(const PLayout *layout) const
|
||||
{
|
||||
return (layout != NULL && util::Is<const Layout>(layout) != NULL &&
|
||||
layout->As<Layout>().OccaEngine() == *this);
|
||||
}
|
||||
|
||||
bool Engine::CheckArray(const PArray *array) const
|
||||
{
|
||||
return (array != NULL && util::Is<const Array>(array) != NULL &&
|
||||
array->As<Array>().OccaEngine() == *this);
|
||||
}
|
||||
|
||||
bool Engine::CheckVector(const PVector *vector) const
|
||||
{
|
||||
return (vector != NULL && util::Is<const Vector>(vector) != NULL &&
|
||||
vector->As<Vector>().OccaEngine() == *this);
|
||||
}
|
||||
|
||||
bool Engine::CheckFESpace(const PFiniteElementSpace *fes) const
|
||||
{
|
||||
return (fes != NULL && util::Is<const FiniteElementSpace>(fes) != NULL &&
|
||||
fes->As<FiniteElementSpace>().OccaEngine() == *this);
|
||||
}
|
||||
|
||||
DLayout Engine::MakeLayout(std::size_t size) const
|
||||
{
|
||||
return DLayout(new Layout(*this, size));
|
||||
}
|
||||
|
||||
DLayout Engine::MakeLayout(const mfem::Array<std::size_t> &offsets) const
|
||||
{
|
||||
MFEM_ASSERT(offsets.Size() == 2,
|
||||
"multiple workers are not supported yet");
|
||||
return DLayout(new Layout(*this, offsets.Last()));
|
||||
}
|
||||
|
||||
DArray Engine::MakeArray(PLayout &layout, std::size_t item_size) const
|
||||
{
|
||||
return DArray(new Array(layout.As<Layout>(), item_size));
|
||||
}
|
||||
|
||||
DVector Engine::MakeVector(PLayout &layout, int type_id) const
|
||||
{
|
||||
MFEM_ASSERT(type_id == ScalarId<double>::value, "type_id " << type_id
|
||||
<< " is not supported");
|
||||
return DVector(new Vector(layout.As<Layout>()));
|
||||
}
|
||||
|
||||
DFiniteElementSpace Engine::MakeFESpace(mfem::FiniteElementSpace &fespace) const
|
||||
{
|
||||
return DFiniteElementSpace(new FiniteElementSpace(*this, fespace));
|
||||
}
|
||||
|
||||
DBilinearForm Engine::MakeBilinearForm(mfem::BilinearForm &bf) const
|
||||
{
|
||||
return DBilinearForm(new BilinearForm(*this, bf));
|
||||
}
|
||||
|
||||
void Engine::AssembleLinearForm(LinearForm &l_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
}
|
||||
|
||||
mfem::Operator *Engine::MakeOperator(const MixedBilinearForm &mbl_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
mfem::Operator *Engine::MakeOperator(const NonlinearForm &nl_form) const
|
||||
{
|
||||
/// FIXME - What will the actual parameters be?
|
||||
MFEM_ABORT("FIXME");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,144 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
#define MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "../base/backend.hpp"
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Engine : public mfem::Engine
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// mfem::Backend *backend;
|
||||
#ifdef MFEM_USE_MPI
|
||||
// MPI_Comm comm;
|
||||
#endif
|
||||
// int num_mem_res;
|
||||
// int num_workers;
|
||||
// MemoryResource **memory_resources;
|
||||
// double *workers_weights;
|
||||
// int *workers_mem_res;
|
||||
|
||||
static bool fileOpenerRegistered;
|
||||
/// An array of OCCA devices. Currently only a single device is supported.
|
||||
::occa::device *device;
|
||||
std::string okl_path;
|
||||
bool force_cuda_aware_mpi;
|
||||
|
||||
void Init(const std::string &engine_spec);
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
Engine(const std::string &engine_spec);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
/// TODO: doxygen
|
||||
Engine(MPI_Comm comm, const std::string &engine_spec);
|
||||
#endif
|
||||
|
||||
/// TODO: doxygen
|
||||
virtual ~Engine() { delete [] device; }
|
||||
|
||||
/**
|
||||
@name OCCA specific interface, used by other objects in the OCCA backend
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Get the associated OCCA device.
|
||||
::occa::device GetDevice(int idx = 0) const { return device[idx]; }
|
||||
|
||||
/// TODO: doxygen
|
||||
const std::string &GetOklPath() const { return okl_path; }
|
||||
|
||||
/// OCCA device memory allocation.
|
||||
::occa::memory Alloc(std::size_t bytes) const
|
||||
{ return GetDevice().malloc(bytes); }
|
||||
|
||||
/// Two mfem::occa::Engine%s are equal if they use the same OCCA device.
|
||||
bool operator==(const Engine &other) const
|
||||
{ return GetDevice() == other.GetDevice(); }
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckEngine(const mfem::Engine *e) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckLayout(const PLayout *layout) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckArray(const PArray *array) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckVector(const PVector *vector) const;
|
||||
|
||||
/// TODO: doxygen
|
||||
bool CheckFESpace(const PFiniteElementSpace *fes) const;
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
void SetForceCudaAwareMPI(bool force = true)
|
||||
{ force_cuda_aware_mpi = force; }
|
||||
|
||||
bool GetForceCudaAwareMPI() const { return force_cuda_aware_mpi; }
|
||||
#endif
|
||||
|
||||
///@}
|
||||
// End: OCCA specific interface
|
||||
|
||||
/**
|
||||
@name Virtual interface: finite element data structures and algorithms
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual DLayout MakeLayout(std::size_t size) const;
|
||||
virtual DLayout MakeLayout(const mfem::Array<std::size_t> &offsets) const;
|
||||
|
||||
virtual DArray MakeArray(PLayout &layout, std::size_t item_size) const;
|
||||
|
||||
virtual DVector MakeVector(PLayout &layout,
|
||||
int type_id = ScalarId<double>::value) const;
|
||||
|
||||
virtual DFiniteElementSpace MakeFESpace(mfem::FiniteElementSpace &
|
||||
fespace) const;
|
||||
|
||||
virtual DBilinearForm MakeBilinearForm(mfem::BilinearForm &bf) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual void AssembleLinearForm(LinearForm &l_form) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual mfem::Operator *MakeOperator(const MixedBilinearForm &mbl_form) const;
|
||||
|
||||
/// FIXME - What will the actual parameters be?
|
||||
virtual mfem::Operator *MakeOperator(const NonlinearForm &nl_form) const;
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_ENGINE_HPP
|
||||
@@ -0,0 +1,532 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "backend.hpp"
|
||||
#include "fespace.hpp"
|
||||
#include "interpolation.hpp"
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
#ifdef OMPI_RELEASE_VERSION
|
||||
#include <mpi-ext.h> // Check for cuda support
|
||||
#endif
|
||||
#endif
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
FiniteElementSpace::FiniteElementSpace(const Engine &e,
|
||||
mfem::FiniteElementSpace &fespace)
|
||||
: PFiniteElementSpace(e, fespace),
|
||||
e_layout(new Layout(e, 0)) // resized in SetupLocalGlobalMaps()
|
||||
{
|
||||
vdim = fespace.GetVDim();
|
||||
ordering = fespace.GetOrdering();
|
||||
|
||||
SetupLocalGlobalMaps();
|
||||
SetupOperators(); // calls virtual methods of 'fes'
|
||||
SetupKernels();
|
||||
}
|
||||
|
||||
FiniteElementSpace::~FiniteElementSpace()
|
||||
{
|
||||
delete restrictionOp;
|
||||
delete prolongationOp;
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupLocalGlobalMaps()
|
||||
{
|
||||
const int elements = fes->GetNE();
|
||||
|
||||
if (elements == 0) { return; }
|
||||
|
||||
// Assuming of finite elements are the same.
|
||||
const mfem::FiniteElement &fe = *fes->GetFE(0);
|
||||
const mfem::TensorBasisElement *el =
|
||||
dynamic_cast<const mfem::TensorBasisElement*>(&fe);
|
||||
|
||||
const mfem::Table &e2dTable = fes->GetElementToDofTable();
|
||||
const int *elementMap = e2dTable.GetJ();
|
||||
|
||||
globalDofs = fes->GetNDofs();
|
||||
localDofs = fe.GetDof();
|
||||
|
||||
e_layout->OccaResize(e2dTable.Size_of_connections());
|
||||
|
||||
int *elementDofMap = new int[localDofs];
|
||||
if (el)
|
||||
{
|
||||
::memcpy(elementDofMap,
|
||||
el->GetDofMap().GetData(),
|
||||
localDofs * sizeof(int));
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < localDofs; ++i)
|
||||
{
|
||||
elementDofMap[i] = i;
|
||||
}
|
||||
}
|
||||
|
||||
// Allocate device offsets and indices
|
||||
globalToLocalOffsets.allocate(GetDevice(),
|
||||
globalDofs + 1);
|
||||
globalToLocalIndices.allocate(GetDevice(),
|
||||
localDofs, elements);
|
||||
localToGlobalMap.allocate(GetDevice(),
|
||||
localDofs, elements);
|
||||
|
||||
int *offsets = globalToLocalOffsets.ptr();
|
||||
int *indices = globalToLocalIndices.ptr();
|
||||
int *l2gMap = localToGlobalMap.ptr();
|
||||
|
||||
// We'll be keeping a count of how many local nodes point
|
||||
// to its global dof
|
||||
for (int i = 0; i <= globalDofs; ++i)
|
||||
{
|
||||
offsets[i] = 0;
|
||||
}
|
||||
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
MFEM_ASSERT(e2dTable.RowSize(e) == localDofs, "");
|
||||
for (int d = 0; d < localDofs; ++d)
|
||||
{
|
||||
const int gid = elementMap[localDofs*e + d];
|
||||
++offsets[gid + 1];
|
||||
}
|
||||
}
|
||||
// Aggregate to find offsets for each global dof
|
||||
for (int i = 1; i <= globalDofs; ++i)
|
||||
{
|
||||
offsets[i] += offsets[i - 1];
|
||||
}
|
||||
// For each global dof, fill in all local nodes that point
|
||||
// to it
|
||||
for (int e = 0; e < elements; ++e)
|
||||
{
|
||||
for (int d = 0; d < localDofs; ++d)
|
||||
{
|
||||
const int gid = elementMap[localDofs*e + elementDofMap[d]];
|
||||
const int lid = localDofs*e + d;
|
||||
indices[offsets[gid]++] = lid;
|
||||
l2gMap[lid] = gid;
|
||||
}
|
||||
}
|
||||
// We shifted the offsets vector by 1 by using it
|
||||
// as a counter. Now we shift it back.
|
||||
for (int i = globalDofs; i > 0; --i)
|
||||
{
|
||||
offsets[i] = offsets[i - 1];
|
||||
}
|
||||
offsets[0] = 0;
|
||||
|
||||
delete [] elementDofMap;
|
||||
|
||||
globalToLocalOffsets.keepInDevice();
|
||||
globalToLocalIndices.keepInDevice();
|
||||
localToGlobalMap.keepInDevice();
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupOperators() const
|
||||
{
|
||||
// Construct 'restrictionOp' and 'prolongationOp'.
|
||||
|
||||
prolongationOp = restrictionOp = NULL;
|
||||
|
||||
const mfem::SparseMatrix *R = fes->GetRestrictionMatrix();
|
||||
const mfem::Operator *P = fes->GetProlongationMatrix();
|
||||
|
||||
if (!P) { return; }
|
||||
|
||||
Layout &v_layout = OccaVLayout();
|
||||
Layout &t_layout = OccaTrueVLayout();
|
||||
|
||||
// Assuming R has one entry per row equal to 1.
|
||||
MFEM_ASSERT(R->Finalized(), "");
|
||||
const int tdofs = R->Height();
|
||||
MFEM_ASSERT(tdofs == (int)t_layout.Size(), "");
|
||||
MFEM_ASSERT(tdofs == R->GetI()[tdofs], "");
|
||||
::occa::array<int> ltdof_ldof(GetDevice(), tdofs, R->GetJ());
|
||||
ltdof_ldof.keepInDevice();
|
||||
|
||||
restrictionOp = new RestrictionOperator(v_layout, t_layout, ltdof_ldof);
|
||||
|
||||
const mfem::SparseMatrix *pmat = dynamic_cast<const mfem::SparseMatrix*>(P);
|
||||
if (pmat)
|
||||
{
|
||||
const mfem::SparseMatrix *pmatT = Transpose(*pmat);
|
||||
|
||||
OccaSparseMatrix *occaP =
|
||||
CreateMappedSparseMatrix(t_layout, v_layout, *pmat);
|
||||
OccaSparseMatrix *occaPT =
|
||||
CreateMappedSparseMatrix(v_layout, t_layout, *pmatT);
|
||||
|
||||
prolongationOp = new ProlongationOperator(*occaP, *occaPT);
|
||||
|
||||
delete occaPT;
|
||||
delete occaP;
|
||||
}
|
||||
#ifdef MFEM_USE_MPI
|
||||
else if (fes->Conforming() && dynamic_cast<ParFiniteElementSpace*>(fes))
|
||||
{
|
||||
ParFiniteElementSpace *pfes = static_cast<ParFiniteElementSpace*>(fes);
|
||||
prolongationOp = new OccaConformingProlongation(*this, *pfes,
|
||||
ltdof_ldof.memory());
|
||||
}
|
||||
#endif
|
||||
else
|
||||
{
|
||||
prolongationOp = new ProlongationOperator(t_layout, v_layout, P);
|
||||
}
|
||||
}
|
||||
|
||||
void FiniteElementSpace::SetupKernels()
|
||||
{
|
||||
::occa::properties props("defines: {"
|
||||
" TILESIZE: 256,"
|
||||
"}");
|
||||
props["defines/NUM_VDIM"] = vdim;
|
||||
|
||||
props["defines/ORDERING_BY_NODES"] = 0;
|
||||
props["defines/ORDERING_BY_VDIM"] = 1;
|
||||
props["defines/VDIM_ORDERING"] = (int) (ordering == Ordering::byVDIM);
|
||||
|
||||
::occa::device device = GetDevice();
|
||||
const std::string &okl_path = OccaEngine().GetOklPath();
|
||||
globalToLocalKernel = device.buildKernel(okl_path + "fespace.okl",
|
||||
"GlobalToLocal",
|
||||
props);
|
||||
localToGlobalKernel = device.buildKernel(okl_path + "fespace.okl",
|
||||
"LocalToGlobal",
|
||||
props);
|
||||
}
|
||||
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
OccaConformingProlongation::OccaConformingProlongation(
|
||||
const FiniteElementSpace &ofes, const mfem::ParFiniteElementSpace &pfes,
|
||||
::occa::memory ltdof_ldof_)
|
||||
|
||||
: Operator(ofes.OccaTrueVLayout(), ofes.OccaVLayout()),
|
||||
shr_ltdof(ofes.OccaEngine()),
|
||||
ext_ldof(ofes.OccaEngine()),
|
||||
shr_buf(shr_ltdof.OccaLayout(), sizeof(double)),
|
||||
ext_buf(ext_ldof.OccaLayout(), sizeof(double)),
|
||||
shr_buf_offsets(NULL), ext_buf_offsets(NULL),
|
||||
ltdof_ldof(ltdof_ldof_),
|
||||
gc(pfes.GroupComm())
|
||||
{
|
||||
MFEM_ASSERT(pfes.Conforming(), "internal error");
|
||||
|
||||
const Engine &engine = ofes.OccaEngine();
|
||||
const std::string &okl_path = engine.GetOklPath();
|
||||
::occa::device device = engine.GetDevice();
|
||||
|
||||
{
|
||||
Table nbr_ltdof;
|
||||
gc.GetNeighborLTDofTable(nbr_ltdof);
|
||||
shr_ltdof.OccaResize(nbr_ltdof.Size_of_connections(), sizeof(int));
|
||||
shr_ltdof.OccaPush(nbr_ltdof.GetJ());
|
||||
shr_buf.OccaResize(&shr_ltdof.OccaLayout(), sizeof(double));
|
||||
shr_buf_offsets = nbr_ltdof.GetI();
|
||||
{
|
||||
mfem::Array<int> shr_ltdof(nbr_ltdof.GetJ(),
|
||||
nbr_ltdof.Size_of_connections());
|
||||
mfem::Array<int> unique_ltdof(shr_ltdof);
|
||||
unique_ltdof.Sort();
|
||||
unique_ltdof.Unique();
|
||||
// Note: the next loop modifies the J array of nbr_ltdof
|
||||
for (int i = 0; i < shr_ltdof.Size(); i++)
|
||||
{
|
||||
shr_ltdof[i] = unique_ltdof.FindSorted(shr_ltdof[i]);
|
||||
MFEM_ASSERT(shr_ltdof[i] != -1, "internal error");
|
||||
}
|
||||
Table unique_shr;
|
||||
Transpose(shr_ltdof, unique_shr, unique_ltdof.Size());
|
||||
|
||||
unq_ltdof = device.malloc(unique_ltdof.Size()*sizeof(int),
|
||||
unique_ltdof.GetData());
|
||||
unq_shr_i = device.malloc((unique_shr.Size()+1)*sizeof(int),
|
||||
unique_shr.GetI());
|
||||
unq_shr_j = device.malloc(unique_shr.Size_of_connections()*sizeof(int),
|
||||
unique_shr.GetJ());
|
||||
}
|
||||
delete [] nbr_ltdof.GetJ();
|
||||
nbr_ltdof.LoseData();
|
||||
}
|
||||
{
|
||||
Table nbr_ldof;
|
||||
gc.GetNeighborLDofTable(nbr_ldof);
|
||||
ext_ldof.OccaResize(nbr_ldof.Size_of_connections(), sizeof(int));
|
||||
ext_ldof.OccaPush(nbr_ldof.GetJ());
|
||||
ext_buf.OccaResize(&ext_ldof.OccaLayout(), sizeof(double));
|
||||
ext_buf_offsets = nbr_ldof.GetI();
|
||||
delete [] nbr_ldof.GetJ();
|
||||
nbr_ldof.LoseData();
|
||||
}
|
||||
host_shr_buf = NULL;
|
||||
host_ext_buf = NULL;
|
||||
// If the device has a separate memory space (e.g. CUDA device) and the MPI
|
||||
// library does not support buffers in that separate memory space, we
|
||||
// allocate separate host buffers to use for MPI communication.
|
||||
if (device.hasSeparateMemorySpace())
|
||||
{
|
||||
bool need_host_buf = true;
|
||||
if (device.mode() == "CUDA")
|
||||
{
|
||||
#ifdef MPIX_CUDA_AWARE_SUPPORT
|
||||
need_host_buf = !MPIX_Query_cuda_support();
|
||||
#endif
|
||||
if (engine.GetForceCudaAwareMPI()) { need_host_buf = false; }
|
||||
if (gc.GetGroupTopology().MyRank() == 0)
|
||||
{
|
||||
mfem::out << "\nOccaConformingProlongation: CUDA-aware MPI: "
|
||||
<< (need_host_buf ? "NO" : "YES") << "\n\n";
|
||||
}
|
||||
}
|
||||
if (need_host_buf)
|
||||
{
|
||||
host_shr_buf = new char[shr_buf.OccaMem().size()];
|
||||
host_ext_buf = new char[ext_buf.OccaMem().size()];
|
||||
}
|
||||
}
|
||||
|
||||
ExtractSubVector = device.buildKernel(okl_path + "mappings.okl",
|
||||
"ExtractSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
SetSubVector = device.buildKernel(okl_path + "mappings.okl",
|
||||
"SetSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
AddSubVector = device.buildKernel(okl_path + "mappings.okl",
|
||||
"AddSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
|
||||
const GroupTopology >opo = gc.GetGroupTopology();
|
||||
int req_counter = 0;
|
||||
for (int nbr = 1; nbr < gtopo.GetNumNeighbors(); nbr++)
|
||||
{
|
||||
const int send_offset = shr_buf_offsets[nbr];
|
||||
const int send_size = shr_buf_offsets[nbr+1] - send_offset;
|
||||
if (send_size > 0) { req_counter++; }
|
||||
|
||||
const int recv_offset = ext_buf_offsets[nbr];
|
||||
const int recv_size = ext_buf_offsets[nbr+1] - recv_offset;
|
||||
if (recv_size > 0) { req_counter++; }
|
||||
}
|
||||
requests = new MPI_Request[req_counter];
|
||||
}
|
||||
|
||||
OccaConformingProlongation::~OccaConformingProlongation()
|
||||
{
|
||||
delete [] requests;
|
||||
delete [] host_ext_buf;
|
||||
delete [] host_shr_buf;
|
||||
delete [] ext_buf_offsets;
|
||||
delete [] shr_buf_offsets;
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::BcastBeginCopy(const ::occa::memory &src,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// shr_buf[i] = src[shr_ltdof[i]]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (shr_ltdof.Size() == 0) { return; }
|
||||
ExtractSubVector((int)shr_ltdof.Size(), shr_ltdof.OccaMem(), src,
|
||||
shr_buf.OccaMem());
|
||||
// If the above kernel is executed asynchronously, wait for it to complete:
|
||||
shr_buf.OccaMem().getDevice().finish();
|
||||
if (host_shr_buf)
|
||||
{
|
||||
shr_buf.OccaMem().copyTo(host_shr_buf);
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::BcastLocalCopy(const ::occa::memory &src,
|
||||
::occa::memory &dst,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// dst[ltdof_ldof[i]] = src[i]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (ltdof_ldof.size<int>() == 0) { return; }
|
||||
SetSubVector((int)ltdof_ldof.size<int>(), ltdof_ldof, src, dst);
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::BcastEndCopy(::occa::memory &dst,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// dst[ext_ldof[i]] = ext_buf[i]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (ext_ldof.Size() == 0) { return; }
|
||||
if (host_ext_buf)
|
||||
{
|
||||
ext_buf.OccaMem().copyFrom(host_ext_buf);
|
||||
}
|
||||
SetSubVector((int)ext_ldof.Size(), ext_ldof.OccaMem(),
|
||||
ext_buf.OccaMem(), dst);
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::ReduceBeginCopy(const ::occa::memory &src,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// ext_buf[i] = src[ext_ldof[i]]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (ext_ldof.Size() == 0) { return; }
|
||||
ExtractSubVector((int)ext_ldof.Size(), ext_ldof.OccaMem(), src,
|
||||
ext_buf.OccaMem());
|
||||
// If the above kernel is executed asynchronously, wait for it to complete:
|
||||
ext_buf.OccaMem().getDevice().finish();
|
||||
if (host_ext_buf)
|
||||
{
|
||||
ext_buf.OccaMem().copyTo(host_ext_buf);
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::ReduceLocalCopy(const ::occa::memory &src,
|
||||
::occa::memory &dst,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// dst[i] = src[ltdof_ldof[i]]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (ltdof_ldof.size<int>() == 0) { return; }
|
||||
ExtractSubVector((int)ltdof_ldof.size<int>(), ltdof_ldof, src, dst);
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::ReduceEndAssemble(::occa::memory &dst,
|
||||
std::size_t item_size) const
|
||||
{
|
||||
// dst[shr_ltdof[i]] += shr_buf[i]
|
||||
MFEM_ASSERT(item_size == sizeof(double), "");
|
||||
if (unq_ltdof.size<int>() == 0) { return; }
|
||||
if (host_shr_buf)
|
||||
{
|
||||
shr_buf.OccaMem().copyFrom(host_shr_buf);
|
||||
}
|
||||
AddSubVector((int)unq_ltdof.size<int>(), unq_ltdof, unq_shr_i, unq_shr_j,
|
||||
shr_buf.OccaMem(), dst);
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
const GroupTopology >opo = gc.GetGroupTopology();
|
||||
|
||||
BcastBeginCopy(x.OccaMem(), sizeof(double)); // copy to 'shr_buf'
|
||||
|
||||
int req_counter = 0;
|
||||
for (int nbr = 1; nbr < gtopo.GetNumNeighbors(); nbr++)
|
||||
{
|
||||
const int send_offset = shr_buf_offsets[nbr];
|
||||
const int send_size = shr_buf_offsets[nbr+1] - send_offset;
|
||||
if (send_size > 0)
|
||||
{
|
||||
void *send_buf;
|
||||
if (host_shr_buf)
|
||||
{
|
||||
send_buf = host_shr_buf + send_offset*sizeof(double);
|
||||
}
|
||||
else
|
||||
{
|
||||
send_buf = (shr_buf.OccaMem() + send_offset*sizeof(double)).ptr();
|
||||
}
|
||||
MPI_Isend(send_buf, send_size, MPI_DOUBLE, gtopo.GetNeighborRank(nbr),
|
||||
41822, gtopo.GetComm(), &requests[req_counter++]);
|
||||
}
|
||||
|
||||
const int recv_offset = ext_buf_offsets[nbr];
|
||||
const int recv_size = ext_buf_offsets[nbr+1] - recv_offset;
|
||||
if (recv_size > 0)
|
||||
{
|
||||
void *recv_buf;
|
||||
if (host_ext_buf)
|
||||
{
|
||||
recv_buf = host_ext_buf + recv_offset*sizeof(double);
|
||||
}
|
||||
else
|
||||
{
|
||||
recv_buf = (ext_buf.OccaMem() + recv_offset*sizeof(double)).ptr();
|
||||
}
|
||||
MPI_Irecv(recv_buf, recv_size, MPI_DOUBLE, gtopo.GetNeighborRank(nbr),
|
||||
41822, gtopo.GetComm(), &requests[req_counter++]);
|
||||
}
|
||||
}
|
||||
|
||||
BcastLocalCopy(x.OccaMem(), y.OccaMem(), sizeof(double));
|
||||
|
||||
MPI_Waitall(req_counter, requests, MPI_STATUSES_IGNORE);
|
||||
|
||||
BcastEndCopy(y.OccaMem(), sizeof(double)); // copy from 'ext_buf'
|
||||
}
|
||||
|
||||
void OccaConformingProlongation::MultTranspose_(const Vector &x,
|
||||
Vector &y) const
|
||||
{
|
||||
const GroupTopology >opo = gc.GetGroupTopology();
|
||||
|
||||
ReduceBeginCopy(x.OccaMem(), sizeof(double)); // copy to 'ext_buf'
|
||||
|
||||
int req_counter = 0;
|
||||
for (int nbr = 1; nbr < gtopo.GetNumNeighbors(); nbr++)
|
||||
{
|
||||
const int send_offset = ext_buf_offsets[nbr];
|
||||
const int send_size = ext_buf_offsets[nbr+1] - send_offset;
|
||||
if (send_size > 0)
|
||||
{
|
||||
void *send_buf;
|
||||
if (host_ext_buf)
|
||||
{
|
||||
send_buf = host_ext_buf + send_offset*sizeof(double);
|
||||
}
|
||||
else
|
||||
{
|
||||
send_buf = (ext_buf.OccaMem() + send_offset*sizeof(double)).ptr();
|
||||
}
|
||||
MPI_Isend(send_buf, send_size, MPI_DOUBLE, gtopo.GetNeighborRank(nbr),
|
||||
41823, gtopo.GetComm(), &requests[req_counter++]);
|
||||
}
|
||||
|
||||
const int recv_offset = shr_buf_offsets[nbr];
|
||||
const int recv_size = shr_buf_offsets[nbr+1] - recv_offset;
|
||||
if (recv_size > 0)
|
||||
{
|
||||
void *recv_buf;
|
||||
if (host_shr_buf)
|
||||
{
|
||||
recv_buf = host_shr_buf + recv_offset*sizeof(double);
|
||||
}
|
||||
else
|
||||
{
|
||||
recv_buf = (shr_buf.OccaMem() + recv_offset*sizeof(double)).ptr();
|
||||
}
|
||||
MPI_Irecv(recv_buf, recv_size, MPI_DOUBLE, gtopo.GetNeighborRank(nbr),
|
||||
41823, gtopo.GetComm(), &requests[req_counter++]);
|
||||
}
|
||||
}
|
||||
|
||||
ReduceLocalCopy(x.OccaMem(), y.OccaMem(), sizeof(double));
|
||||
|
||||
MPI_Waitall(req_counter, requests, MPI_STATUSES_IGNORE);
|
||||
|
||||
ReduceEndAssemble(y.OccaMem(), sizeof(double)); // assemble from 'shr_buf'
|
||||
}
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,210 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
#define MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "engine.hpp"
|
||||
#include "operator.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
/// TODO: doxygen
|
||||
class FiniteElementSpace : public mfem::PFiniteElementSpace
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// mfem::FiniteElementSpace *fes;
|
||||
|
||||
SharedPtr<Layout> e_layout;
|
||||
|
||||
::occa::array<int> globalToLocalOffsets;
|
||||
::occa::array<int> globalToLocalIndices;
|
||||
::occa::array<int> localToGlobalMap;
|
||||
::occa::kernel globalToLocalKernel, localToGlobalKernel;
|
||||
|
||||
mfem::Ordering::Type ordering;
|
||||
|
||||
int globalDofs, localDofs;
|
||||
int vdim;
|
||||
|
||||
mutable Operator *prolongationOp, *restrictionOp;
|
||||
|
||||
void SetupLocalGlobalMaps();
|
||||
void SetupOperators() const; // calls virtual methods of 'fes' !!!
|
||||
void SetupKernels();
|
||||
|
||||
public:
|
||||
/// TODO: doxygen
|
||||
FiniteElementSpace(const Engine &e, mfem::FiniteElementSpace &fespace);
|
||||
|
||||
/// Virtual destructor
|
||||
virtual ~FiniteElementSpace();
|
||||
|
||||
/// TODO: doxygen
|
||||
const Engine &OccaEngine() const { return engine->As<Engine>(); }
|
||||
|
||||
/// TODO: doxygen
|
||||
::occa::device GetDevice(int idx = 0) const
|
||||
{ return OccaEngine().GetDevice(idx); }
|
||||
|
||||
mfem::Mesh* GetMesh() const { return fes->GetMesh(); }
|
||||
|
||||
Layout &OccaVLayout() const
|
||||
{ return *fes->GetVLayout().As<Layout>(); }
|
||||
|
||||
Layout &OccaTrueVLayout() const
|
||||
{ return *fes->GetTrueVLayout().As<Layout>(); }
|
||||
|
||||
Layout &OccaEVLayout() { return *e_layout; }
|
||||
|
||||
bool hasTensorBasis() const
|
||||
{ return dynamic_cast<const mfem::TensorBasisElement*>(fes->GetFE(0)); }
|
||||
|
||||
mfem::Ordering::Type GetOrdering() const { return ordering; }
|
||||
|
||||
int GetGlobalDofs() const { return globalDofs; }
|
||||
int GetLocalDofs() const { return localDofs; }
|
||||
|
||||
int GetDim() const { return fes->GetMesh()->Dimension(); }
|
||||
int GetVDim() const { return vdim; }
|
||||
|
||||
int GetVSize() const { return globalDofs * vdim; }
|
||||
int GetTrueVSize() const { return fes->GetTrueVSize(); }
|
||||
int GetGlobalVSize() const { return globalDofs*vdim; /* FIXME: MPI */ }
|
||||
int GetGlobalTrueVSize() const { return fes->GetTrueVSize(); }
|
||||
|
||||
int GetNE() const { return fes->GetNE(); }
|
||||
|
||||
const mfem::FiniteElementCollection *FEColl() const
|
||||
{ return fes->FEColl(); }
|
||||
const mfem::FiniteElement *GetFE(const int idx) const
|
||||
{ return fes->GetFE(idx); }
|
||||
|
||||
virtual const mfem::Operator *GetProlongationOperator() const
|
||||
{ return prolongationOp; }
|
||||
|
||||
virtual const mfem::Operator *GetRestrictionOperator() const
|
||||
{ return restrictionOp; }
|
||||
|
||||
virtual const mfem::Operator *GetInterpolationOperator(
|
||||
const mfem::QuadratureSpace &qspace) const
|
||||
{ return NULL; /* FIXME */ }
|
||||
|
||||
virtual const mfem::Operator *GetGradientOperator(
|
||||
const mfem::QuadratureSpace &qspace) const
|
||||
{ return NULL; /* FIXME */ }
|
||||
|
||||
const ::occa::array<int> GetLocalToGlobalMap() const
|
||||
{ return localToGlobalMap; }
|
||||
|
||||
/// L-vector to E-vector
|
||||
void GlobalToLocal(const Vector &globalVec, Vector &localVec) const
|
||||
{
|
||||
globalToLocalKernel(globalDofs,
|
||||
localDofs * fes->GetNE(),
|
||||
globalToLocalOffsets,
|
||||
globalToLocalIndices,
|
||||
globalVec.OccaMem(), localVec.OccaMem());
|
||||
}
|
||||
|
||||
/// E-vector to L-vector, transpose of GlobalToLocal
|
||||
void LocalToGlobal(const Vector &localVec, Vector &globalVec) const
|
||||
{
|
||||
localToGlobalKernel(globalDofs,
|
||||
localDofs * fes->GetNE(),
|
||||
globalToLocalOffsets,
|
||||
globalToLocalIndices,
|
||||
localVec.OccaMem(), globalVec.OccaMem());
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
|
||||
/// OCCA version of mfem::ConformingProlongationOperator
|
||||
class OccaConformingProlongation : public Operator
|
||||
{
|
||||
protected:
|
||||
// size(shr_buf)=size(shr_ltdof)
|
||||
// size(ext_buf)=size(ext_ldof)
|
||||
Array shr_ltdof, ext_ldof;
|
||||
mutable Array shr_buf, ext_buf;
|
||||
mutable char *host_shr_buf, *host_ext_buf;
|
||||
// Offsets into {shr,ext}_buf; size is num. neighbors, i.e.
|
||||
// gc.GetGroupTopology().GetNumNeighbors():
|
||||
int *shr_buf_offsets, *ext_buf_offsets;
|
||||
|
||||
::occa::memory ltdof_ldof; // shared with the restriction operator
|
||||
|
||||
::occa::memory unq_ltdof; // enumeration of the unique ltdofs in shr_ltdof
|
||||
::occa::memory unq_shr_i, unq_shr_j;
|
||||
|
||||
::occa::kernel ExtractSubVector, SetSubVector, AddSubVector;
|
||||
|
||||
MPI_Request *requests;
|
||||
|
||||
const GroupCommunicator &gc;
|
||||
|
||||
// Kernel: copy ltdofs from 'src' to 'shr_buf' - prepare for send.
|
||||
// shr_buf[i] = src[shr_ltdof[i]]
|
||||
void BcastBeginCopy(const ::occa::memory &src, std::size_t item_size) const;
|
||||
// Kernel: copy ltdofs from 'src' to ldofs in 'dst'.
|
||||
// dst[ltdof_ldof[i]] = src[i]
|
||||
void BcastLocalCopy(const ::occa::memory &src, ::occa::memory &dst,
|
||||
std::size_t item_size) const;
|
||||
// Kernel: copy ext. dofs from 'ext_buf' to 'dst' - after recv.
|
||||
// dst[ext_ldof[i]] = ext_buf[i]
|
||||
void BcastEndCopy(::occa::memory &dst, std::size_t item_size) const;
|
||||
|
||||
// Kernel: copy ext. dofs from 'src' to 'ext_buf' - prepare for send.
|
||||
// ext_buf[i] = src[ext_ldof[i]]
|
||||
void ReduceBeginCopy(const ::occa::memory &src, std::size_t item_size) const;
|
||||
// Kernel: copy owned ldofs from 'src' to ltdofs in 'dst'.
|
||||
// dst[i] = src[ltdof_ldof[i]]
|
||||
void ReduceLocalCopy(const ::occa::memory &src, ::occa::memory &dst,
|
||||
std::size_t item_size) const;
|
||||
// Kernel: assemble dofs from 'shr_buf' into to 'dst' - after recv.
|
||||
// dst[shr_ltdof[i]] += shr_buf[i]
|
||||
void ReduceEndAssemble(::occa::memory &dst, std::size_t item_size) const;
|
||||
|
||||
public:
|
||||
OccaConformingProlongation(const FiniteElementSpace &ofes,
|
||||
const mfem::ParFiniteElementSpace &pfes,
|
||||
::occa::memory ltdof_ldof_);
|
||||
|
||||
virtual ~OccaConformingProlongation();
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
#endif // MFEM_USE_MPI
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_FE_SPACE_HPP
|
||||
@@ -0,0 +1,67 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over entries
|
||||
================================================
|
||||
*/
|
||||
|
||||
#if VDIM_ORDERING == ORDERING_BY_VDIM
|
||||
typedef double *Global_t @dim(NUM_VDIM, globalEntries);
|
||||
typedef double *Local_t @dim(NUM_VDIM, localEntries);
|
||||
#else
|
||||
typedef double *Global_t @dim(NUM_VDIM, globalEntries) @dimOrder(1, 0);
|
||||
typedef double *Local_t @dim(NUM_VDIM, localEntries) @dimOrder(1, 0);
|
||||
#endif
|
||||
|
||||
@kernel void GlobalToLocal(const int globalEntries,
|
||||
const int localEntries,
|
||||
@restrict const int * offsets,
|
||||
@restrict const int * indices,
|
||||
@restrict const Global_t globalX,
|
||||
@restrict Local_t localX) {
|
||||
|
||||
for (int i = 0; i < globalEntries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < globalEntries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double dofValue = globalX(v, i);
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
localX(v, indices[j]) = dofValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void LocalToGlobal(const int globalEntries,
|
||||
const int localEntries,
|
||||
@restrict const int * offsets,
|
||||
@restrict const int * indices,
|
||||
@restrict const Local_t localX,
|
||||
@restrict Global_t globalX) {
|
||||
|
||||
for (int i = 0; i < globalEntries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < globalEntries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double dofValue = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
dofValue += localX(v, indices[j]);
|
||||
}
|
||||
globalX(v, i) = dofValue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef STORE_JACOBIAN
|
||||
# define STORE_JACOBIAN 1
|
||||
#endif
|
||||
|
||||
#ifndef STORE_JACOBIAN_INV
|
||||
# define STORE_JACOBIAN_INV 1
|
||||
#endif
|
||||
|
||||
#ifndef STORE_JACOBIAN_DET
|
||||
# define STORE_JACOBIAN_DET 1
|
||||
#endif
|
||||
|
||||
typedef double* Local1D_t @dim(1, NUM_DOFS, numElements);
|
||||
typedef double* Local2D_t @dim(2, NUM_DOFS, numElements);
|
||||
typedef double* Local3D_t @dim(3, NUM_DOFS, numElements);
|
||||
|
||||
typedef double* QLocal_t @dim(NUM_QUAD, numElements);
|
||||
|
||||
typedef double* DofToQuadD1D_t @dim(NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD2D_t @dim(2, NUM_QUAD, NUM_DOFS);
|
||||
typedef double* DofToQuadD3D_t @dim(3, NUM_QUAD, NUM_DOFS);
|
||||
|
||||
typedef double* Jacobian1D_t @dim(NUM_QUAD, numElements);
|
||||
typedef double* Jacobian2D_t @dim(2, 2, NUM_QUAD, numElements);
|
||||
typedef double* Jacobian3D_t @dim(3, 3, NUM_QUAD, numElements);
|
||||
|
||||
@kernel void InitGeometryInfo1D(const int numElements,
|
||||
@restrict const DofToQuadD1D_t dofToQuadD,
|
||||
@restrict const Local1D_t nodes,
|
||||
@restrict Jacobian1D_t J,
|
||||
@restrict Jacobian1D_t invJ,
|
||||
@restrict QLocal_t detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[NUM_DOFS];
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes[d] = nodes(0, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(q, d);
|
||||
J11 += wx * s_nodes[d];
|
||||
}
|
||||
#if STORE_JACOBIAN
|
||||
J(q, e) = J11;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
invJ(q, e) = 1.0 / J11;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = J11;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void InitGeometryInfo2D(const int numElements,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const Local2D_t nodes,
|
||||
@restrict Jacobian2D_t J,
|
||||
@restrict Jacobian2D_t invJ,
|
||||
@restrict QLocal_t detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[2 * NUM_DOFS] @dim(2, NUM_DOFS);
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes(0, d) = nodes(0, d, e);
|
||||
s_nodes(1, d) = nodes(1, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0, J12 = 0;
|
||||
double J21 = 0, J22 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(0, q, d);
|
||||
const double wy = dofToQuadD(1, q, d);
|
||||
const double x = s_nodes(0, d);
|
||||
const double y = s_nodes(1, d);
|
||||
J11 += (wx * x); J12 += (wx * y);
|
||||
J21 += (wy * x); J22 += (wy * y);
|
||||
}
|
||||
#if STORE_JACOBIAN_INV || STORE_JACOBIAN_DET
|
||||
const double r_detJ = (J11 * J22) - (J12 * J21);
|
||||
#endif
|
||||
#if STORE_JACOBIAN
|
||||
J(0, 0, q, e) = J11; J(1, 0, q, e) = J12;
|
||||
J(0, 1, q, e) = J21; J(1, 1, q, e) = J22;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
const double r_idetJ = 1.0 / r_detJ;
|
||||
invJ(0, 0, q, e) = J22 * r_idetJ;
|
||||
invJ(1, 0, q, e) = -J12 * r_idetJ;
|
||||
|
||||
invJ(0, 1, q, e) = -J21 * r_idetJ;
|
||||
invJ(1, 1, q, e) = J11 * r_idetJ;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = r_detJ;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void InitGeometryInfo3D(const int numElements,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const Local3D_t nodes,
|
||||
@restrict Jacobian3D_t J,
|
||||
@restrict Jacobian3D_t invJ,
|
||||
@restrict QLocal_t detJ) {
|
||||
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_nodes[3 * NUM_DOFS] @dim(3, NUM_DOFS);
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
for (int d = q; d < NUM_DOFS; d += NUM_QUAD) {
|
||||
s_nodes(0, d) = nodes(0, d, e);
|
||||
s_nodes(1, d) = nodes(1, d, e);
|
||||
s_nodes(2, d) = nodes(2, d, e);
|
||||
}
|
||||
}
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
double J11 = 0, J12 = 0, J13 = 0;
|
||||
double J21 = 0, J22 = 0, J23 = 0;
|
||||
double J31 = 0, J32 = 0, J33 = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const double wx = dofToQuadD(0, q, d);
|
||||
const double wy = dofToQuadD(1, q, d);
|
||||
const double wz = dofToQuadD(2, q, d);
|
||||
const double x = s_nodes(0, d);
|
||||
const double y = s_nodes(1, d);
|
||||
const double z = s_nodes(2, d);
|
||||
J11 += (wx * x); J12 += (wx * y); J13 += (wx * z);
|
||||
J21 += (wy * x); J22 += (wy * y); J23 += (wy * z);
|
||||
J31 += (wz * x); J32 += (wz * y); J33 += (wz * z);
|
||||
}
|
||||
#if STORE_JACOBIAN_INV || STORE_JACOBIAN_DET
|
||||
const double r_detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
#endif
|
||||
#if STORE_JACOBIAN
|
||||
J(0, 0, q, e) = J11; J(1, 0, q, e) = J12; J(2, 0, q, e) = J13;
|
||||
J(0, 1, q, e) = J21; J(1, 1, q, e) = J22; J(2, 1, q, e) = J23;
|
||||
J(0, 2, q, e) = J31; J(1, 2, q, e) = J32; J(2, 2, q, e) = J33;
|
||||
#endif
|
||||
#if STORE_JACOBIAN_INV
|
||||
const double r_idetJ = 1.0 / r_detJ;
|
||||
invJ(0, 0, q, e) = r_idetJ * ((J22 * J33) - (J23 * J32));
|
||||
invJ(1, 0, q, e) = r_idetJ * ((J32 * J13) - (J33 * J12));
|
||||
invJ(2, 0, q, e) = r_idetJ * ((J12 * J23) - (J13 * J22));
|
||||
|
||||
invJ(0, 1, q, e) = r_idetJ * ((J23 * J31) - (J21 * J33));
|
||||
invJ(1, 1, q, e) = r_idetJ * ((J33 * J11) - (J31 * J13));
|
||||
invJ(2, 1, q, e) = r_idetJ * ((J13 * J21) - (J11 * J23));
|
||||
|
||||
invJ(0, 2, q, e) = r_idetJ * ((J21 * J32) - (J22 * J31));
|
||||
invJ(1, 2, q, e) = r_idetJ * ((J31 * J12) - (J32 * J11));
|
||||
invJ(2, 2, q, e) = r_idetJ * ((J11 * J22) - (J12 * J21));
|
||||
#endif
|
||||
#if STORE_JACOBIAN_DET
|
||||
detJ(q, e) = r_detJ;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "gridfunc.hpp"
|
||||
#include "bilininteg.hpp"
|
||||
#include "../../fem/gridfunc.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
std::map<std::string, ::occa::kernel> gridFunctionKernels;
|
||||
|
||||
::occa::kernel GetGridFunctionKernel(::occa::device device,
|
||||
FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir)
|
||||
{
|
||||
const int numQuad = ir.GetNPoints();
|
||||
|
||||
const FiniteElement &fe = *(fespace.GetFE(0));
|
||||
const int dim = fe.GetDim();
|
||||
const int vdim = fespace.GetVDim();
|
||||
|
||||
std::stringstream ss;
|
||||
ss << ::occa::hash(device)
|
||||
<< "FEColl : " << fespace.FEColl()->Name()
|
||||
<< "Quad: " << numQuad
|
||||
<< "Dim: " << dim
|
||||
<< "VDim: " << vdim;
|
||||
std::string hash = ss.str();
|
||||
|
||||
// Kernel defines
|
||||
::occa::properties props;
|
||||
props["defines/NUM_VDIM"] = vdim;
|
||||
|
||||
SetProperties(fespace, ir, props);
|
||||
|
||||
::occa::kernel kernel = gridFunctionKernels[hash];
|
||||
if (!kernel.isInitialized())
|
||||
{
|
||||
const std::string &okl_path = fespace.OccaEngine().GetOklPath();
|
||||
kernel = device.buildKernel(okl_path + "gridfunc.okl",
|
||||
stringWithDim("GridFuncToQuad", dim),
|
||||
props);
|
||||
}
|
||||
return kernel;
|
||||
}
|
||||
|
||||
void ToQuad(const IntegrationRule &ir, FiniteElementSpace &fespace, Vector &gf,
|
||||
Vector &quadValues)
|
||||
{
|
||||
const Engine &engine = fespace.OccaEngine();
|
||||
::occa::device device = engine.GetDevice();
|
||||
|
||||
OccaDofQuadMaps &maps = OccaDofQuadMaps::Get(device, fespace, ir);
|
||||
|
||||
const int elements = fespace.GetNE();
|
||||
const int numQuad = ir.GetNPoints();
|
||||
quadValues.OccaResize(numQuad * elements, sizeof(double));
|
||||
|
||||
::occa::kernel g2qKernel = GetGridFunctionKernel(device, fespace, ir);
|
||||
g2qKernel(elements,
|
||||
maps.dofToQuad,
|
||||
fespace.GetLocalToGlobalMap(),
|
||||
gf.OccaMem(),
|
||||
quadValues.OccaMem());
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,55 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
#define MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "fespace.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
class IntegrationRule;
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
// TODO: make this object part of the backend or the engine.
|
||||
extern std::map<std::string, ::occa::kernel> gridFunctionKernels;
|
||||
|
||||
// TODO: make this a method of the backend or the engine.
|
||||
::occa::kernel GetGridFunctionKernel(::occa::device device,
|
||||
FiniteElementSpace &fespace,
|
||||
const mfem::IntegrationRule &ir);
|
||||
|
||||
// ToQuad version without the deprecated class.
|
||||
//
|
||||
// FIXME: This is the action of a global B matrix, mapping L-vector to Q-vector,
|
||||
// so it should be made into an operator that can be constructed by the
|
||||
// FE space class. A batched version, where only a subset of the elements
|
||||
// are processed should be defined as well.
|
||||
//
|
||||
// The abstract operator construction method in the FE space class is:
|
||||
// PFiniteElementSpace::GetInterpolationOperator(...)
|
||||
void ToQuad(const IntegrationRule &ir, FiniteElementSpace &ofespace, Vector &gf,
|
||||
Vector &quadValues);
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_GRID_FUNC_HPP
|
||||
@@ -0,0 +1,26 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
#ifdef USING_TENSOR_OPS
|
||||
# ifdef OCCA_USING_CPU
|
||||
# include "mfem-occa://gridfunc/tensor/cpu.okl"
|
||||
# else
|
||||
# include "mfem-occa://gridfunc/tensor/gpuHighOrder.okl"
|
||||
# endif
|
||||
#else
|
||||
# ifdef OCCA_USING_CPU
|
||||
# include "mfem-occa://gridfunc/simplex/cpu.okl"
|
||||
# else
|
||||
# include "mfem-occa://gridfunc/simplex/gpuHighOrder.okl"
|
||||
# endif
|
||||
#endif
|
||||
@@ -0,0 +1,63 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal_t out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
double r_out = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_out += r_gf * dofToQuad(d, q);
|
||||
}
|
||||
out(v, d, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal_t out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
double r_out = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_out += r_gf * dofToQuad(d, q);
|
||||
}
|
||||
out(v, d, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,79 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal_t out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gf[NUM_VDIM][NUM_DOFS];
|
||||
|
||||
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff; @inner) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
s_gf[v][d] = gf[v + gid*NUM_VDIM]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double r_out = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_out += s_gf[v][d] * dofToQuad(d, q);
|
||||
}
|
||||
out(v, q, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal_t out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_gf[NUM_VDIM][NUM_DOFS];
|
||||
|
||||
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff; @inner) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
|
||||
const int gid = l2gMap(d, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
s_gf[v][d] = gf[v + gid*NUM_VDIM]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
double r_out = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_out += s_gf[v][d] * dofToQuad(d, q);
|
||||
}
|
||||
out(v, q, e) = r_out;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,188 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void GridFuncToQuad1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap1D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal1D_t out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_out[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[v][qx] = 0;
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[v][qx] += r_gf * dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, e) = r_out[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap2D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal2D_t out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double out_xy[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double out_x[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
out_x[v][qy] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, dy, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
out_x[v][qy] += r_gf * dofToQuad(qy, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double d2q = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] += d2q * out_x[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, qy, e) = out_xy[v][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap3D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QVLocal3D_t out) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double out_xyz[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xyz[v][qz][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double out_xy[NUM_VDIM][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double out_x[NUM_VDIM][NUM_QUAD_1D];
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_x[v][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const int gid = l2gMap(dx, dy, dz, e);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
const double r_gf = gf[v + gid*NUM_VDIM];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_x[v][qx] += r_gf * dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xy[v][qy][qx] += wy * out_x[v][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out_xyz[v][qz][qy][qx] += wz * out_xy[v][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int v = 0; v < NUM_VDIM; ++v) {
|
||||
out(v, qx, qy, qz, e) = out_xyz[v][qz][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,183 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void GridFuncToQuad1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap1D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QLocal1D_t out) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@exclusive double r_out[NUM_QUAD_1D];
|
||||
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
|
||||
s_dofToQuad[i] = dofToQuad[i];
|
||||
}
|
||||
}
|
||||
|
||||
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
|
||||
if (e < numElements) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double r_gf = gf[l2gMap(dx, e)];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_out[qx] += r_gf * s_dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
out(qx, e) = r_out[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void GridFuncToQuad2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap2D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QLocal2D_t out) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
}
|
||||
}
|
||||
|
||||
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
|
||||
if (e < numElements) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
double r_x[NUM_DOFS_1D];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
r_x[dy] = gf[l2gMap(dx, dy, e)];
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double xy = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
xy += r_x[dy] * s_dofToQuad(qy, dy);
|
||||
}
|
||||
s_xy(dx, qy) = xy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
if (qy < NUM_QUAD_1D) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
double val = 0;
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
val += s_xy(dx, qy) * s_dofToQuad(qx, dx);
|
||||
}
|
||||
out(qx, qy, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void GridFuncToQuad3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DLocalMap3D_t l2gMap,
|
||||
@restrict const double * gf,
|
||||
@restrict QLocal3D_t out) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
|
||||
// Store xy planes in shared memory
|
||||
@shared double s_z[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
|
||||
// Store z axis as registers
|
||||
@exclusive double r_qz[NUM_QUAD_1D];
|
||||
|
||||
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
const int id = (y * NUM_MAX_1D) + x;
|
||||
// Fetch Q <--> D maps
|
||||
if (id < NUM_QUAD_DOFS_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
}
|
||||
// Initialize our Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double val = gf[l2gMap(dx, dy, dz, e)];
|
||||
// Calculate D -> Q in the Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_qz[qz] += val * s_dofToQuad(qz, dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// For each xy plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
// Fill xy plane at given z position
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
s_z(dx, dy) = r_qz[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
double val = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = s_dofToQuad(qy, dy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = s_dofToQuad(qx, dx);
|
||||
val += wx * wy * s_z(dx, dy);
|
||||
}
|
||||
}
|
||||
out(qx, qy, qz, e) = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,119 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "interpolation.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
RestrictionOperator::RestrictionOperator(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> indices)
|
||||
: Operator(in_layout, out_layout)
|
||||
{
|
||||
trueIndices = indices;
|
||||
|
||||
::occa::device device = in_layout.OccaEngine().GetDevice();
|
||||
const std::string &okl_path = in_layout.OccaEngine().GetOklPath();
|
||||
multOp = device.buildKernel(okl_path + "mappings.okl",
|
||||
"ExtractSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
|
||||
multTransposeOp = device.buildKernel(okl_path + "mappings.okl",
|
||||
"SetSubVector",
|
||||
"defines: { TILESIZE: 256 }");
|
||||
}
|
||||
|
||||
void RestrictionOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
// y[i] = x[trueIndices[i]]
|
||||
multOp(height, trueIndices, x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
void RestrictionOperator::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
y.OccaFill<double>(0.0);
|
||||
// y[trueIndices[i]] = x[i]
|
||||
multTransposeOp(height, trueIndices, x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
|
||||
ProlongationOperator::ProlongationOperator(OccaSparseMatrix &multOp_,
|
||||
OccaSparseMatrix &multTransposeOp_)
|
||||
: Operator(multOp_),
|
||||
pmat(NULL),
|
||||
multOp(multOp_),
|
||||
multTransposeOp(multTransposeOp_)
|
||||
{ }
|
||||
|
||||
ProlongationOperator::ProlongationOperator(Layout &in_layout,
|
||||
Layout &out_layout,
|
||||
const mfem::Operator *pmat_)
|
||||
: Operator(in_layout, out_layout),
|
||||
pmat(pmat_),
|
||||
multOp(*this),
|
||||
multTransposeOp(*this)
|
||||
{ }
|
||||
|
||||
void ProlongationOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
MFEM_VERIFY(pmat == NULL, "");
|
||||
multOp.Mult_(x, y);
|
||||
}
|
||||
|
||||
void ProlongationOperator::MultTranspose_(const Vector &x, Vector &y) const
|
||||
{
|
||||
MFEM_VERIFY(pmat == NULL, "");
|
||||
multTransposeOp.Mult_(x, y);
|
||||
}
|
||||
|
||||
void ProlongationOperator::Mult(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
if (pmat)
|
||||
{
|
||||
// FIXME: create an OCCA version of 'pmat'
|
||||
x.Pull();
|
||||
y.Pull(false);
|
||||
pmat->Mult(x, y);
|
||||
y.Push();
|
||||
}
|
||||
else
|
||||
{
|
||||
multOp.Mult(x, y);
|
||||
}
|
||||
}
|
||||
|
||||
void ProlongationOperator::MultTranspose(const mfem::Vector &x,
|
||||
mfem::Vector &y) const
|
||||
{
|
||||
if (pmat)
|
||||
{
|
||||
// FIXME: create an OCCA version of 'pmat'
|
||||
x.Pull();
|
||||
y.Pull(false);
|
||||
pmat->MultTranspose(x, y);
|
||||
y.Push();
|
||||
}
|
||||
else
|
||||
{
|
||||
multTransposeOp.Mult(x, y);
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,73 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
#define MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "vector.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "sparsemat.hpp"
|
||||
#include "../../fem/fem.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class RestrictionOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
::occa::array<int> trueIndices; // ldof = trueIndices[ltdof]
|
||||
::occa::kernel multOp, multTransposeOp;
|
||||
|
||||
public:
|
||||
RestrictionOperator(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> indices);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
class ProlongationOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
const mfem::Operator *pmat;
|
||||
OccaSparseMatrix multOp, multTransposeOp;
|
||||
|
||||
public:
|
||||
ProlongationOperator(OccaSparseMatrix &multOp_,
|
||||
OccaSparseMatrix &multTransposeOp_);
|
||||
|
||||
ProlongationOperator(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::Operator *pmat_);
|
||||
|
||||
// overrides
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const;
|
||||
|
||||
// overrides
|
||||
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const;
|
||||
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const;
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_INTERPOLATION_HPP
|
||||
@@ -0,0 +1,40 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "layout.hpp"
|
||||
#include "../../general/array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
void Layout::Resize(std::size_t new_size)
|
||||
{
|
||||
size = new_size;
|
||||
}
|
||||
|
||||
void Layout::Resize(const Array<std::size_t> &offsets)
|
||||
{
|
||||
MFEM_ASSERT(offsets.Size() == 2,
|
||||
"multiple workers are not supported yet");
|
||||
size = offsets.Last();
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,67 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "../base/layout.hpp"
|
||||
#include "engine.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Layout : public PLayout
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// SharedPtr<const mfem::Engine> engine;
|
||||
// std::size_t size;
|
||||
|
||||
public:
|
||||
Layout(const Engine &e, std::size_t s = 0) : PLayout(e, s) { }
|
||||
|
||||
const Engine &OccaEngine() const
|
||||
{ return *static_cast<const Engine *>(engine.Get()); }
|
||||
|
||||
void OccaResize(std::size_t new_size) { size = new_size; }
|
||||
|
||||
virtual ~Layout() { }
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
/// Resize the layout
|
||||
virtual void Resize(std::size_t new_size);
|
||||
|
||||
/// Resize the layout based on the given worker offsets
|
||||
virtual void Resize(const Array<std::size_t> &offsets);
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_LAYOUT_HPP
|
||||
@@ -0,0 +1,76 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over entries
|
||||
================================================
|
||||
*/
|
||||
|
||||
@kernel void ExtractSubVector(const int entries,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
out[i] = in[indices[i]]; // indices can be repeated
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void SetSubVector(const int entries,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
out[indices[i]] = in[i]; // indices CANNOT be repeated
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void AddSubVector(const int num_unique_dst_indices,
|
||||
@restrict const int *unique_dst_indices,
|
||||
@restrict const int *unique_to_src_offsets,
|
||||
@restrict const int *unique_to_src_indices,
|
||||
@restrict const double *src,
|
||||
@restrict double *dst) {
|
||||
|
||||
for (int i = 0; i < num_unique_dst_indices; ++i;
|
||||
@tile(TILESIZE, @outer, @inner)) {
|
||||
|
||||
if (i < num_unique_dst_indices) {
|
||||
const int dst_idx = unique_dst_indices[i];
|
||||
double sum = dst[dst_idx];
|
||||
const int end = unique_to_src_offsets[i+1];
|
||||
for (int j = unique_to_src_offsets[i]; j != end; ++j) {
|
||||
sum += src[unique_to_src_indices[j]];
|
||||
}
|
||||
dst[dst_idx] = sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MapSubVector(const int entries,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int fromIdx = indices[2*i + 0]; // fromIdx indices can be repeated
|
||||
const int toIdx = indices[2*i + 1]; // toIdx indices CANNOT be repeated
|
||||
out[toIdx] = in[fromIdx];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * ((J11 * J22) - (J21 * J12));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD2D_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double s = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
s += solIn(d, e) * quadToDof(d, q);
|
||||
}
|
||||
s *= oper(q, e);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += (s * quadToDof(d, q));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD3D_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double r_sol[NUM_DOFS];
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] = 0;
|
||||
}
|
||||
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
double s = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
s += solIn(d, e) * quadToDof(d, q);
|
||||
}
|
||||
s *= oper(q, e);
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
r_sol[d] += (s * quadToDof(d, q));
|
||||
}
|
||||
}
|
||||
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
solOut(d, e) += r_sol[d];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,129 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * ((J11 * J22) - (J21 * J12));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD2D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD2D_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_sol[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M2_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M2_INNER_BATCH) {
|
||||
double s = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
s += solIn(d, e) * dofToQuad(d, q);
|
||||
}
|
||||
s_sol[q] = s * oper(q, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M2_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M2_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_sol += (s_sol[q] * quadToDof(d, q));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuadD3D_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDofD3D_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DLocal_t solIn,
|
||||
@restrict DLocal_t solOut) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
@shared double s_sol[NUM_QUAD];
|
||||
|
||||
for (int qOff = 0; qOff < M3_INNER_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD; q += M3_INNER_BATCH) {
|
||||
double s = 0;
|
||||
for (int d = 0; d < NUM_DOFS; ++d) {
|
||||
s += solIn(d, e) * dofToQuad(d, q);
|
||||
}
|
||||
s_sol[q] = s * oper(q, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int dOff = 0; dOff < M3_INNER_BATCH; ++dOff) {
|
||||
for (int d = dOff; d < NUM_DOFS; d += M3_INNER_BATCH) {
|
||||
double r_sol = 0;
|
||||
for (int q = 0; q < NUM_QUAD; ++q) {
|
||||
r_sol += (s_sol[q] * quadToDof(d, q));
|
||||
}
|
||||
solOut(d, e) += r_sol;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,281 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
const double detJ = J(q, e);
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal1D_t oper,
|
||||
@restrict const DLocal1D_t solIn,
|
||||
@restrict DLocal1D_t solOut) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_x[NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] = 0;
|
||||
}
|
||||
// sol_x{qx} = dofToQuad{qx,dx} * sol{dx}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] += s * dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
// sol_x{q} *= oper{q}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] *= oper(qx, e);
|
||||
}
|
||||
// sol{dx} = quadToDof{dx,qx} * sol_x{qx}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, e) += sol_x[qx] * quadToDof(dx, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_2D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
const double detJ = ((J11 * J22) - (J21 * J12));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal2D_t oper,
|
||||
@restrict const DLocal2D_t solIn,
|
||||
@restrict DLocal2D_t solOut) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_xy[NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double sol_x[NUM_QUAD_1D];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
sol_x[qy] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] += dofToQuad(qx, dx) * s;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double d2q = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] += d2q * sol_x[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] *= oper(qx, qy, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double sol_x[NUM_DOFS_1D];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[dx] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double s = sol_xy[qy][qx];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[dx] += quadToDof(dx, qx) * s;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double q2d = quadToDof(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, e) += q2d * sol_x[dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_3D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal3D_t oper,
|
||||
@restrict const DLocal3D_t solIn,
|
||||
@restrict DLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_xyz[NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[qz][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double sol_xy[NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double sol_x[NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] = 0;
|
||||
}
|
||||
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[qx] += dofToQuad(qx, dx) * s;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[qy][qx] += wy * sol_x[qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[qz][qy][qx] += wz * sol_xy[qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[qz][qy][qx] *= oper(qx, qy, qz, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
double sol_xy[NUM_DOFS_1D][NUM_DOFS_1D];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_xy[dy][dx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double sol_x[NUM_DOFS_1D];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[dx] = 0;
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double s = sol_xyz[qz][qy][qx];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[dx] += quadToDof(dx, qx) * s;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_xy[dy][dx] += wy * sol_x[dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = quadToDof(dz, qz);
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(dx, dy, dz, e) += wz * sol_xy[dy][dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,341 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A1_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A1_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF * J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal1D_t oper,
|
||||
@restrict const DLocal1D_t solIn,
|
||||
@restrict DLocal1D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M1_ELEMENT_BATCHES; @outer) {
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double r_sol[NUM_QUAD_1D];
|
||||
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
for (int i = el; i < NUM_QUAD_DOFS_1D; i += M1_INNER_ELEMENT_BATCH) {
|
||||
s_dofToQuad[i] = dofToQuad[i];
|
||||
s_quadToDof[i] = quadToDof[i];
|
||||
}
|
||||
}
|
||||
|
||||
for (int b = 0; b < M1_OUTER_ELEMENT_BATCH; ++b) {
|
||||
for (int el = 0; el < M1_INNER_ELEMENT_BATCH; ++el; @inner) {
|
||||
const int e = eOff + b*M1_INNER_ELEMENT_BATCH + el;
|
||||
if (e < numElements) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_sol[qx] = 0;
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double s = solIn(dx, e);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_sol[qx] += s * s_dofToQuad(qx, dx);
|
||||
}
|
||||
}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
r_sol[qx] *= oper(qx, e);
|
||||
}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += r_sol[qx] * s_quadToDof(dx, qx);
|
||||
}
|
||||
solOut(dx, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A2_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A2_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A2_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_2D; q += A2_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * ((J11 * J22) - (J21 * J12));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal2D_t oper,
|
||||
@restrict const DLocal2D_t solIn,
|
||||
@restrict DLocal2D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int eOff = 0; eOff < numElements; eOff += M2_ELEMENT_BATCH; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in @shared memory
|
||||
@shared double s_xy[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
@shared double s_xy2[NUM_QUAD_2D] @dim(NUM_QUAD_1D, NUM_QUAD_1D);
|
||||
|
||||
@exclusive double r_x[NUM_MAX_1D];
|
||||
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
for (int id = x; id < NUM_QUAD_DOFS_1D; id += NUM_MAX_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
}
|
||||
}
|
||||
|
||||
for (int e = eOff; e < (eOff + M2_ELEMENT_BATCH); ++e) {
|
||||
if (e < numElements) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
s_xy(dx, qy) = 0;
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
r_x[dy] = solIn(dx, dy, e);
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double xy = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
xy += r_x[dy] * s_dofToQuad(qy, dy);
|
||||
}
|
||||
s_xy(dx, qy) = xy;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
if (qy < NUM_QUAD_1D) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
double s = 0;
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
s += s_xy(dx, qy) * s_dofToQuad(qx, dx);
|
||||
}
|
||||
s_xy2(qx, qy) = s * oper(qx, qy, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if (qx < NUM_QUAD_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
s_xy(dy, qx) = 0;
|
||||
}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
r_x[qy] = s_xy2(qx, qy);
|
||||
}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double s = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
s += r_x[qy] * s_quadToDof(dy, qy);
|
||||
}
|
||||
s_xy(dy, qx) = s;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if (dx < NUM_DOFS_1D) {
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double s = 0;
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
s += (s_xy(dy, qx) * s_quadToDof(dx, qx));
|
||||
}
|
||||
solOut(dx, dy, e) += s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
const double *quadWeights,
|
||||
const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
QLocal_t oper) {
|
||||
for (int eOff = 0; eOff < numElements; eOff += A3_ELEMENT_BATCH; @outer) {
|
||||
for (int e = eOff; e < (eOff + A3_ELEMENT_BATCH); ++e; @inner) {
|
||||
if (e < numElements) {
|
||||
for (int qOff = 0; qOff < A3_QUAD_BATCH; ++qOff; @inner) {
|
||||
for (int q = qOff; q < NUM_QUAD_3D; q += A3_QUAD_BATCH) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal3D_t oper,
|
||||
@restrict const DLocal3D_t solIn,
|
||||
@restrict DLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
// Store dof <--> quad mappings
|
||||
@shared double s_dofToQuad[NUM_QUAD_DOFS_1D] @dim(NUM_QUAD_1D, NUM_DOFS_1D);
|
||||
@shared double s_quadToDof[NUM_QUAD_DOFS_1D] @dim(NUM_DOFS_1D, NUM_QUAD_1D);
|
||||
|
||||
// Store xy planes in @shared memory
|
||||
@shared double s_xy[NUM_MAX_2D] @dim(NUM_MAX_1D, NUM_MAX_1D);
|
||||
|
||||
// Store z axis as registers
|
||||
@exclusive double r_z[NUM_QUAD_1D];
|
||||
@exclusive double r_z2[NUM_DOFS_1D];
|
||||
|
||||
for (int y = 0; y < NUM_MAX_1D; ++y; @inner) {
|
||||
for (int x = 0; x < NUM_MAX_1D; ++x; @inner) {
|
||||
const int id = (y * NUM_MAX_1D) + x;
|
||||
// Fetch Q <--> D maps
|
||||
if (id < NUM_QUAD_DOFS_1D) {
|
||||
s_dofToQuad[id] = dofToQuad[id];
|
||||
s_quadToDof[id] = quadToDof[id];
|
||||
}
|
||||
// Initialize our Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_z[qz] = 0;
|
||||
}
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
r_z2[dz] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double s = solIn(dx, dy, dz, e);
|
||||
// Calculate D -> Q in the Z axis
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
r_z[qz] += s * s_dofToQuad(qz, dz);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// For each xy plane
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
// Fill xy plane at given z position
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
s_xy(dx, dy) = r_z[qz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Calculate Dxyz, xDyz, xyDz in plane
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
double s = 0;
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = s_dofToQuad(qy, dy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
const double wx = s_dofToQuad(qx, dx);
|
||||
s += wx * wy * s_xy(dx, dy);
|
||||
}
|
||||
}
|
||||
|
||||
s *= oper(qx, qy, qz, e);
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = s_quadToDof(dz, qz);
|
||||
r_z2[dz] += wz * s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@barrier("s_xy_sync_1");
|
||||
}
|
||||
// Iterate over xy planes to compute solution
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
// Place xy plane in @shared memory
|
||||
for (int qy = 0; qy < NUM_MAX_1D; ++qy; @inner) {
|
||||
for (int qx = 0; qx < NUM_MAX_1D; ++qx; @inner) {
|
||||
if ((qx < NUM_QUAD_1D) && (qy < NUM_QUAD_1D)) {
|
||||
s_xy(qx, qy) = r_z2[dz];
|
||||
}
|
||||
}
|
||||
}
|
||||
// Finalize solution in xy plane
|
||||
for (int dy = 0; dy < NUM_MAX_1D; ++dy; @inner) {
|
||||
for (int dx = 0; dx < NUM_MAX_1D; ++dx; @inner) {
|
||||
if ((dx < NUM_DOFS_1D) && (dy < NUM_DOFS_1D)) {
|
||||
double solZ = 0;
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = s_quadToDof(dy, qy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const double wx = s_quadToDof(dx, qx);
|
||||
solZ += wx * wy * s_xy(qx, qy);
|
||||
}
|
||||
}
|
||||
solOut(dx, dy, dz, e) += solZ;
|
||||
}
|
||||
}
|
||||
}
|
||||
@barrier("s_xy_sync_2");
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
@@ -0,0 +1,142 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "operator.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
// FIXME: move this object to the Backend?
|
||||
::occa::kernelBuilder OccaConstrainedOperator::mapDofBuilder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_map_dofs",
|
||||
|
||||
"const int idx = v2[i];"
|
||||
"v0[idx] = v1[idx];",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
// FIXME: move this object to the Backend?
|
||||
::occa::kernelBuilder OccaConstrainedOperator::clearDofBuilder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"vector_clear_dofs",
|
||||
|
||||
"v0[v1[i]] = 0.0;",
|
||||
|
||||
"defines: {"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'int',"
|
||||
" TILESIZE: 128,"
|
||||
"}");
|
||||
|
||||
OccaConstrainedOperator::OccaConstrainedOperator(
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_)
|
||||
|
||||
: Operator(A_->InLayout()->As<Layout>()),
|
||||
z(OutLayout_()),
|
||||
w(OutLayout_()),
|
||||
mfem_z((z.DontDelete(), z)),
|
||||
mfem_w((w.DontDelete(), w))
|
||||
{
|
||||
Setup(OutLayout_().OccaEngine().GetDevice(), A_, constraintList_, own_A_);
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::Setup(::occa::device device_,
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_)
|
||||
{
|
||||
device = device_;
|
||||
|
||||
A = A_;
|
||||
own_A = own_A_;
|
||||
|
||||
constraintIndices = constraintList_.Size();
|
||||
if (constraintList_.Size() > 0)
|
||||
{
|
||||
constraintList = constraintList_.Get_PArray()->As<Array>().OccaMem();
|
||||
}
|
||||
else
|
||||
{
|
||||
// constraintList is not used
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::EliminateRHS(const Vector &x, Vector &b) const
|
||||
{
|
||||
::occa::kernel mapDofs = mapDofBuilder.build(device);
|
||||
|
||||
w.OccaFill(0.0);
|
||||
|
||||
if (constraintIndices)
|
||||
{
|
||||
mapDofs(constraintIndices, w.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
|
||||
A->Mult(mfem_w, mfem_z);
|
||||
|
||||
b.Axpby<double>(1.0, b, -1.0, z);
|
||||
|
||||
if (constraintIndices)
|
||||
{
|
||||
mapDofs(constraintIndices, b.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
}
|
||||
|
||||
void OccaConstrainedOperator::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
mfem::Vector mfem_y(y);
|
||||
if (constraintIndices == 0)
|
||||
{
|
||||
A->Mult(x.Wrap(), mfem_y);
|
||||
return;
|
||||
}
|
||||
|
||||
::occa::kernel mapDofs = mapDofBuilder.build(device);
|
||||
::occa::kernel clearDofs = clearDofBuilder.build(device);
|
||||
|
||||
// z.OccaAssign(x); // z = x
|
||||
// Is Axpy faster than DtoD copy on Volta?
|
||||
z.Axpby(1.0, x, 0.0, x);
|
||||
|
||||
clearDofs(constraintIndices, z.OccaMem(), constraintList);
|
||||
|
||||
A->Mult(mfem_z, mfem_y);
|
||||
|
||||
mapDofs(constraintIndices, y.OccaMem(), x.OccaMem(), constraintList);
|
||||
}
|
||||
|
||||
OccaConstrainedOperator::~OccaConstrainedOperator()
|
||||
{
|
||||
if (own_A)
|
||||
{
|
||||
delete A;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,127 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
#define MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "../../linalg/operator.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class Operator : public mfem::Operator
|
||||
{
|
||||
public:
|
||||
/// Creare an operator with the same dimensions as @a orig.
|
||||
Operator(const Operator &orig)
|
||||
: mfem::Operator(orig) { }
|
||||
|
||||
Operator(Layout &layout)
|
||||
: mfem::Operator(layout) { }
|
||||
|
||||
Operator(Layout &in_layout, Layout &out_layout)
|
||||
: mfem::Operator(in_layout, out_layout) { }
|
||||
|
||||
Layout &InLayout_() const { return in_layout->As<Layout>(); }
|
||||
|
||||
Layout &OutLayout_() const { return out_layout->As<Layout>(); }
|
||||
|
||||
virtual void Mult_(const Vector &x, Vector &y) const = 0;
|
||||
|
||||
virtual void MultTranspose_(const Vector &x, Vector &y) const
|
||||
{ MFEM_ABORT("method is not supported"); }
|
||||
|
||||
// override
|
||||
virtual void Mult(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
Mult_(x.Get_PVector()->As<Vector>(),
|
||||
y.Get_PVector()->As<Vector>());
|
||||
}
|
||||
|
||||
// override
|
||||
virtual void MultTranspose(const mfem::Vector &x, mfem::Vector &y) const
|
||||
{
|
||||
MultTranspose_(x.Get_PVector()->As<Vector>(),
|
||||
y.Get_PVector()->As<Vector>());
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
class OccaConstrainedOperator : public Operator
|
||||
{
|
||||
protected:
|
||||
::occa::device device;
|
||||
|
||||
mfem::Operator *A; //< The unconstrained Operator.
|
||||
bool own_A; //< Ownership flag for A.
|
||||
::occa::memory constraintList; //< List of constrained indices/dofs.
|
||||
int constraintIndices;
|
||||
mutable Vector z, w; //< Auxiliary vectors.
|
||||
mutable mfem::Vector mfem_z, mfem_w; // Wrap z, w
|
||||
|
||||
static ::occa::kernelBuilder mapDofBuilder, clearDofBuilder;
|
||||
|
||||
public:
|
||||
/** @brief Constructor from a general Operator and a list of essential
|
||||
indices/dofs.
|
||||
|
||||
Specify the unconstrained operator @a *A and a @a list of indices to
|
||||
constrain, i.e. each entry @a list[i] represents an essential-dof. If the
|
||||
ownership flag @a own_A is true, the operator @a *A will be destroyed
|
||||
when this object is destroyed. */
|
||||
OccaConstrainedOperator(mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_ = false);
|
||||
|
||||
void Setup(::occa::device device_,
|
||||
mfem::Operator *A_,
|
||||
const mfem::Array<int> &constraintList_,
|
||||
bool own_A_ = false);
|
||||
|
||||
/** @brief Eliminate "essential boundary condition" values specified in @a x
|
||||
from the given right-hand side @a b.
|
||||
|
||||
Performs the following steps:
|
||||
|
||||
z = A((0,x_b)); b_i -= z_i; b_b = x_b;
|
||||
|
||||
where the "_b" subscripts denote the essential (boundary) indices/dofs of
|
||||
the vectors, and "_i" -- the rest of the entries. */
|
||||
void EliminateRHS(const Vector &x, Vector &b) const;
|
||||
|
||||
/** @brief Constrained operator action.
|
||||
|
||||
Performs the following steps:
|
||||
|
||||
z = A((x_i,0)); y_i = z_i; y_b = x_b;
|
||||
|
||||
where the "_b" subscripts denote the essential (boundary) indices/dofs of
|
||||
the vectors, and "_i" -- the rest of the entries. */
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
|
||||
// Destructor: destroys the unconstrained Operator @a A if @a own_A is true.
|
||||
virtual ~OccaConstrainedOperator();
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_OPERATOR_HPP
|
||||
@@ -0,0 +1,57 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
/*
|
||||
---[ Defines Known At Compile-Time ]------------
|
||||
TILESIZE : Tilesize for iterating over dofs
|
||||
================================================
|
||||
*/
|
||||
|
||||
@kernel void Mult(const int entries,
|
||||
@restrict const int *offsets,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *weights,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
double value = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
value += weights[j] * in[indices[j]];
|
||||
}
|
||||
out[i] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MappedMult(const int entries,
|
||||
@restrict const int *offsets,
|
||||
@restrict const int *indices,
|
||||
@restrict const double *weights,
|
||||
@restrict const int *outIndices,
|
||||
@restrict const double *in,
|
||||
@restrict double *out) {
|
||||
|
||||
for (int i = 0; i < entries; ++i; @tile(TILESIZE, @outer, @inner)) {
|
||||
if (i < entries) {
|
||||
const int offset = offsets[i];
|
||||
const int nextOffset = offsets[i + 1];
|
||||
double value = 0;
|
||||
for (int j = offset; j < nextOffset; ++j) {
|
||||
value += weights[j] * in[indices[j]];
|
||||
}
|
||||
out[outIndices[i]] = value;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,221 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "sparsemat.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props)
|
||||
: Operator(in_layout, out_layout)
|
||||
{
|
||||
Setup(in_layout.OccaEngine().GetDevice(), m, props);
|
||||
}
|
||||
|
||||
OccaSparseMatrix::OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> offsets_,
|
||||
::occa::array<int> indices_,
|
||||
::occa::array<double> weights_,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props)
|
||||
: Operator(in_layout, out_layout),
|
||||
offsets(offsets_),
|
||||
indices(indices_),
|
||||
weights(weights_),
|
||||
reorderIndices(reorderIndices_),
|
||||
mappedIndices(mappedIndices_)
|
||||
{
|
||||
SetupKernel(in_layout.OccaEngine().GetDevice(), props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
Setup(device, m, ::occa::array<int>(), ::occa::array<int>(), props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Setup(::occa::device device, const SparseMatrix &m,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
MFEM_ASSERT(m.Finalized(), "");
|
||||
MFEM_ASSERT(m.Height() == height, "");
|
||||
MFEM_ASSERT(m.Width() == width, "");
|
||||
|
||||
const int nnz = m.GetI()[height];
|
||||
offsets.allocate(device,
|
||||
height + 1, m.GetI());
|
||||
indices.allocate(device,
|
||||
nnz, m.GetJ());
|
||||
weights.allocate(device,
|
||||
nnz, m.GetData());
|
||||
|
||||
offsets.keepInDevice();
|
||||
indices.keepInDevice();
|
||||
weights.keepInDevice();
|
||||
|
||||
reorderIndices = reorderIndices_;
|
||||
mappedIndices = mappedIndices_;
|
||||
|
||||
SetupKernel(device, props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::SetupKernel(::occa::device device,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
const bool hasOutIndices = mappedIndices.isInitialized();
|
||||
|
||||
const ::occa::properties defaultProps("defines: {"
|
||||
" TILESIZE: 256,"
|
||||
"}");
|
||||
|
||||
const std::string &okl_path = InLayout_().OccaEngine().GetOklPath();
|
||||
mapKernel = device.buildKernel(okl_path + "mappings.okl",
|
||||
"MapSubVector",
|
||||
defaultProps + props);
|
||||
|
||||
multKernel = device.buildKernel(okl_path + "sparse.okl",
|
||||
hasOutIndices ? "MappedMult" : "Mult",
|
||||
defaultProps + props);
|
||||
}
|
||||
|
||||
void OccaSparseMatrix::Mult_(const Vector &x, Vector &y) const
|
||||
{
|
||||
if (reorderIndices.isInitialized() ||
|
||||
mappedIndices.isInitialized())
|
||||
{
|
||||
if (reorderIndices.isInitialized())
|
||||
{
|
||||
mapKernel((int) (reorderIndices.size() / 2),
|
||||
reorderIndices,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
if (mappedIndices.isInitialized())
|
||||
{
|
||||
multKernel((int) (mappedIndices.size()),
|
||||
offsets, indices, weights,
|
||||
mappedIndices,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
multKernel((int) height,
|
||||
offsets, indices, weights,
|
||||
x.OccaMem(), y.OccaMem());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
OccaSparseMatrix* CreateMappedSparseMatrix(Layout &in_layout,
|
||||
Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props)
|
||||
{
|
||||
const int mHeight = m.Height();
|
||||
// const int mWidth = m.Width();
|
||||
|
||||
// Count indices that are only reordered (true dofs)
|
||||
const int *I = m.GetI();
|
||||
const int *J = m.GetJ();
|
||||
const double *D = m.GetData();
|
||||
|
||||
int trueCount = 0;
|
||||
for (int i = 0; i < mHeight; ++i)
|
||||
{
|
||||
trueCount += ((I[i + 1] - I[i]) == 1);
|
||||
}
|
||||
const int dupCount = (mHeight - trueCount);
|
||||
|
||||
// Create the reordering map for entries that aren't modified (true dofs)
|
||||
::occa::device device(in_layout.OccaEngine().GetDevice());
|
||||
::occa::array<int> reorderIndices(device,
|
||||
2 * trueCount);
|
||||
::occa::array<int> mappedIndices, offsets, indices;
|
||||
::occa::array<double> weights;
|
||||
|
||||
if (dupCount)
|
||||
{
|
||||
mappedIndices.allocate(device,
|
||||
dupCount);
|
||||
}
|
||||
int trueIdx = 0, dupIdx = 0;
|
||||
for (int i = 0; i < mHeight; ++i)
|
||||
{
|
||||
const int i1 = I[i];
|
||||
if ((I[i + 1] - i1) == 1)
|
||||
{
|
||||
reorderIndices[trueIdx++] = J[i1];
|
||||
reorderIndices[trueIdx++] = i;
|
||||
}
|
||||
else
|
||||
{
|
||||
mappedIndices[dupIdx++] = i;
|
||||
}
|
||||
}
|
||||
reorderIndices.keepInDevice();
|
||||
|
||||
if (dupCount)
|
||||
{
|
||||
mappedIndices.keepInDevice();
|
||||
|
||||
// Extract sparse matrix without reordered identity
|
||||
const int dupNnz = I[mHeight] - trueCount;
|
||||
|
||||
offsets.allocate(device,
|
||||
dupCount + 1);
|
||||
indices.allocate(device,
|
||||
dupNnz);
|
||||
weights.allocate(device,
|
||||
dupNnz);
|
||||
|
||||
int nnz = 0;
|
||||
offsets[0] = 0;
|
||||
for (int i = 0; i < dupCount; ++i)
|
||||
{
|
||||
const int idx = mappedIndices[i];
|
||||
const int offStart = I[idx];
|
||||
const int offEnd = I[idx + 1];
|
||||
offsets[i + 1] = offsets[i] + (offEnd - offStart);
|
||||
for (int j = offStart; j < offEnd; ++j)
|
||||
{
|
||||
indices[nnz] = J[j];
|
||||
weights[nnz] = D[j];
|
||||
++nnz;
|
||||
}
|
||||
}
|
||||
|
||||
offsets.keepInDevice();
|
||||
indices.keepInDevice();
|
||||
weights.keepInDevice();
|
||||
}
|
||||
|
||||
return new OccaSparseMatrix(in_layout, out_layout,
|
||||
offsets, indices, weights,
|
||||
reorderIndices, mappedIndices,
|
||||
props);
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,89 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
#define MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "vector.hpp"
|
||||
#include "engine.hpp"
|
||||
#include "operator.hpp"
|
||||
#include "../../linalg/sparsemat.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
/// TODO: doxygen
|
||||
class OccaSparseMatrix : public Operator
|
||||
{
|
||||
protected:
|
||||
::occa::array<int> offsets, indices;
|
||||
::occa::array<double> weights;
|
||||
::occa::array<int> reorderIndices, mappedIndices;
|
||||
::occa::kernel mapKernel, multKernel;
|
||||
|
||||
void Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props);
|
||||
|
||||
void Setup(::occa::device device, const mfem::SparseMatrix &m,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props);
|
||||
|
||||
void SetupKernel(::occa::device device,
|
||||
const ::occa::properties &props);
|
||||
|
||||
public:
|
||||
/// Construct an empty OccaSparseMatrix.
|
||||
OccaSparseMatrix(const Operator &orig)
|
||||
: Operator(orig) { }
|
||||
|
||||
// Implicitly defined copy constructor.
|
||||
|
||||
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
OccaSparseMatrix(Layout &in_layout, Layout &out_layout,
|
||||
::occa::array<int> offsets_,
|
||||
::occa::array<int> indices_,
|
||||
::occa::array<double> weights_,
|
||||
::occa::array<int> reorderIndices_,
|
||||
::occa::array<int> mappedIndices_,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
const ::occa::array<int> &GetReorderIndices() const
|
||||
{ return reorderIndices; }
|
||||
|
||||
// override
|
||||
virtual void Mult_(const Vector &x, Vector &y) const;
|
||||
};
|
||||
|
||||
|
||||
/// TODO: doxygen
|
||||
OccaSparseMatrix* CreateMappedSparseMatrix(
|
||||
Layout &in_layout, Layout &out_layout,
|
||||
const mfem::SparseMatrix &m,
|
||||
const ::occa::properties &props = ::occa::properties());
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_SPARSE_MAT_HPP
|
||||
@@ -0,0 +1,81 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "url_handler.hpp"
|
||||
#include "../../general/error.hpp"
|
||||
#include <cstdlib>
|
||||
#include <sys/stat.h>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
FileOpener::FileOpener(const std::string &prefix,
|
||||
const std::string &env_variable)
|
||||
: pfx(prefix)
|
||||
{
|
||||
const char *env_path = getenv(env_variable.c_str());
|
||||
if (!env_path) { return; }
|
||||
std::string path(env_path);
|
||||
for (std::size_t start = 0, end; start < path.size(); start = end + 1)
|
||||
{
|
||||
end = path.find(':', start);
|
||||
if (end == std::string::npos)
|
||||
{
|
||||
AddDir(path.substr(start, end));
|
||||
break;
|
||||
}
|
||||
AddDir(path.substr(start, end - start));
|
||||
}
|
||||
}
|
||||
|
||||
bool FileOpener::AddDir(const std::string &dir)
|
||||
{
|
||||
if (dir.size() == 0 || dir[0] != '/') { return false; }
|
||||
struct stat dir_stat;
|
||||
if (stat(dir.c_str(), &dir_stat)) { return false; }
|
||||
if (!S_ISDIR(dir_stat.st_mode)) { return false; }
|
||||
paths.push_back(dir + (*dir.rbegin() == '/' ? "" : "/"));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool FileOpener::handles(const std::string &filename)
|
||||
{
|
||||
return filename.size() >= pfx.size() &&
|
||||
filename.compare(0, pfx.size(), pfx) == 0;
|
||||
}
|
||||
|
||||
std::string FileOpener::expand(const std::string &filename)
|
||||
{
|
||||
std::string sfx(filename.substr(pfx.size()));
|
||||
for (std::size_t i = 0; i < paths.size(); i++)
|
||||
{
|
||||
std::string file = paths[i] + sfx;
|
||||
struct stat file_stat;
|
||||
if (stat(file.c_str(), &file_stat) == 0 && S_ISREG(file_stat.st_mode))
|
||||
{
|
||||
return file;
|
||||
}
|
||||
}
|
||||
MFEM_ABORT("invalid url: " << filename);
|
||||
return sfx;
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,47 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
#define MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
class FileOpener : public ::occa::io::fileOpener
|
||||
{
|
||||
protected:
|
||||
std::string pfx; // prefix, e.g. "mfem://"
|
||||
std::vector<std::string> paths; // paths to search for prefix replacement
|
||||
|
||||
public:
|
||||
FileOpener(const std::string &prefix, const std::string &env_variable);
|
||||
|
||||
bool AddDir(const std::string &dir);
|
||||
|
||||
virtual bool handles(const std::string &filename);
|
||||
virtual std::string expand(const std::string &filename);
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_URL_HANDLER_HPP
|
||||
@@ -0,0 +1,22 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
typedef double* Local_t @dim(numDofs, numElements);
|
||||
|
||||
@kernel void InitLocalVector(const int numElements,
|
||||
const int numDofs,
|
||||
@restrict Local_t sol) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int d = 0; d < numDofs; ++d; @inner) {
|
||||
sol(d, e) = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,196 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include "vector.hpp"
|
||||
#include "../../linalg/vector.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
PVector *Vector::DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const
|
||||
{
|
||||
MFEM_ASSERT(buffer_type_id == ScalarId<double>::value, "");
|
||||
Vector *new_vector = new Vector(OccaLayout());
|
||||
if (copy_data)
|
||||
{
|
||||
new_vector->slice.copyFrom(slice);
|
||||
}
|
||||
if (buffer)
|
||||
{
|
||||
*buffer = new_vector->GetBuffer();
|
||||
}
|
||||
return new_vector;
|
||||
}
|
||||
|
||||
void Vector::DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const
|
||||
{
|
||||
// Can be called when Size() == 0, e.g. when an MPI-parallel vector has a
|
||||
// local size of 0.
|
||||
|
||||
MFEM_ASSERT(result_type_id == ScalarId<double>::value, "");
|
||||
double *res = (double *)result;
|
||||
const Vector &xp = x.As<Vector>();
|
||||
MFEM_ASSERT(this->Size() == xp.Size(), "");
|
||||
*res = ::occa::linalg::dot<double, double, double>(this->slice, xp.slice);
|
||||
|
||||
#ifdef MFEM_USE_MPI
|
||||
double local_dot = *res;
|
||||
if (IsParallel())
|
||||
{
|
||||
MPI_Allreduce(&local_dot, res, 1, MPI_DOUBLE, MPI_SUM,
|
||||
OccaEngine().GetComm());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void Vector::DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id)
|
||||
{
|
||||
//
|
||||
// TODO: move all kernel builders to class mfem::occa::Backend
|
||||
//
|
||||
static ::occa::kernelBuilder axpby1_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby1",
|
||||
"v0[i] = c0 * v1[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
static ::occa::kernelBuilder axpby2_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby2",
|
||||
"v0[i] = c0 * v0[i] + c1 * v1[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" CTYPE1: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
static ::occa::kernelBuilder axpby3_builder =
|
||||
::occa::linalg::customLinearMethod(
|
||||
"mfem_occa_axpby3",
|
||||
"v0[i] = c0 * v1[i] + c1 * v2[i];",
|
||||
"defines: {"
|
||||
" CTYPE0: 'double',"
|
||||
" CTYPE1: 'double',"
|
||||
" VTYPE0: 'double',"
|
||||
" VTYPE1: 'double',"
|
||||
" VTYPE2: 'double',"
|
||||
" TILESIZE: '128',"
|
||||
"}");
|
||||
|
||||
// called only when Size() != 0
|
||||
|
||||
MFEM_ASSERT(ab_type_id == ScalarId<double>::value, "");
|
||||
const double da = *static_cast<const double *>(a);
|
||||
const double db = *static_cast<const double *>(b);
|
||||
MFEM_ASSERT(da == 0.0 || dynamic_cast<const Vector *>(&x) != NULL,
|
||||
"invalid Vector x");
|
||||
MFEM_ASSERT(db == 0.0 || dynamic_cast<const Vector *>(&y) != NULL,
|
||||
"invalid Vector y");
|
||||
const Vector *xp = static_cast<const Vector *>(&x);
|
||||
const Vector *yp = static_cast<const Vector *>(&y);
|
||||
|
||||
MFEM_ASSERT(da == 0.0 || this->Size() == xp->Size(), "");
|
||||
MFEM_ASSERT(db == 0.0 || this->Size() == yp->Size(), "");
|
||||
|
||||
if (da == 0.0)
|
||||
{
|
||||
if (db == 0.0)
|
||||
{
|
||||
OccaFill(da);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (this->slice == yp->slice)
|
||||
{
|
||||
// *this *= db
|
||||
::occa::linalg::operator_mult_eq(slice, db);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = db * y
|
||||
::occa::kernel kernel = axpby1_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), db, slice, yp->slice);
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (db == 0.0)
|
||||
{
|
||||
if (this->slice == xp->slice)
|
||||
{
|
||||
// *this *= da
|
||||
::occa::linalg::operator_mult_eq(slice, da);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = da * x
|
||||
::occa::kernel kernel = axpby1_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), da, slice, xp->slice);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
MFEM_ASSERT(xp->slice != yp->slice, "invalid input");
|
||||
if (this->slice == xp->slice)
|
||||
{
|
||||
// *this = da * (*this) + db * y
|
||||
::occa::kernel kernel = axpby2_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), da, db, slice, yp->slice);
|
||||
}
|
||||
else if (this->slice == yp->slice)
|
||||
{
|
||||
// *this = da * x + db * (*this)
|
||||
::occa::kernel kernel = axpby2_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), db, da, slice, xp->slice);
|
||||
}
|
||||
else
|
||||
{
|
||||
// *this = da * x + db * y
|
||||
::occa::kernel kernel = axpby3_builder.build(slice.getDevice());
|
||||
kernel((int)Size(), da, db, slice, xp->slice, yp->slice);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mfem::Vector Vector::Wrap()
|
||||
{
|
||||
return mfem::Vector(*this);
|
||||
}
|
||||
|
||||
const mfem::Vector Vector::Wrap() const
|
||||
{
|
||||
return mfem::Vector(*const_cast<Vector*>(this));
|
||||
}
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
@@ -0,0 +1,83 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
#define MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
|
||||
#include "../../config/config.hpp"
|
||||
#if defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#include <occa.hpp>
|
||||
#include "../base/vector.hpp"
|
||||
#include "array.hpp"
|
||||
|
||||
namespace mfem
|
||||
{
|
||||
|
||||
namespace occa
|
||||
{
|
||||
|
||||
// FIXME: Once XL fixes this code quirk we can remove this #ifdef switch
|
||||
#ifdef __ibmxl__
|
||||
class Vector : public Array, public PVector
|
||||
#else
|
||||
class Vector : virtual public Array, public PVector
|
||||
#endif
|
||||
{
|
||||
protected:
|
||||
//
|
||||
// Inherited fields
|
||||
//
|
||||
// DLayout layout;
|
||||
|
||||
/**
|
||||
@name Virtual interface
|
||||
*/
|
||||
///@{
|
||||
|
||||
virtual PVector *DoVectorClone(bool copy_data, void **buffer,
|
||||
int buffer_type_id) const;
|
||||
|
||||
virtual void DoDotProduct(const PVector &x, void *result,
|
||||
int result_type_id) const;
|
||||
|
||||
virtual void DoAxpby(const void *a, const PVector &x,
|
||||
const void *b, const PVector &y,
|
||||
int ab_type_id);
|
||||
|
||||
///@}
|
||||
// End: Virtual interface
|
||||
|
||||
public:
|
||||
Vector(const Engine &e)
|
||||
: PArray(*(new Layout(e, 0))), Array(e), PVector(*layout)
|
||||
{ }
|
||||
|
||||
Vector(Layout <)
|
||||
: PArray(lt), Array(lt, sizeof(double)), PVector(lt)
|
||||
{ }
|
||||
|
||||
mfem::Vector Wrap();
|
||||
|
||||
const mfem::Vector Wrap() const;
|
||||
|
||||
#if defined(MFEM_USE_MPI)
|
||||
bool IsParallel() const { return (OccaEngine().GetComm() != MPI_COMM_NULL); }
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace mfem::occa
|
||||
|
||||
} // namespace mfem
|
||||
|
||||
#endif // defined(MFEM_USE_BACKENDS) && defined(MFEM_USE_OCCA)
|
||||
|
||||
#endif // MFEM_BACKENDS_OCCA_VECTOR_HPP
|
||||
@@ -0,0 +1,321 @@
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#include "mfem-occa://defines.okl"
|
||||
|
||||
//---[ 1D ]-----------------------------
|
||||
@kernel void Assemble1D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian1D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_1D; ++q; @inner) {
|
||||
oper(q, e) = quadWeights[q] * COEFF * J(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd1D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DVLocal1D_t solIn,
|
||||
@restrict DVLocal1D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_x[1][NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] = 0;
|
||||
}
|
||||
// sol_x{qx} = dofToQuad{qx,dx} * sol{dx}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] += dofToQuad(qx, dx) * solIn(0, dx, e);
|
||||
}
|
||||
}
|
||||
// sol_x{q} *= oper{q}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] *= oper(qx, e);
|
||||
}
|
||||
// sol{dx} = quadToDof{dx,qx} * sol_x{qx}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(0, dx, e) += sol_x[0][qx] * quadToDof(dx, qx);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 2D ]-----------------------------
|
||||
@kernel void Assemble2D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian2D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_2D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e);
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * ((J11 * J22) - (J21 * J12));
|
||||
}
|
||||
} // e
|
||||
}
|
||||
|
||||
@kernel void MultAdd2D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DVLocal2D_t solIn,
|
||||
@restrict DVLocal2D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy=0; dummy<1; ++dummy; @inner) {
|
||||
double sol_xy[2][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[0][qx][qy] = 0;
|
||||
sol_xy[1][qx][qy] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double sol_x[2][NUM_QUAD_1D];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
sol_x[0][qy] = 0;
|
||||
sol_x[1][qy] = 0;
|
||||
}
|
||||
|
||||
// sol_x{vd, dx, qy} = dofToQuad{qy, dy} * sol{vd, dx, dy}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
sol_x[0][qy] += dofToQuad(qy, dx) * solIn(0, dx, dy, e);
|
||||
sol_x[1][qy] += dofToQuad(qy, dx) * solIn(1, dx, dy, e);
|
||||
}
|
||||
}
|
||||
|
||||
// sol_xy{qx, qy} = dofToQuad{qx, dx} * sol_x{dx, qy}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double d2q = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[0][qx][qy] += d2q * sol_x[0][qx];
|
||||
sol_xy[1][qx][qy] += d2q * sol_x[1][qx];
|
||||
}
|
||||
}
|
||||
} // dy
|
||||
|
||||
// sol_xy{qx, qy} = sol_xy{q} *= oper{q, e}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_2D_ID(qx, qy);
|
||||
sol_xy[0][qx][qy] *= oper(q, e);
|
||||
sol_xy[1][qx][qy] *= oper(q, e);
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double sol_x[2][NUM_DOFS_1D];
|
||||
for (int dx = 0; dx < NUM_QUAD_1D; ++dx) {
|
||||
sol_x[0][dx] = 0;
|
||||
sol_x[1][dx] = 0;
|
||||
}
|
||||
|
||||
// sol_x{qx, dy} = quadToDof{dy, qy} * sol_xy{qx, qy}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[0][dx] += quadToDof(dx, qx) * sol_xy[0][qx][qy];
|
||||
sol_x[1][dx] += quadToDof(dx, qx) * sol_xy[1][qx][qy];
|
||||
}
|
||||
}
|
||||
|
||||
// sol{dx, dy, e} = quadToDof{dx, qx} * sol_x{qx, dy}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double q2d = quadToDof(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(0, dx, dy, e) += q2d * sol_x[0][dx];
|
||||
solOut(1, dx, dy, e) += q2d * sol_x[1][dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
} // dummy
|
||||
} // e
|
||||
}
|
||||
//======================================
|
||||
|
||||
|
||||
//---[ 3D ]-----------------------------
|
||||
@kernel void Assemble3D(const int numElements,
|
||||
@restrict const double * quadWeights,
|
||||
@restrict const Jacobian3D_t J,
|
||||
COEFF_ARGS
|
||||
@restrict QLocal_t oper) {
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int q = 0; q < NUM_QUAD_3D; ++q; @inner) {
|
||||
const double J11 = J(0, 0, q, e), J12 = J(1, 0, q, e), J13 = J(2, 0, q, e);
|
||||
const double J21 = J(0, 1, q, e), J22 = J(1, 1, q, e), J23 = J(2, 1, q, e);
|
||||
const double J31 = J(0, 2, q, e), J32 = J(1, 2, q, e), J33 = J(2, 2, q, e);
|
||||
|
||||
const double detJ = ((J11 * J22 * J33) + (J12 * J23 * J31) + (J13 * J21 * J32) -
|
||||
(J13 * J22 * J31) - (J12 * J21 * J33) - (J11 * J23 * J32));
|
||||
|
||||
oper(q, e) = quadWeights[q] * COEFF * detJ;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@kernel void MultAdd3D(const int numElements,
|
||||
@restrict const DofToQuad_t dofToQuad,
|
||||
@restrict const DofToQuad_t dofToQuadD,
|
||||
@restrict const QuadToDof_t quadToDof,
|
||||
@restrict const QuadToDof_t quadToDofD,
|
||||
@restrict const QLocal_t oper,
|
||||
@restrict const DVLocal3D_t solIn,
|
||||
@restrict DVLocal3D_t solOut) {
|
||||
// Iterate over elements
|
||||
for (int e = 0; e < numElements; ++e; @outer) {
|
||||
for (int dummy = 0; dummy < 1; ++dummy; @inner) {
|
||||
double sol_xyz[3][NUM_QUAD_1D][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[0][qz][qy][qx] = 0;
|
||||
sol_xyz[1][qz][qy][qx] = 0;
|
||||
sol_xyz[2][qz][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
double sol_xy[3][NUM_QUAD_1D][NUM_QUAD_1D];
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[0][qy][qx] = 0;
|
||||
sol_xy[1][qy][qx] = 0;
|
||||
sol_xy[2][qy][qx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
double sol_x[3][NUM_QUAD_1D];
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] = 0;
|
||||
sol_x[1][qx] = 0;
|
||||
sol_x[2][qx] = 0;
|
||||
}
|
||||
|
||||
// sol_x{qx} = dofToQuad{qx, dx} * sol{dx, dy, dz, e}
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_x[0][qx] += dofToQuad(qx, dx) * solIn(0, dx, dy, dz, e);
|
||||
sol_x[1][qx] += dofToQuad(qx, dx) * solIn(1, dx, dy, dz, e);
|
||||
sol_x[2][qx] += dofToQuad(qx, dx) * solIn(2, dx, dy, dz, e);
|
||||
}
|
||||
}
|
||||
|
||||
// sol_xy{qx, qy} = dofToQuad{qy, dy} * sol_x{dx}
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
const double wy = dofToQuad(qy, dy);
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xy[0][qy][qx] += wy * sol_x[0][qx];
|
||||
sol_xy[1][qy][qx] += wy * sol_x[1][qx];
|
||||
sol_xy[2][qy][qx] += wy * sol_x[2][qx];
|
||||
}
|
||||
}
|
||||
} // dy
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
const double wz = dofToQuad(qz, dz);
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
sol_xyz[0][qz][qy][qx] += wz * sol_xy[0][qy][qx];
|
||||
sol_xyz[1][qz][qy][qx] += wz * sol_xy[1][qy][qx];
|
||||
sol_xyz[2][qz][qy][qx] += wz * sol_xy[2][qy][qx];
|
||||
}
|
||||
}
|
||||
}
|
||||
} // dz
|
||||
|
||||
// sol_xyz{qz, qy, qx} *= oper{q, e}
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
const int q = QUAD_3D_ID(qx, qy, qz);
|
||||
sol_xyz[0][qz][qy][qx] *= oper(q, e);
|
||||
sol_xyz[1][qz][qy][qx] *= oper(q, e);
|
||||
sol_xyz[2][qz][qy][qx] *= oper(q, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int qz = 0; qz < NUM_QUAD_1D; ++qz) {
|
||||
double sol_xy[3][NUM_DOFS_1D][NUM_DOFS_1D];
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_xy[0][dy][dx] = 0;
|
||||
sol_xy[1][dy][dx] = 0;
|
||||
sol_xy[2][dy][dx] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int qy = 0; qy < NUM_QUAD_1D; ++qy) {
|
||||
double sol_x[3][NUM_DOFS_1D];
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[0][dx] = 0;
|
||||
sol_x[1][dx] = 0;
|
||||
sol_x[2][dx] = 0;
|
||||
}
|
||||
|
||||
// sol_x{dx} = quadToDof{dx, qx} * sol_xyz{qz, qy, qx}
|
||||
for (int qx = 0; qx < NUM_QUAD_1D; ++qx) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_x[0][dx] += quadToDof(dx, qx) * sol_xyz[0][qz][qy][qx];
|
||||
sol_x[1][dx] += quadToDof(dx, qx) * sol_xyz[1][qz][qy][qx];
|
||||
sol_x[2][dx] += quadToDof(dx, qx) * sol_xyz[2][qz][qy][qx];
|
||||
}
|
||||
}
|
||||
|
||||
// sol_xy{dy, dx} = quadToDof{dy, qy} * sol_x{dx}
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
const double wy = quadToDof(dy, qy);
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
sol_xy[0][dy][dx] += wy * sol_x[0][dx];
|
||||
sol_xy[1][dy][dx] += wy * sol_x[1][dx];
|
||||
sol_xy[2][dy][dx] += wy * sol_x[2][dx];
|
||||
}
|
||||
}
|
||||
} // qy
|
||||
|
||||
for (int dz = 0; dz < NUM_DOFS_1D; ++dz) {
|
||||
const double wz = quadToDof(dz, qz);
|
||||
for (int dy = 0; dy < NUM_DOFS_1D; ++dy) {
|
||||
for (int dx = 0; dx < NUM_DOFS_1D; ++dx) {
|
||||
solOut(0, dx, dy, dz, e) += wz * sol_xy[0][dy][dx];
|
||||
solOut(1, dx, dy, dz, e) += wz * sol_xy[1][dy][dx];
|
||||
solOut(2, dx, dy, dz, e) += wz * sol_xy[2][dy][dx];
|
||||
}
|
||||
}
|
||||
}
|
||||
} // qz
|
||||
} // dummy
|
||||
} // e
|
||||
|
||||
}
|
||||
//======================================
|
||||
@@ -1,13 +1,13 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
set(HDRS
|
||||
config.hpp
|
||||
|
||||
+4
-103
@@ -74,7 +74,7 @@
|
||||
|
||||
IF (NOT COMMAND PRINT_VAR)
|
||||
FUNCTION(PRINT_VAR VAR_NAME)
|
||||
MESSAGE(STATUS "${VAR_NAME} = '${${VAR_NAME}}'")
|
||||
MESSAGE("-- " "${VAR_NAME} = '${${VAR_NAME}}'")
|
||||
ENDFUNCTION()
|
||||
ENDIF()
|
||||
|
||||
@@ -166,116 +166,17 @@ IF (USE_XSDK_DEFAULTS)
|
||||
ENDIF()
|
||||
XSDK_HANDLE_LANG_DEFAULTS(Fortran FC "FFLAGS;FCFLAGS")
|
||||
ENDIF()
|
||||
|
||||
|
||||
# Set XSDK defaults for other CMake variables
|
||||
|
||||
|
||||
IF ("${BUILD_SHARED_LIBS}" STREQUAL "")
|
||||
MESSAGE("-- " "XSDK: Setting default BUILD_SHARED_LIBS=TRUE")
|
||||
SET(BUILD_SHARED_LIBS TRUE CACHE BOOL "Set by default in XSDK mode")
|
||||
ENDIF()
|
||||
|
||||
|
||||
IF ("${CMAKE_BUILD_TYPE}" STREQUAL "")
|
||||
MESSAGE("-- " "XSDK: Setting default CMAKE_BUILD_TYPE=DEBUG")
|
||||
SET(CMAKE_BUILD_TYPE DEBUG CACHE STRING "Set by default in XSDK mode")
|
||||
ENDIF()
|
||||
|
||||
ENDIF()
|
||||
|
||||
|
||||
##################################################################################
|
||||
#
|
||||
# MFEM-specific additions: set TPL MFEM_USE_* defaults
|
||||
#
|
||||
##################################################################################
|
||||
|
||||
IF (DEFINED TPL_ENABLE_MPI)
|
||||
SET(MFEM_USE_MPI ${TPL_ENABLE_MPI} CACHE BOOL "Enable MPI parallel build" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_METIS)
|
||||
SET(MFEM_USE_METIS ${TPL_ENABLE_METIS} CACHE BOOL "Enable METIS usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_ZLIB)
|
||||
SET(MFEM_USE_ZLIB ${TPL_ENABLE_ZLIB} CACHE BOOL "Enable zlib for compressed data streams." FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_LIBUNWIND)
|
||||
SET(MFEM_USE_LIBUNWIND ${TPL_ENABLE_LIBUNWIND} CACHE BOOL "Enable backtrace for errors." FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_LAPACK)
|
||||
SET(MFEM_USE_LAPACK ${TPL_ENABLE_LAPACK} CACHE BOOL "Enable LAPACK usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_SUNDIALS)
|
||||
SET(MFEM_USE_SUNDIALS ${TPL_ENABLE_SUNDIALS} CACHE BOOL "Enable SUNDIALS usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_MESQUITE)
|
||||
SET(MFEM_USE_MESQUITE ${TPL_ENABLE_MESQUITE} CACHE BOOL "Enable MESQUITE usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_SUITESPARSE)
|
||||
SET(MFEM_USE_SUITESPARSE ${TPL_ENABLE_SUITESPARSE} CACHE BOOL "Enable SuiteSparse usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_SUPERLU)
|
||||
SET(MFEM_USE_SUPERLU ${TPL_ENABLE_SUPERLU} CACHE BOOL "Enable SuperLU_DIST usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_STRUMPACK)
|
||||
SET(MFEM_USE_STRUMPACK ${TPL_ENABLE_STRUMPACK} CACHE BOOL "Enable STRUMPACK usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_GINKGO)
|
||||
SET(MFEM_USE_GINKGO ${TPL_ENABLE_GINKGO} CACHE BOOL "Enable GINKGO usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_GNUTLS)
|
||||
SET(MFEM_USE_GNUTLS ${TPL_ENABLE_GNUTLS} CACHE BOOL "Enable GNUTLS usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_NETCDF)
|
||||
SET(MFEM_USE_NETCDF ${TPL_ENABLE_NETCDF} CACHE BOOL "Enable NETCDF usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_PETSC)
|
||||
SET(MFEM_USE_PETSC ${TPL_ENABLE_PETSC} CACHE BOOL "Enable PETSc support." FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_SLEPC)
|
||||
SET(MFEM_USE_SLEPC ${TPL_ENABLE_SLEPC} CACHE BOOL "Enable SLEPc support." FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_MPFR)
|
||||
SET(MFEM_USE_MPFR ${TPL_ENABLE_MPFR} CACHE BOOL "Enable MPFR usage." FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_SIDRE)
|
||||
SET(MFEM_USE_SIDRE ${TPL_ENABLE_SIDRE} CACHE BOOL "Enable Axom/Sidre usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_CONDUIT)
|
||||
SET(MFEM_USE_CONDUIT ${TPL_ENABLE_CONDUIT} CACHE BOOL "Enable Conduit usage" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_PUMI)
|
||||
SET(MFEM_USE_PUMI ${TPL_ENABLE_PUMI} CACHE BOOL "Enable PUMI" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_CUDA)
|
||||
SET(MFEM_USE_CUDA ${TPL_ENABLE_CUDA} CACHE BOOL "Enable CUDA" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_OCCA)
|
||||
SET(MFEM_USE_OCCA ${TPL_ENABLE_OCCA} CACHE BOOL "Enable OCCA" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_RAJA)
|
||||
SET(MFEM_USE_RAJA ${TPL_ENABLE_RAJA} CACHE BOOL "Enable RAJA" FORCE)
|
||||
ENDIF()
|
||||
|
||||
IF (DEFINED TPL_ENABLE_UMPIRE)
|
||||
SET(MFEM_USE_UMPIRE ${TPL_ENABLE_UMPIRE} CACHE BOOL "Enable Umpire" FORCE)
|
||||
ENDIF()
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
# the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
# reserved. See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
include(${CMAKE_CURRENT_LIST_DIR}/MFEMConfigVersion.cmake)
|
||||
|
||||
@@ -20,40 +20,26 @@ set(MFEM_USE_METIS @MFEM_USE_METIS@)
|
||||
set(MFEM_USE_METIS_5 @MFEM_USE_METIS_5@)
|
||||
set(MFEM_DEBUG @MFEM_DEBUG@)
|
||||
set(MFEM_USE_EXCEPTIONS @MFEM_USE_EXCEPTIONS@)
|
||||
set(MFEM_USE_ZLIB @MFEM_USE_ZLIB@)
|
||||
set(MFEM_USE_GZSTREAM @MFEM_USE_GZSTREAM@)
|
||||
set(MFEM_USE_LIBUNWIND @MFEM_USE_LIBUNWIND@)
|
||||
set(MFEM_USE_LAPACK @MFEM_USE_LAPACK@)
|
||||
set(MFEM_THREAD_SAFE @MFEM_THREAD_SAFE@)
|
||||
set(MFEM_USE_OPENMP @MFEM_USE_OPENMP@)
|
||||
set(MFEM_USE_LEGACY_OPENMP @MFEM_USE_LEGACY_OPENMP@)
|
||||
set(MFEM_USE_MEMALLOC @MFEM_USE_MEMALLOC@)
|
||||
set(MFEM_TIMER_TYPE @MFEM_TIMER_TYPE@)
|
||||
set(MFEM_USE_SUNDIALS @MFEM_USE_SUNDIALS@)
|
||||
set(MFEM_USE_MESQUITE @MFEM_USE_MESQUITE@)
|
||||
set(MFEM_USE_SUITESPARSE @MFEM_USE_SUITESPARSE@)
|
||||
set(MFEM_USE_SUPERLU @MFEM_USE_SUPERLU@)
|
||||
set(MFEM_USE_MUMPS @MFEM_USE_MUMPS@)
|
||||
set(MFEM_USE_STRUMPACK @MFEM_USE_STRUMPACK@)
|
||||
set(MFEM_USE_GINKGO @MFEM_USE_GINKGO@)
|
||||
set(MFEM_USE_AMGX @MFEM_USE_AMGX@)
|
||||
set(MFEM_USE_HIOP @MFEM_USE_HIOP@)
|
||||
set(MFEM_USE_GECKO @MFEM_USE_GECKO@)
|
||||
set(MFEM_USE_GNUTLS @MFEM_USE_GNUTLS@)
|
||||
set(MFEM_USE_GSLIB @MFEM_USE_GSLIB@)
|
||||
set(MFEM_USE_NETCDF @MFEM_USE_NETCDF@)
|
||||
set(MFEM_USE_PETSC @MFEM_USE_PETSC@)
|
||||
set(MFEM_USE_SLEPC @MFEM_USE_SLEPC@)
|
||||
set(MFEM_USE_MPFR @MFEM_USE_MPFR@)
|
||||
set(MFEM_USE_SIDRE @MFEM_USE_SIDRE@)
|
||||
set(MFEM_USE_CONDUIT @MFEM_USE_CONDUIT@)
|
||||
set(MFEM_USE_PUMI @MFEM_USE_PUMI@)
|
||||
set(MFEM_USE_CUDA @MFEM_USE_CUDA@)
|
||||
set(MFEM_USE_OCCA @MFEM_USE_OCCA@)
|
||||
set(MFEM_USE_RAJA @MFEM_USE_RAJA@)
|
||||
set(MFEM_USE_CEED @MFEM_USE_CEED@)
|
||||
set(MFEM_USE_UMPIRE @MFEM_USE_UMPIRE@)
|
||||
set(MFEM_USE_SIMD @MFEM_USE_SIMD@)
|
||||
set(MFEM_USE_ADIOS2 @MFEM_USE_ADIOS2@)
|
||||
set(MFEM_USE_CALIPER @MFEM_USE_CALIPER@)
|
||||
|
||||
set(MFEM_CXX_COMPILER "@CMAKE_CXX_COMPILER@")
|
||||
set(MFEM_CXX_FLAGS "@CMAKE_CXX_FLAGS@")
|
||||
|
||||
+17
-69
@@ -1,13 +1,13 @@
|
||||
// Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
// LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
// Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at
|
||||
// the Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights
|
||||
// reserved. See file COPYRIGHT for details.
|
||||
//
|
||||
// This file is part of the MFEM library. For more information and source code
|
||||
// availability visit https://mfem.org.
|
||||
// availability see http://mfem.org.
|
||||
//
|
||||
// MFEM is free software; you can redistribute it and/or modify it under the
|
||||
// terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
// CONTRIBUTING.md for details.
|
||||
// terms of the GNU Lesser General Public License (as published by the Free
|
||||
// Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
#ifndef MFEM_CONFIG_HEADER
|
||||
#define MFEM_CONFIG_HEADER
|
||||
@@ -30,12 +30,6 @@
|
||||
#define MFEM_VERSION_MINOR (((MFEM_VERSION)/100)%100)
|
||||
#define MFEM_VERSION_PATCH ((MFEM_VERSION)%100)
|
||||
|
||||
// MFEM source directory.
|
||||
#define MFEM_SOURCE_DIR "@MFEM_SOURCE_DIR@"
|
||||
|
||||
// MFEM install directory.
|
||||
#define MFEM_INSTALL_DIR "@MFEM_INSTALL_DIR@"
|
||||
|
||||
// Description of the git commit used to build MFEM.
|
||||
#cmakedefine MFEM_GIT_STRING "@MFEM_GIT_STRING@"
|
||||
|
||||
@@ -49,8 +43,8 @@
|
||||
// Throw an exception on errors.
|
||||
#cmakedefine MFEM_USE_EXCEPTIONS
|
||||
|
||||
// Enable zlib in MFEM.
|
||||
#cmakedefine MFEM_USE_ZLIB
|
||||
// Enable gzstream in MFEM.
|
||||
#cmakedefine MFEM_USE_GZSTREAM
|
||||
|
||||
// Enable backtraces for mfem_error through libunwind.
|
||||
#cmakedefine MFEM_USE_LIBUNWIND
|
||||
@@ -68,15 +62,9 @@
|
||||
// allocation and de-allocation.
|
||||
#cmakedefine MFEM_THREAD_SAFE
|
||||
|
||||
// Enable the OpenMP backend.
|
||||
// Enable experimental OpenMP support. Requires MFEM_THREAD_SAFE.
|
||||
#cmakedefine MFEM_USE_OPENMP
|
||||
|
||||
// [Deprecated] Enable experimental OpenMP support. Requires MFEM_THREAD_SAFE.
|
||||
#cmakedefine MFEM_USE_LEGACY_OPENMP
|
||||
|
||||
// Internal MFEM option: enable group/batch allocation for some small objects.
|
||||
#cmakedefine MFEM_USE_MEMALLOC
|
||||
|
||||
// Enable MFEM functionality based on the Mesquite library.
|
||||
#cmakedefine MFEM_USE_MESQUITE
|
||||
|
||||
@@ -86,74 +74,33 @@
|
||||
// Enable MFEM functionality based on the SuperLU_DIST library.
|
||||
#cmakedefine MFEM_USE_SUPERLU
|
||||
|
||||
// Enable MFEM functionality based on the MUMPS library.
|
||||
#cmakedefine MFEM_USE_MUMPS
|
||||
|
||||
// Enable MFEM functionality based on the STRUMPACK library.
|
||||
#cmakedefine MFEM_USE_STRUMPACK
|
||||
|
||||
// Enable functionality based on the Ginkgo library
|
||||
#cmakedefine MFEM_USE_GINKGO
|
||||
// Internal MFEM option: enable group/batch allocation for some small objects.
|
||||
#cmakedefine MFEM_USE_MEMALLOC
|
||||
|
||||
// Enable MFEM functionality based on the AmgX library
|
||||
#cmakedefine MFEM_USE_AMGX
|
||||
// Enable functionality based on the Gecko library
|
||||
#cmakedefine MFEM_USE_GECKO
|
||||
|
||||
// Enable MFEM functionality based on the GnuTLS library
|
||||
#cmakedefine MFEM_USE_GNUTLS
|
||||
|
||||
// Enable MFEM functionality based on the GSLIB library
|
||||
#cmakedefine MFEM_USE_GSLIB
|
||||
|
||||
// Enable MFEM functionality based on the NetCDF library
|
||||
#cmakedefine MFEM_USE_NETCDF
|
||||
|
||||
// Enable MFEM functionality based on the PETSc library
|
||||
#cmakedefine MFEM_USE_PETSC
|
||||
|
||||
// Enable MFEM functionality based on the SLEPc library
|
||||
#cmakedefine MFEM_USE_SLEPC
|
||||
|
||||
// Enable MFEM functionality based on the Sidre library
|
||||
#cmakedefine MFEM_USE_SIDRE
|
||||
|
||||
// Enable the use of SIMD in the high performance templated classes
|
||||
#cmakedefine MFEM_USE_SIMD
|
||||
|
||||
// Enable MFEM functionality based on Conduit
|
||||
#cmakedefine MFEM_USE_CONDUIT
|
||||
|
||||
// Enable MFEM functionality based on the PUMI library
|
||||
#cmakedefine MFEM_USE_PUMI
|
||||
|
||||
// Enable MFEM functionality based on the HiOp library
|
||||
#cmakedefine MFEM_USE_HIOP
|
||||
|
||||
// Build the GPU/CUDA-enabled version of the MFEM library.
|
||||
// Requires a CUDA compiler (nvcc).
|
||||
#cmakedefine MFEM_USE_CUDA
|
||||
|
||||
// Build the HIP-enabled version of the MFEM library.
|
||||
// Requires a HIP compiler (hipcc).
|
||||
#cmakedefine MFEM_USE_HIP
|
||||
|
||||
// Enable MFEM functionality based on the RAJA library
|
||||
#cmakedefine MFEM_USE_RAJA
|
||||
|
||||
// Enable MFEM functionality based on the OCCA library
|
||||
#cmakedefine MFEM_USE_OCCA
|
||||
|
||||
// Enable MFEM functionality based on the libCEED library
|
||||
#cmakedefine MFEM_USE_CEED
|
||||
|
||||
// Enable MFEM functionality based on the Umpire library
|
||||
#cmakedefine MFEM_USE_UMPIRE
|
||||
|
||||
// Enable MFEM functionality based on the ADIOS2 library
|
||||
#cmakedefine MFEM_USE_ADIOS2
|
||||
|
||||
// Enable MFEM functionality based on the Caliper library
|
||||
#cmakedefine MFEM_USE_CALIPER
|
||||
|
||||
// Which library functions to use in class StopWatch for measuring time.
|
||||
// For a list of the available options, see INSTALL.
|
||||
// If not defined, an option is selected automatically.
|
||||
@@ -162,6 +109,10 @@
|
||||
// Enable MFEM functionality based on the SUNDIALS libraries.
|
||||
#cmakedefine MFEM_USE_SUNDIALS
|
||||
|
||||
// Windows specific options
|
||||
// Macro needed to get defines like M_PI from <cmath>. (Visual Studio C++ only?)
|
||||
#cmakedefine _USE_MATH_DEFINES
|
||||
|
||||
// Version of HYPRE used for building MFEM.
|
||||
#cmakedefine MFEM_HYPRE_VERSION @MFEM_HYPRE_VERSION@
|
||||
|
||||
@@ -169,7 +120,4 @@
|
||||
// library.
|
||||
#cmakedefine MFEM_USE_SIMMETRIX
|
||||
|
||||
// Enable interface to the MKL CPardiso library.
|
||||
#cmakedefine MFEM_USE_MKL_CPARDISO
|
||||
|
||||
#endif // MFEM_CONFIG_HEADER
|
||||
|
||||
@@ -1,64 +0,0 @@
|
||||
#------------------------------------------------------------------------------#
|
||||
# Distributed under the OSI-approved Apache License, Version 2.0. See
|
||||
# accompanying file Copyright.txt for details.
|
||||
#------------------------------------------------------------------------------#
|
||||
#
|
||||
# FindADIOS2
|
||||
# -----------
|
||||
#
|
||||
# Try to find the ADIOS2 library
|
||||
#
|
||||
# This module defines the following variables:
|
||||
#
|
||||
# ADIOS2_FOUND - System has ADIOS2
|
||||
# ADIOS2_INCLUDE_DIRS - The ADIOS2 include directory
|
||||
# ADIOS2_LIBRARIES - Link these to use ADIOS2
|
||||
#
|
||||
# and the following imported targets:
|
||||
# ADIOS2::ADIOS2 - The ADIOS2 compression library target
|
||||
#
|
||||
# You can also set the following variable to help guide the search:
|
||||
# ADIOS2_DIR - The install prefix for ADIOS2 containing the
|
||||
# include and lib folders
|
||||
# Note: this can be set as a CMake variable or an
|
||||
# environment variable. If specified as a CMake
|
||||
# variable, it will override any setting specified
|
||||
# as an environment variable.
|
||||
|
||||
if(NOT ADIOS2_FOUND)
|
||||
if((NOT ADIOS2_DIR) AND (NOT (ENV{ADIOS2_DIR} STREQUAL "")))
|
||||
set(ADIOS2_DIR "$ENV{ADIOS2_DIR}")
|
||||
endif()
|
||||
if(ADIOS2_DIR)
|
||||
set(ADIOS2_INCLUDE_OPTS HINTS ${ADIOS2_DIR}/include NO_DEFAULT_PATHS)
|
||||
set(ADIOS2_LIBRARY_OPTS
|
||||
HINTS ${ADIOS2_DIR}/lib ${ADIOS2_DIR}/lib64
|
||||
NO_DEFAULT_PATHS
|
||||
)
|
||||
endif()
|
||||
|
||||
find_path(ADIOS2_INCLUDE_DIR adios2.h ${ADIOS2_INCLUDE_OPTS})
|
||||
|
||||
# adios2 version 2.5.0
|
||||
find_library(ADIOS2_LIBRARY NAMES adios2 ${ADIOS2_LIBRARY_OPTS})
|
||||
|
||||
# adios2 version 2.6.0 and onwards
|
||||
if(NOT ADIOS2_LIBRARY)
|
||||
find_library(ADIOS2_CXX11_MPI_LIBRARY NAMES adios2_cxx11_mpi ${ADIOS2_LIBRARY_OPTS})
|
||||
find_library(ADIOS2_CXX11_LIBRARY NAMES adios2_cxx11 ${ADIOS2_LIBRARY_OPTS})
|
||||
set(ADIOS2_LIBRARY ${ADIOS2_CXX11_MPI_LIBRARY} ${ADIOS2_CXX11_LIBRARY})
|
||||
if(MFEM_USE_MPI)
|
||||
add_definitions(-DADIOS2_USE_MPI)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
find_package_handle_standard_args(ADIOS2
|
||||
FOUND_VAR ADIOS2_FOUND
|
||||
REQUIRED_VARS ADIOS2_LIBRARY ADIOS2_INCLUDE_DIR
|
||||
)
|
||||
if(ADIOS2_FOUND)
|
||||
set(ADIOS2_INCLUDE_DIRS ${ADIOS2_INCLUDE_DIR})
|
||||
set(ADIOS2_LIBRARIES ${ADIOS2_LIBRARY})
|
||||
endif()
|
||||
endif()
|
||||
@@ -1,24 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Defines the following variables:
|
||||
# - AMGX_FOUND
|
||||
# - AMGX_LIBRARIES
|
||||
# - AMGX_INCLUDE_DIRS
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
set(AMGX_REQUIRED_LIBRARIES cusparse cusolver cublas cublasLt nvToolsExt)
|
||||
mfem_find_package(AMGX AMGX AMGX_DIR "include" "amgx_c.h" "lib" "amgx"
|
||||
"Paths to headers required by AMGX." "Libraries required by AMGX.")
|
||||
# Make sure the library location is locked down
|
||||
foreach(lib ${AMGX_REQUIRED_LIBRARIES})
|
||||
list(APPEND AMGX_LIBRARIES ${CUDA_TOOLKIT_ROOT_DIR}/lib64/lib${lib}${CMAKE_SHARED_LIBRARY_SUFFIX})
|
||||
endforeach()
|
||||
@@ -1,13 +1,13 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Defines the following variables:
|
||||
# - AXOM_FOUND
|
||||
@@ -18,4 +18,6 @@ include(MfemCmakeUtilities)
|
||||
# Note: components are enabled based on the find_package() parameters.
|
||||
mfem_find_package(Axom AXOM AXOM_DIR "include" "" "lib" ""
|
||||
"Paths to headers required by Axom." "Libraries required by Axom."
|
||||
ADD_COMPONENT Axom "include" axom/config.hpp "lib" axom)
|
||||
ADD_COMPONENT Sidre "include" sidre/sidre.hpp "lib" sidre
|
||||
ADD_COMPONENT SLIC "include" slic/slic.hpp "lib" slic
|
||||
ADD_COMPONENT axom_utils "include" axom_utils/Utilities.hpp "lib" axom_utils)
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Check for "abi::__cxa_demangle" in <cxxabi.h>.
|
||||
# Defines the variables:
|
||||
|
||||
@@ -1,22 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Defines the following variables:
|
||||
# - CALIPER_FOUND
|
||||
# - CALIPER_LIBRARIES
|
||||
# - CALIPER_INCLUDE_DIRS
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(Caliper CALIPER CALIPER_DIR
|
||||
"include" "caliper/cali.h"
|
||||
"lib" "caliper"
|
||||
"Paths to headers required by Caliper."
|
||||
"Libraries required by Caliper.")
|
||||
@@ -1,13 +1,13 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Defines the following variables:
|
||||
# - CONDUIT_FOUND
|
||||
|
||||
@@ -1,19 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Defines the following variables:
|
||||
# - GSLIB_FOUND
|
||||
# - GSLIB_LIBRARIES
|
||||
# - GSLIB_INCLUDE_DIRS
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(gslib GSLIB GSLIB_DIR "include" gslib.h "lib" gs
|
||||
"Paths to headers required by GSLIB." "Libraries required by GSLIB.")
|
||||
@@ -0,0 +1,19 @@
|
||||
# Copyright (c) 2010, Lawrence Livermore National Security, LLC. Produced at the
|
||||
# Lawrence Livermore National Laboratory. LLNL-CODE-443211. All Rights reserved.
|
||||
# See file COPYRIGHT for details.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability see http://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the GNU Lesser General Public License (as published by the Free
|
||||
# Software Foundation) version 2.1 dated February 1999.
|
||||
|
||||
# Defines the following variables:
|
||||
# - GECKO_FOUND
|
||||
# - GECKO_LIBRARIES
|
||||
# - GECKO_INCLUDE_DIRS
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(Gecko GECKO GECKO_DIR "include;inc" graph.h "lib" gecko
|
||||
"Paths to headers required by Gecko." "Libraries required by Gecko.")
|
||||
@@ -1,36 +0,0 @@
|
||||
# Copyright (c) 2010-2021, Lawrence Livermore National Security, LLC. Produced
|
||||
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
|
||||
# LICENSE and NOTICE for details. LLNL-CODE-806117.
|
||||
#
|
||||
# This file is part of the MFEM library. For more information and source code
|
||||
# availability visit https://mfem.org.
|
||||
#
|
||||
# MFEM is free software; you can redistribute it and/or modify it under the
|
||||
# terms of the BSD-3 license. We welcome feedback and contributions, see file
|
||||
# CONTRIBUTING.md for details.
|
||||
|
||||
# Sets the following variables:
|
||||
# - HIOP_FOUND
|
||||
# - HIOP_INCLUDE_DIRS
|
||||
# - HIOP_LIBRARIES
|
||||
|
||||
include(MfemCmakeUtilities)
|
||||
mfem_find_package(HIOP HIOP HIOP_DIR
|
||||
"include" "hiopInterface.hpp"
|
||||
"lib" "hiop"
|
||||
"Paths to headers required by HIOP."
|
||||
"Libraries required by HIOP.")
|
||||
|
||||
# this test fails with parallel MFEM since mpi.h is not available (cxx compiler is used for some reason)
|
||||
# CHECK_BUILD HIOP_VERSION_OK TRUE
|
||||
#"
|
||||
##include <hiopInterface.hpp>
|
||||
#using namespace hiop;
|
||||
#int main(int argc, char *argv[])
|
||||
#{
|
||||
# MPI_Init(&argc, &argv);
|
||||
# MPI_Comm comm = MPI_COMM_WORLD;
|
||||
#
|
||||
# return 0;
|
||||
#}
|
||||
#")
|
||||
@@ -1,692 +0,0 @@
|
||||
###############################################################################
|
||||
# FindHIP.cmake
|
||||
###############################################################################
|
||||
include(CheckCXXCompilerFlag)
|
||||
###############################################################################
|
||||
# SET: Variable defaults
|
||||
###############################################################################
|
||||
# User defined flags
|
||||
set(HIP_HIPCC_FLAGS "" CACHE STRING "Semicolon delimited flags for HIPCC")
|
||||
set(HIP_HCC_FLAGS "" CACHE STRING "Semicolon delimited flags for HCC")
|
||||
set(HIP_CLANG_FLAGS "" CACHE STRING "Semicolon delimited flags for CLANG")
|
||||
set(HIP_NVCC_FLAGS "" CACHE STRING "Semicolon delimted flags for NVCC")
|
||||
mark_as_advanced(HIP_HIPCC_FLAGS HIP_HCC_FLAGS HIP_CLANG_FLAGS HIP_NVCC_FLAGS)
|
||||
|
||||
set(_hip_configuration_types ${CMAKE_CONFIGURATION_TYPES} ${CMAKE_BUILD_TYPE} Debug MinSizeRel Release RelWithDebInfo)
|
||||
list(REMOVE_DUPLICATES _hip_configuration_types)
|
||||
foreach(config ${_hip_configuration_types})
|
||||
string(TOUPPER ${config} config_upper)
|
||||
set(HIP_HIPCC_FLAGS_${config_upper} "" CACHE STRING "Semicolon delimited flags for HIPCC")
|
||||
set(HIP_HCC_FLAGS_${config_upper} "" CACHE STRING "Semicolon delimited flags for HCC")
|
||||
set(HIP_CLANG_FLAGS_${config_upper} "" CACHE STRING "Semicolon delimited flags for CLANG")
|
||||
set(HIP_NVCC_FLAGS_${config_upper} "" CACHE STRING "Semicolon delimited flags for NVCC")
|
||||
mark_as_advanced(HIP_HIPCC_FLAGS_${config_upper} HIP_HCC_FLAGS_${config_upper} HIP_CLANG_FLAGS_${config_upper} HIP_NVCC_FLAGS_${config_upper})
|
||||
endforeach()
|
||||
option(HIP_HOST_COMPILATION_CPP "Host code compilation mode" ON)
|
||||
option(HIP_VERBOSE_BUILD "Print out the commands run while compiling the HIP source file. With the Makefile generator this defaults to VERBOSE variable specified on the command line, but can be forced on with this option." OFF)
|
||||
mark_as_advanced(HIP_HOST_COMPILATION_CPP)
|
||||
|
||||
###############################################################################
|
||||
# FIND: HIP and associated helper binaries
|
||||
###############################################################################
|
||||
|
||||
get_filename_component(_IMPORT_PREFIX "${CMAKE_CURRENT_LIST_DIR}/../" REALPATH)
|
||||
|
||||
# HIP is supported on Linux only
|
||||
if(UNIX AND NOT APPLE AND NOT CYGWIN)
|
||||
# Search for HIP installation
|
||||
if(NOT HIP_ROOT_DIR)
|
||||
# Search in user specified path first
|
||||
find_path(
|
||||
HIP_ROOT_DIR
|
||||
NAMES bin/hipconfig
|
||||
PATHS
|
||||
"$ENV{ROCM_PATH}/hip"
|
||||
ENV HIP_PATH
|
||||
${_IMPORT_PREFIX}
|
||||
/opt/rocm/hip
|
||||
DOC "HIP installed location"
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
if(NOT EXISTS ${HIP_ROOT_DIR})
|
||||
if(HIP_FIND_REQUIRED)
|
||||
message(FATAL_ERROR "Specify HIP_ROOT_DIR")
|
||||
elseif(NOT HIP_FIND_QUIETLY)
|
||||
message("HIP_ROOT_DIR not found or specified")
|
||||
endif()
|
||||
endif()
|
||||
# And push it back to the cache
|
||||
set(HIP_ROOT_DIR ${HIP_ROOT_DIR} CACHE PATH "HIP installed location" FORCE)
|
||||
endif()
|
||||
|
||||
# Find HIPCC executable
|
||||
find_program(
|
||||
HIP_HIPCC_EXECUTABLE
|
||||
NAMES hipcc
|
||||
PATHS
|
||||
"${HIP_ROOT_DIR}"
|
||||
ENV ROCM_PATH
|
||||
ENV HIP_PATH
|
||||
/opt/rocm
|
||||
/opt/rocm/hip
|
||||
PATH_SUFFIXES bin
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
if(NOT HIP_HIPCC_EXECUTABLE)
|
||||
# Now search in default paths
|
||||
find_program(HIP_HIPCC_EXECUTABLE hipcc)
|
||||
endif()
|
||||
mark_as_advanced(HIP_HIPCC_EXECUTABLE)
|
||||
|
||||
# Find HIPCONFIG executable
|
||||
find_program(
|
||||
HIP_HIPCONFIG_EXECUTABLE
|
||||
NAMES hipconfig
|
||||
PATHS
|
||||
"${HIP_ROOT_DIR}"
|
||||
ENV ROCM_PATH
|
||||
ENV HIP_PATH
|
||||
/opt/rocm
|
||||
/opt/rocm/hip
|
||||
PATH_SUFFIXES bin
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
if(NOT HIP_HIPCONFIG_EXECUTABLE)
|
||||
# Now search in default paths
|
||||
find_program(HIP_HIPCONFIG_EXECUTABLE hipconfig)
|
||||
endif()
|
||||
mark_as_advanced(HIP_HIPCONFIG_EXECUTABLE)
|
||||
|
||||
# Find HIPCC_CMAKE_LINKER_HELPER executable
|
||||
find_program(
|
||||
HIP_HIPCC_CMAKE_LINKER_HELPER
|
||||
NAMES hipcc_cmake_linker_helper
|
||||
PATHS
|
||||
"${HIP_ROOT_DIR}"
|
||||
ENV ROCM_PATH
|
||||
ENV HIP_PATH
|
||||
/opt/rocm
|
||||
/opt/rocm/hip
|
||||
PATH_SUFFIXES bin
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
if(NOT HIP_HIPCC_CMAKE_LINKER_HELPER)
|
||||
# Now search in default paths
|
||||
find_program(HIP_HIPCC_CMAKE_LINKER_HELPER hipcc_cmake_linker_helper)
|
||||
endif()
|
||||
mark_as_advanced(HIP_HIPCC_CMAKE_LINKER_HELPER)
|
||||
|
||||
if(HIP_HIPCONFIG_EXECUTABLE AND NOT HIP_VERSION)
|
||||
# Compute the version
|
||||
execute_process(
|
||||
COMMAND ${HIP_HIPCONFIG_EXECUTABLE} --version
|
||||
OUTPUT_VARIABLE _hip_version
|
||||
ERROR_VARIABLE _hip_error
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
ERROR_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
if(NOT _hip_error)
|
||||
set(HIP_VERSION ${_hip_version} CACHE STRING "Version of HIP as computed from hipcc")
|
||||
else()
|
||||
set(HIP_VERSION "0.0.0" CACHE STRING "Version of HIP as computed by FindHIP()")
|
||||
endif()
|
||||
mark_as_advanced(HIP_VERSION)
|
||||
endif()
|
||||
if(HIP_VERSION)
|
||||
string(REPLACE "." ";" _hip_version_list "${HIP_VERSION}")
|
||||
list(GET _hip_version_list 0 HIP_VERSION_MAJOR)
|
||||
list(GET _hip_version_list 1 HIP_VERSION_MINOR)
|
||||
list(GET _hip_version_list 2 HIP_VERSION_PATCH)
|
||||
set(HIP_VERSION_STRING "${HIP_VERSION}")
|
||||
endif()
|
||||
|
||||
if(HIP_HIPCONFIG_EXECUTABLE AND NOT HIP_PLATFORM)
|
||||
# Compute the platform
|
||||
execute_process(
|
||||
COMMAND ${HIP_HIPCONFIG_EXECUTABLE} --platform
|
||||
OUTPUT_VARIABLE _hip_platform
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
set(HIP_PLATFORM ${_hip_platform} CACHE STRING "HIP platform as computed by hipconfig")
|
||||
mark_as_advanced(HIP_PLATFORM)
|
||||
endif()
|
||||
|
||||
if(HIP_HIPCONFIG_EXECUTABLE AND NOT HIP_COMPILER)
|
||||
# Compute the compiler
|
||||
execute_process(
|
||||
COMMAND ${HIP_HIPCONFIG_EXECUTABLE} --compiler
|
||||
OUTPUT_VARIABLE _hip_compiler
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
set(HIP_COMPILER ${_hip_compiler} CACHE STRING "HIP compiler as computed by hipconfig")
|
||||
mark_as_advanced(HIP_COMPILER)
|
||||
endif()
|
||||
|
||||
if(HIP_HIPCONFIG_EXECUTABLE AND NOT HIP_RUNTIME)
|
||||
# Compute the runtime
|
||||
execute_process(
|
||||
COMMAND ${HIP_HIPCONFIG_EXECUTABLE} --runtime
|
||||
OUTPUT_VARIABLE _hip_runtime
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
set(HIP_RUNTIME ${_hip_runtime} CACHE STRING "HIP runtime as computed by hipconfig")
|
||||
mark_as_advanced(HIP_RUNTIME)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
find_package_handle_standard_args(
|
||||
HIP
|
||||
REQUIRED_VARS
|
||||
HIP_ROOT_DIR
|
||||
HIP_HIPCC_EXECUTABLE
|
||||
HIP_HIPCONFIG_EXECUTABLE
|
||||
HIP_PLATFORM
|
||||
HIP_COMPILER
|
||||
HIP_RUNTIME
|
||||
VERSION_VAR HIP_VERSION
|
||||
)
|
||||
|
||||
###############################################################################
|
||||
# Set HIP CMAKE Flags
|
||||
###############################################################################
|
||||
# Copy the invocation styles from CXX to HIP
|
||||
set(CMAKE_HIP_ARCHIVE_CREATE ${CMAKE_CXX_ARCHIVE_CREATE})
|
||||
set(CMAKE_HIP_ARCHIVE_APPEND ${CMAKE_CXX_ARCHIVE_APPEND})
|
||||
set(CMAKE_HIP_ARCHIVE_FINISH ${CMAKE_CXX_ARCHIVE_FINISH})
|
||||
set(CMAKE_SHARED_LIBRARY_SONAME_HIP_FLAG ${CMAKE_SHARED_LIBRARY_SONAME_CXX_FLAG})
|
||||
set(CMAKE_SHARED_LIBRARY_CREATE_HIP_FLAGS ${CMAKE_SHARED_LIBRARY_CREATE_CXX_FLAGS})
|
||||
set(CMAKE_SHARED_LIBRARY_HIP_FLAGS ${CMAKE_SHARED_LIBRARY_CXX_FLAGS})
|
||||
#set(CMAKE_SHARED_LIBRARY_LINK_HIP_FLAGS ${CMAKE_SHARED_LIBRARY_LINK_CXX_FLAGS})
|
||||
set(CMAKE_SHARED_LIBRARY_RUNTIME_HIP_FLAG ${CMAKE_SHARED_LIBRARY_RUNTIME_CXX_FLAG})
|
||||
set(CMAKE_SHARED_LIBRARY_RUNTIME_HIP_FLAG_SEP ${CMAKE_SHARED_LIBRARY_RUNTIME_CXX_FLAG_SEP})
|
||||
set(CMAKE_SHARED_LIBRARY_LINK_STATIC_HIP_FLAGS ${CMAKE_SHARED_LIBRARY_LINK_STATIC_CXX_FLAGS})
|
||||
set(CMAKE_SHARED_LIBRARY_LINK_DYNAMIC_HIP_FLAGS ${CMAKE_SHARED_LIBRARY_LINK_DYNAMIC_CXX_FLAGS})
|
||||
|
||||
set(HIP_CLANG_PARALLEL_BUILD_COMPILE_OPTIONS "")
|
||||
set(HIP_CLANG_PARALLEL_BUILD_LINK_OPTIONS "")
|
||||
|
||||
if("${HIP_COMPILER}" STREQUAL "nvcc")
|
||||
# Set the CMake Flags to use the nvcc Compiler.
|
||||
set(CMAKE_HIP_CREATE_SHARED_LIBRARY "${HIP_HIPCC_CMAKE_LINKER_HELPER} <CMAKE_SHARED_LIBRARY_CXX_FLAGS> <LANGUAGE_COMPILE_FLAGS> <LINK_FLAGS> <CMAKE_SHARED_LIBRARY_CREATE_CXX_FLAGS> <SONAME_FLAG><TARGET_SONAME> -o <TARGET> <OBJECTS> <LINK_LIBRARIES>")
|
||||
set(CMAKE_HIP_CREATE_SHARED_MODULE "${HIP_HIPCC_CMAKE_LINKER_HELPER} <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> <SONAME_FLAG><TARGET_SONAME> -o <TARGET> <LINK_LIBRARIES> -shared" )
|
||||
set(CMAKE_HIP_LINK_EXECUTABLE "${HIP_HIPCC_CMAKE_LINKER_HELPER} <FLAGS> <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> -o <TARGET> <LINK_LIBRARIES>")
|
||||
elseif("${HIP_COMPILER}" STREQUAL "hcc")
|
||||
# Set the CMake Flags to use the hcc Compiler.
|
||||
set(CMAKE_HIP_CREATE_SHARED_LIBRARY "${HIP_HIPCC_CMAKE_LINKER_HELPER} ${HCC_HOME} <CMAKE_SHARED_LIBRARY_CXX_FLAGS> <LANGUAGE_COMPILE_FLAGS> <LINK_FLAGS> <CMAKE_SHARED_LIBRARY_CREATE_CXX_FLAGS> <SONAME_FLAG><TARGET_SONAME> -o <TARGET> <OBJECTS> <LINK_LIBRARIES>")
|
||||
set(CMAKE_HIP_CREATE_SHARED_MODULE "${HIP_HIPCC_CMAKE_LINKER_HELPER} ${HCC_HOME} <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> <SONAME_FLAG><TARGET_SONAME> -o <TARGET> <LINK_LIBRARIES> -shared" )
|
||||
set(CMAKE_HIP_LINK_EXECUTABLE "${HIP_HIPCC_CMAKE_LINKER_HELPER} ${HCC_HOME} <FLAGS> <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> -o <TARGET> <LINK_LIBRARIES>")
|
||||
elseif("${HIP_COMPILER}" STREQUAL "clang")
|
||||
#Number of parallel jobs by default is 1
|
||||
if(NOT DEFINED HIP_CLANG_NUM_PARALLEL_JOBS)
|
||||
set(HIP_CLANG_NUM_PARALLEL_JOBS 1)
|
||||
endif()
|
||||
#Add support for parallel build and link
|
||||
if(${CMAKE_CXX_COMPILER_ID} STREQUAL "Clang")
|
||||
check_cxx_compiler_flag("-parallel-jobs=1" HIP_CLANG_SUPPORTS_PARALLEL_JOBS)
|
||||
endif()
|
||||
if(HIP_CLANG_NUM_PARALLEL_JOBS GREATER 1)
|
||||
if(${HIP_CLANG_SUPPORTS_PARALLEL_JOBS})
|
||||
set(HIP_CLANG_PARALLEL_BUILD_COMPILE_OPTIONS "-Wno-format-nonliteral -parallel-jobs=${HIP_CLANG_NUM_PARALLEL_JOBS}")
|
||||
set(HIP_CLANG_PARALLEL_BUILD_LINK_OPTIONS "-parallel-jobs=${HIP_CLANG_NUM_PARALLEL_JOBS}")
|
||||
else()
|
||||
message("clang compiler doesn't support parallel jobs")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Set the CMake Flags to use the HIP-Clang Compiler.
|
||||
set(CMAKE_HIP_CREATE_SHARED_LIBRARY "${HIP_HIPCC_CMAKE_LINKER_HELPER} ${HIP_CLANG_PATH} ${HIP_CLANG_PARALLEL_BUILD_LINK_OPTIONS} <CMAKE_SHARED_LIBRARY_CXX_FLAGS> <LANGUAGE_COMPILE_FLAGS> <LINK_FLAGS> <CMAKE_SHARED_LIBRARY_CREATE_CXX_FLAGS> <SONAME_FLAG><TARGET_SONAME> -o <TARGET> <OBJECTS> <LINK_LIBRARIES>")
|
||||
set(CMAKE_HIP_CREATE_SHARED_MODULE "${HIP_HIPCC_CMAKE_LINKER_HELPER} ${HIP_CLANG_PATH} ${HIP_CLANG_PARALLEL_BUILD_LINK_OPTIONS} <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> <SONAME_FLAG><TARGET_SONAME> -o <TARGET> <LINK_LIBRARIES> -shared" )
|
||||
set(CMAKE_HIP_LINK_EXECUTABLE "${HIP_HIPCC_CMAKE_LINKER_HELPER} ${HIP_CLANG_PATH} ${HIP_CLANG_PARALLEL_BUILD_LINK_OPTIONS} <FLAGS> <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> -o <TARGET> <LINK_LIBRARIES>")
|
||||
|
||||
if("${HIP_RUNTIME}" STREQUAL "rocclr")
|
||||
if(TARGET host)
|
||||
message(STATUS "host interface - found")
|
||||
set(HIP_HOST_INTERFACE host)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
###############################################################################
|
||||
# MACRO: Locate helper files
|
||||
###############################################################################
|
||||
macro(HIP_FIND_HELPER_FILE _name _extension)
|
||||
set(_hip_full_name "${_name}.${_extension}")
|
||||
get_filename_component(CMAKE_CURRENT_LIST_DIR "${CMAKE_CURRENT_LIST_FILE}" PATH)
|
||||
set(HIP_${_name} "${CMAKE_CURRENT_LIST_DIR}/FindHIP/${_hip_full_name}")
|
||||
if(NOT EXISTS "${HIP_${_name}}")
|
||||
set(error_message "${_hip_full_name} not found in ${CMAKE_CURRENT_LIST_DIR}/FindHIP")
|
||||
if(HIP_FIND_REQUIRED)
|
||||
message(FATAL_ERROR "${error_message}")
|
||||
else()
|
||||
if(NOT HIP_FIND_QUIETLY)
|
||||
message(STATUS "${error_message}")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
# Set this variable as internal, so the user isn't bugged with it.
|
||||
set(HIP_${_name} ${HIP_${_name}} CACHE INTERNAL "Location of ${_full_name}" FORCE)
|
||||
endmacro()
|
||||
|
||||
###############################################################################
|
||||
hip_find_helper_file(run_make2cmake cmake)
|
||||
hip_find_helper_file(run_hipcc cmake)
|
||||
###############################################################################
|
||||
|
||||
###############################################################################
|
||||
# MACRO: Reset compiler flags
|
||||
###############################################################################
|
||||
macro(HIP_RESET_FLAGS)
|
||||
unset(HIP_HIPCC_FLAGS)
|
||||
unset(HIP_HCC_FLAGS)
|
||||
unset(HIP_CLANG_FLAGS)
|
||||
unset(HIP_NVCC_FLAGS)
|
||||
foreach(config ${_hip_configuration_types})
|
||||
string(TOUPPER ${config} config_upper)
|
||||
unset(HIP_HIPCC_FLAGS_${config_upper})
|
||||
unset(HIP_HCC_FLAGS_${config_upper})
|
||||
unset(HIP_CLANG_FLAGS_${config_upper})
|
||||
unset(HIP_NVCC_FLAGS_${config_upper})
|
||||
endforeach()
|
||||
endmacro()
|
||||
|
||||
###############################################################################
|
||||
# MACRO: Separate the options from the sources
|
||||
###############################################################################
|
||||
macro(HIP_GET_SOURCES_AND_OPTIONS _sources _cmake_options _hipcc_options _hcc_options _clang_options _nvcc_options)
|
||||
set(${_sources})
|
||||
set(${_cmake_options})
|
||||
set(${_hipcc_options})
|
||||
set(${_hcc_options})
|
||||
set(${_clang_options})
|
||||
set(${_nvcc_options})
|
||||
set(_hipcc_found_options FALSE)
|
||||
set(_hcc_found_options FALSE)
|
||||
set(_clang_found_options FALSE)
|
||||
set(_nvcc_found_options FALSE)
|
||||
foreach(arg ${ARGN})
|
||||
if("x${arg}" STREQUAL "xHIPCC_OPTIONS")
|
||||
set(_hipcc_found_options TRUE)
|
||||
set(_hcc_found_options FALSE)
|
||||
set(_clang_found_options FALSE)
|
||||
set(_nvcc_found_options FALSE)
|
||||
elseif("x${arg}" STREQUAL "xHCC_OPTIONS")
|
||||
set(_hipcc_found_options FALSE)
|
||||
set(_hcc_found_options TRUE)
|
||||
set(_clang_found_options FALSE)
|
||||
set(_nvcc_found_options FALSE)
|
||||
elseif("x${arg}" STREQUAL "xCLANG_OPTIONS")
|
||||
set(_hipcc_found_options FALSE)
|
||||
set(_hcc_found_options FALSE)
|
||||
set(_clang_found_options TRUE)
|
||||
set(_nvcc_found_options FALSE)
|
||||
elseif("x${arg}" STREQUAL "xNVCC_OPTIONS")
|
||||
set(_hipcc_found_options FALSE)
|
||||
set(_hcc_found_options FALSE)
|
||||
set(_clang_found_options FALSE)
|
||||
set(_nvcc_found_options TRUE)
|
||||
elseif(
|
||||
"x${arg}" STREQUAL "xEXCLUDE_FROM_ALL" OR
|
||||
"x${arg}" STREQUAL "xSTATIC" OR
|
||||
"x${arg}" STREQUAL "xSHARED" OR
|
||||
"x${arg}" STREQUAL "xMODULE"
|
||||
)
|
||||
list(APPEND ${_cmake_options} ${arg})
|
||||
else()
|
||||
if(_hipcc_found_options)
|
||||
list(APPEND ${_hipcc_options} ${arg})
|
||||
elseif(_hcc_found_options)
|
||||
list(APPEND ${_hcc_options} ${arg})
|
||||
elseif(_clang_found_options)
|
||||
list(APPEND ${_clang_options} ${arg})
|
||||
elseif(_nvcc_found_options)
|
||||
list(APPEND ${_nvcc_options} ${arg})
|
||||
else()
|
||||
# Assume this is a file
|
||||
list(APPEND ${_sources} ${arg})
|
||||
endif()
|
||||
endif()
|
||||
endforeach()
|
||||
endmacro()
|
||||
|
||||
###############################################################################
|
||||
# MACRO: Add include directories to pass to the hipcc command
|
||||
###############################################################################
|
||||
set(HIP_HIPCC_INCLUDE_ARGS_USER "")
|
||||
macro(HIP_INCLUDE_DIRECTORIES)
|
||||
foreach(dir ${ARGN})
|
||||
list(APPEND HIP_HIPCC_INCLUDE_ARGS_USER $<$<BOOL:${dir}>:-I${dir}>)
|
||||
endforeach()
|
||||
endmacro()
|
||||
|
||||
###############################################################################
|
||||
# FUNCTION: Helper to avoid clashes of files with the same basename but different paths
|
||||
###############################################################################
|
||||
function(HIP_COMPUTE_BUILD_PATH path build_path)
|
||||
# Convert to cmake style paths
|
||||
file(TO_CMAKE_PATH "${path}" bpath)
|
||||
if(IS_ABSOLUTE "${bpath}")
|
||||
string(FIND "${bpath}" "${CMAKE_CURRENT_BINARY_DIR}" _binary_dir_pos)
|
||||
if(_binary_dir_pos EQUAL 0)
|
||||
file(RELATIVE_PATH bpath "${CMAKE_CURRENT_BINARY_DIR}" "${bpath}")
|
||||
else()
|
||||
file(RELATIVE_PATH bpath "${CMAKE_CURRENT_SOURCE_DIR}" "${bpath}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Remove leading /
|
||||
string(REGEX REPLACE "^[/]+" "" bpath "${bpath}")
|
||||
# Avoid absolute paths by removing ':'
|
||||
string(REPLACE ":" "_" bpath "${bpath}")
|
||||
# Avoid relative paths that go up the tree
|
||||
string(REPLACE "../" "__/" bpath "${bpath}")
|
||||
# Avoid spaces
|
||||
string(REPLACE " " "_" bpath "${bpath}")
|
||||
# Strip off the filename
|
||||
get_filename_component(bpath "${bpath}" PATH)
|
||||
|
||||
set(${build_path} "${bpath}" PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
###############################################################################
|
||||
# MACRO: Parse OPTIONS from ARGN & set variables prefixed by _option_prefix
|
||||
###############################################################################
|
||||
macro(HIP_PARSE_HIPCC_OPTIONS _option_prefix)
|
||||
set(_hip_found_config)
|
||||
foreach(arg ${ARGN})
|
||||
# Determine if we are dealing with a per-configuration flag
|
||||
foreach(config ${_hip_configuration_types})
|
||||
string(TOUPPER ${config} config_upper)
|
||||
if(arg STREQUAL "${config_upper}")
|
||||
set(_hip_found_config _${arg})
|
||||
# Clear arg to prevent it from being processed anymore
|
||||
set(arg)
|
||||
endif()
|
||||
endforeach()
|
||||
if(arg)
|
||||
list(APPEND ${_option_prefix}${_hip_found_config} "${arg}")
|
||||
endif()
|
||||
endforeach()
|
||||
endmacro()
|
||||
|
||||
###############################################################################
|
||||
# MACRO: Try and include dependency file if it exists
|
||||
###############################################################################
|
||||
macro(HIP_INCLUDE_HIPCC_DEPENDENCIES dependency_file)
|
||||
set(HIP_HIPCC_DEPEND)
|
||||
set(HIP_HIPCC_DEPEND_REGENERATE FALSE)
|
||||
|
||||
# Create the dependency file if it doesn't exist
|
||||
if(NOT EXISTS ${dependency_file})
|
||||
file(WRITE ${dependency_file} "# Generated by: FindHIP.cmake. Do not edit.\n")
|
||||
endif()
|
||||
# Include the dependency file
|
||||
include(${dependency_file})
|
||||
|
||||
# Verify the existence of all the included files
|
||||
if(HIP_HIPCC_DEPEND)
|
||||
foreach(f ${HIP_HIPCC_DEPEND})
|
||||
if(NOT EXISTS ${f})
|
||||
# If they aren't there, regenerate the file again
|
||||
set(HIP_HIPCC_DEPEND_REGENERATE TRUE)
|
||||
endif()
|
||||
endforeach()
|
||||
else()
|
||||
# No dependencies, so regenerate the file
|
||||
set(HIP_HIPCC_DEPEND_REGENERATE TRUE)
|
||||
endif()
|
||||
|
||||
# Regenerate the dependency file if needed
|
||||
if(HIP_HIPCC_DEPEND_REGENERATE)
|
||||
set(HIP_HIPCC_DEPEND ${dependency_file})
|
||||
file(WRITE ${dependency_file} "# Generated by: FindHIP.cmake. Do not edit.\n")
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
###############################################################################
|
||||
# MACRO: Prepare cmake commands for the target
|
||||
###############################################################################
|
||||
macro(HIP_PREPARE_TARGET_COMMANDS _target _format _generated_files _source_files)
|
||||
set(_hip_flags "")
|
||||
string(TOUPPER "${CMAKE_BUILD_TYPE}" _hip_build_configuration)
|
||||
if(HIP_HOST_COMPILATION_CPP)
|
||||
set(HIP_C_OR_CXX CXX)
|
||||
else()
|
||||
set(HIP_C_OR_CXX C)
|
||||
endif()
|
||||
set(generated_extension ${CMAKE_${HIP_C_OR_CXX}_OUTPUT_EXTENSION})
|
||||
|
||||
# Initialize list of includes with those specified by the user. Append with
|
||||
# ones specified to cmake directly.
|
||||
set(HIP_HIPCC_INCLUDE_ARGS ${HIP_HIPCC_INCLUDE_ARGS_USER})
|
||||
|
||||
# Add the include directories
|
||||
set(include_directories_generator "$<TARGET_PROPERTY:${_target},INCLUDE_DIRECTORIES>")
|
||||
list(APPEND HIP_HIPCC_INCLUDE_ARGS "$<$<BOOL:${include_directories_generator}>:-I$<JOIN:${include_directories_generator}, -I>>")
|
||||
|
||||
get_directory_property(_hip_include_directories INCLUDE_DIRECTORIES)
|
||||
list(REMOVE_DUPLICATES _hip_include_directories)
|
||||
if(_hip_include_directories)
|
||||
foreach(dir ${_hip_include_directories})
|
||||
list(APPEND HIP_HIPCC_INCLUDE_ARGS $<$<BOOL:${dir}>:-I${dir}>)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
HIP_GET_SOURCES_AND_OPTIONS(_hip_sources _hip_cmake_options _hipcc_options _hcc_options _clang_options _nvcc_options ${ARGN})
|
||||
HIP_PARSE_HIPCC_OPTIONS(HIP_HIPCC_FLAGS ${_hipcc_options})
|
||||
HIP_PARSE_HIPCC_OPTIONS(HIP_HCC_FLAGS ${_hcc_options})
|
||||
HIP_PARSE_HIPCC_OPTIONS(HIP_CLANG_FLAGS ${_clang_options})
|
||||
HIP_PARSE_HIPCC_OPTIONS(HIP_NVCC_FLAGS ${_nvcc_options})
|
||||
|
||||
# Add the compile definitions
|
||||
set(compile_definition_generator "$<TARGET_PROPERTY:${_target},COMPILE_DEFINITIONS>")
|
||||
list(APPEND HIP_HIPCC_FLAGS "$<$<BOOL:${compile_definition_generator}>:-D$<JOIN:${compile_definition_generator}, -D>>")
|
||||
|
||||
# Check if we are building shared library.
|
||||
set(_hip_build_shared_libs FALSE)
|
||||
list(FIND _hip_cmake_options SHARED _hip_found_SHARED)
|
||||
list(FIND _hip_cmake_options MODULE _hip_found_MODULE)
|
||||
if(_hip_found_SHARED GREATER -1 OR _hip_found_MODULE GREATER -1)
|
||||
set(_hip_build_shared_libs TRUE)
|
||||
endif()
|
||||
list(FIND _hip_cmake_options STATIC _hip_found_STATIC)
|
||||
if(_hip_found_STATIC GREATER -1)
|
||||
set(_hip_build_shared_libs FALSE)
|
||||
endif()
|
||||
|
||||
# If we are building a shared library, add extra flags to HIP_HIPCC_FLAGS
|
||||
if(_hip_build_shared_libs)
|
||||
list(APPEND HIP_HCC_FLAGS "-fPIC")
|
||||
list(APPEND HIP_CLANG_FLAGS "-fPIC")
|
||||
list(APPEND HIP_NVCC_FLAGS "--shared -Xcompiler '-fPIC'")
|
||||
endif()
|
||||
|
||||
# Set host compiler
|
||||
set(HIP_HOST_COMPILER "${CMAKE_${HIP_C_OR_CXX}_COMPILER}")
|
||||
|
||||
# Set compiler flags
|
||||
set(_HIP_HOST_FLAGS "set(CMAKE_HOST_FLAGS ${CMAKE_${HIP_C_OR_CXX}_FLAGS})")
|
||||
set(_HIP_HIPCC_FLAGS "set(HIP_HIPCC_FLAGS ${HIP_HIPCC_FLAGS})")
|
||||
set(_HIP_HCC_FLAGS "set(HIP_HCC_FLAGS ${HIP_HCC_FLAGS})")
|
||||
set(_HIP_CLANG_FLAGS "set(HIP_CLANG_FLAGS ${HIP_CLANG_FLAGS})")
|
||||
set(_HIP_NVCC_FLAGS "set(HIP_NVCC_FLAGS ${HIP_NVCC_FLAGS})")
|
||||
foreach(config ${_hip_configuration_types})
|
||||
string(TOUPPER ${config} config_upper)
|
||||
set(_HIP_HOST_FLAGS "${_HIP_HOST_FLAGS}\nset(CMAKE_HOST_FLAGS_${config_upper} ${CMAKE_${HIP_C_OR_CXX}_FLAGS_${config_upper}})")
|
||||
set(_HIP_HIPCC_FLAGS "${_HIP_HIPCC_FLAGS}\nset(HIP_HIPCC_FLAGS_${config_upper} ${HIP_HIPCC_FLAGS_${config_upper}})")
|
||||
set(_HIP_HCC_FLAGS "${_HIP_HCC_FLAGS}\nset(HIP_HCC_FLAGS_${config_upper} ${HIP_HCC_FLAGS_${config_upper}})")
|
||||
set(_HIP_CLANG_FLAGS "${_HIP_CLANG_FLAGS}\nset(HIP_CLANG_FLAGS_${config_upper} ${HIP_CLANG_FLAGS_${config_upper}})")
|
||||
set(_HIP_NVCC_FLAGS "${_HIP_NVCC_FLAGS}\nset(HIP_NVCC_FLAGS_${config_upper} ${HIP_NVCC_FLAGS_${config_upper}})")
|
||||
endforeach()
|
||||
|
||||
# Reset the output variable
|
||||
set(_hip_generated_files "")
|
||||
set(_hip_source_files "")
|
||||
|
||||
# Iterate over all arguments and create custom commands for all source files
|
||||
foreach(file ${ARGN})
|
||||
# Ignore any file marked as a HEADER_FILE_ONLY
|
||||
get_source_file_property(_is_header ${file} HEADER_FILE_ONLY)
|
||||
# Allow per source file overrides of the format. Also allows compiling non .cu files.
|
||||
get_source_file_property(_hip_source_format ${file} HIP_SOURCE_PROPERTY_FORMAT)
|
||||
if((${file} MATCHES "\\.cu$" OR _hip_source_format) AND NOT _is_header)
|
||||
set(host_flag FALSE)
|
||||
else()
|
||||
set(host_flag TRUE)
|
||||
endif()
|
||||
|
||||
if(NOT host_flag)
|
||||
# Determine output directory
|
||||
HIP_COMPUTE_BUILD_PATH("${file}" hip_build_path)
|
||||
set(hip_compile_output_dir "${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/${_target}.dir/${hip_build_path}")
|
||||
|
||||
get_filename_component(basename ${file} NAME)
|
||||
set(generated_file_path "${hip_compile_output_dir}/${CMAKE_CFG_INTDIR}")
|
||||
set(generated_file_basename "${_target}_generated_${basename}${generated_extension}")
|
||||
|
||||
# Set file names
|
||||
set(generated_file "${generated_file_path}/${generated_file_basename}")
|
||||
set(cmake_dependency_file "${hip_compile_output_dir}/${generated_file_basename}.depend")
|
||||
set(custom_target_script_pregen "${hip_compile_output_dir}/${generated_file_basename}.cmake.pre-gen")
|
||||
set(custom_target_script "${hip_compile_output_dir}/${generated_file_basename}.cmake")
|
||||
|
||||
# Set properties for object files
|
||||
set_source_files_properties("${generated_file}"
|
||||
PROPERTIES
|
||||
EXTERNAL_OBJECT true # This is an object file not to be compiled, but only be linked
|
||||
)
|
||||
|
||||
# Don't add CMAKE_CURRENT_SOURCE_DIR if the path is already an absolute path
|
||||
get_filename_component(file_path "${file}" PATH)
|
||||
if(IS_ABSOLUTE "${file_path}")
|
||||
set(source_file "${file}")
|
||||
else()
|
||||
set(source_file "${CMAKE_CURRENT_SOURCE_DIR}/${file}")
|
||||
endif()
|
||||
|
||||
# Bring in the dependencies
|
||||
HIP_INCLUDE_HIPCC_DEPENDENCIES(${cmake_dependency_file})
|
||||
|
||||
# Configure the build script
|
||||
configure_file("${HIP_run_hipcc}" "${custom_target_script_pregen}" @ONLY)
|
||||
file(GENERATE
|
||||
OUTPUT "${custom_target_script}"
|
||||
INPUT "${custom_target_script_pregen}"
|
||||
)
|
||||
set(main_dep DEPENDS ${source_file})
|
||||
if(CMAKE_GENERATOR MATCHES "Makefiles")
|
||||
set(verbose_output "$(VERBOSE)")
|
||||
elseif(HIP_VERBOSE_BUILD)
|
||||
set(verbose_output ON)
|
||||
else()
|
||||
set(verbose_output OFF)
|
||||
endif()
|
||||
|
||||
# Create up the comment string
|
||||
file(RELATIVE_PATH generated_file_relative_path "${CMAKE_BINARY_DIR}" "${generated_file}")
|
||||
set(hip_build_comment_string "Building HIPCC object ${generated_file_relative_path}")
|
||||
|
||||
# Build the generated file and dependency file
|
||||
add_custom_command(
|
||||
OUTPUT ${generated_file}
|
||||
# These output files depend on the source_file and the contents of cmake_dependency_file
|
||||
${main_dep}
|
||||
DEPENDS ${HIP_HIPCC_DEPEND}
|
||||
DEPENDS ${custom_target_script}
|
||||
# Make sure the output directory exists before trying to write to it.
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory "${generated_file_path}"
|
||||
COMMAND ${CMAKE_COMMAND} ARGS
|
||||
-D verbose:BOOL=${verbose_output}
|
||||
-D build_configuration:STRING=${_hip_build_configuration}
|
||||
-D "generated_file:STRING=${generated_file}"
|
||||
-P "${custom_target_script}"
|
||||
WORKING_DIRECTORY "${hip_compile_output_dir}"
|
||||
COMMENT "${hip_build_comment_string}"
|
||||
)
|
||||
|
||||
# Make sure the build system knows the file is generated
|
||||
set_source_files_properties(${generated_file} PROPERTIES GENERATED TRUE)
|
||||
list(APPEND _hip_generated_files ${generated_file})
|
||||
list(APPEND _hip_source_files ${file})
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
# Set the return parameter
|
||||
set(${_generated_files} ${_hip_generated_files})
|
||||
set(${_source_files} ${_hip_source_files})
|
||||
endmacro()
|
||||
|
||||
###############################################################################
|
||||
# HIP_ADD_EXECUTABLE
|
||||
###############################################################################
|
||||
macro(HIP_ADD_EXECUTABLE hip_target)
|
||||
# Separate the sources from the options
|
||||
HIP_GET_SOURCES_AND_OPTIONS(_sources _cmake_options _hipcc_options _hcc_options _clang_options _nvcc_options ${ARGN})
|
||||
HIP_PREPARE_TARGET_COMMANDS(${hip_target} OBJ _generated_files _source_files ${_sources} HIPCC_OPTIONS ${_hipcc_options} HCC_OPTIONS ${_hcc_options} CLANG_OPTIONS ${_clang_options} NVCC_OPTIONS ${_nvcc_options})
|
||||
if(_source_files)
|
||||
list(REMOVE_ITEM _sources ${_source_files})
|
||||
endif()
|
||||
if("${HIP_COMPILER}" STREQUAL "hcc")
|
||||
if("x${HCC_HOME}" STREQUAL "x")
|
||||
if (DEFINED ENV{ROCM_PATH})
|
||||
set(HCC_HOME "$ENV{ROCM_PATH}/hcc")
|
||||
elseif(DEFINED ENV{HIP_PATH})
|
||||
set(HCC_HOME "$ENV{HIP_PATH}/../hcc")
|
||||
else()
|
||||
set(HCC_HOME "/opt/rocm/hcc")
|
||||
endif()
|
||||
endif()
|
||||
set(CMAKE_HIP_LINK_EXECUTABLE "${HIP_HIPCC_CMAKE_LINKER_HELPER} ${HCC_HOME} <FLAGS> <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> -o <TARGET> <LINK_LIBRARIES>")
|
||||
elseif("${HIP_COMPILER}" STREQUAL "clang")
|
||||
if("x${HIP_CLANG_PATH}" STREQUAL "x")
|
||||
if(DEFINED ENV{HIP_CLANG_PATH})
|
||||
set(HIP_CLANG_PATH $ENV{HIP_CLANG_PATH})
|
||||
elseif(DEFINED ENV{ROCM_PATH})
|
||||
set(HIP_CLANG_PATH "$ENV{ROCM_PATH}/llvm/bin")
|
||||
elseif(DEFINED ENV{HIP_PATH})
|
||||
set(HIP_CLANG_PATH "$ENV{HIP_PATH}/../llvm/bin")
|
||||
else()
|
||||
set(HIP_CLANG_PATH "/opt/rocm/llvm/bin")
|
||||
endif()
|
||||
endif()
|
||||
set(CMAKE_HIP_LINK_EXECUTABLE "${HIP_HIPCC_CMAKE_LINKER_HELPER} ${HIP_CLANG_PATH} ${HIP_CLANG_PARALLEL_BUILD_LINK_OPTIONS} <FLAGS> <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> -o <TARGET> <LINK_LIBRARIES>")
|
||||
else()
|
||||
set(CMAKE_HIP_LINK_EXECUTABLE "${HIP_HIPCC_CMAKE_LINKER_HELPER} <FLAGS> <CMAKE_CXX_LINK_FLAGS> <LINK_FLAGS> <OBJECTS> -o <TARGET> <LINK_LIBRARIES>")
|
||||
endif()
|
||||
if ("${_sources}" STREQUAL "")
|
||||
add_executable(${hip_target} ${_cmake_options} ${_generated_files} "")
|
||||
else()
|
||||
add_executable(${hip_target} ${_cmake_options} ${_generated_files} ${_sources})
|
||||
endif()
|
||||
set_target_properties(${hip_target} PROPERTIES LINKER_LANGUAGE HIP)
|
||||
# Link with host
|
||||
if (HIP_HOST_INTERFACE)
|
||||
# hip rt should be rocclr, compiler should be clang
|
||||
target_link_libraries(${hip_target} ${HIP_HOST_INTERFACE})
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
###############################################################################
|
||||
# HIP_ADD_LIBRARY
|
||||
###############################################################################
|
||||
macro(HIP_ADD_LIBRARY hip_target)
|
||||
# Separate the sources from the options
|
||||
HIP_GET_SOURCES_AND_OPTIONS(_sources _cmake_options _hipcc_options _hcc_options _clang_options _nvcc_options ${ARGN})
|
||||
HIP_PREPARE_TARGET_COMMANDS(${hip_target} OBJ _generated_files _source_files ${_sources} ${_cmake_options} HIPCC_OPTIONS ${_hipcc_options} HCC_OPTIONS ${_hcc_options} CLANG_OPTIONS ${_clang_options} NVCC_OPTIONS ${_nvcc_options})
|
||||
if(_source_files)
|
||||
list(REMOVE_ITEM _sources ${_source_files})
|
||||
endif()
|
||||
if ("${_sources}" STREQUAL "")
|
||||
add_library(${hip_target} ${_cmake_options} ${_generated_files} "")
|
||||
else()
|
||||
add_library(${hip_target} ${_cmake_options} ${_generated_files} ${_sources})
|
||||
endif()
|
||||
set_target_properties(${hip_target} PROPERTIES LINKER_LANGUAGE ${HIP_C_OR_CXX})
|
||||
# Link with host
|
||||
if (HIP_HOST_INTERFACE)
|
||||
# hip rt should be rocclr, compiler should be clang
|
||||
target_link_libraries(${hip_target} ${HIP_HOST_INTERFACE})
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
# vim: ts=4:sw=4:expandtab:smartindent
|
||||
@@ -1,182 +0,0 @@
|
||||
###############################################################################
|
||||
# Runs commands using HIPCC
|
||||
###############################################################################
|
||||
|
||||
###############################################################################
|
||||
# This file runs the hipcc commands to produce the desired output file
|
||||
# along with the dependency file needed by CMake to compute dependencies.
|
||||
#
|
||||
# Input variables:
|
||||
#
|
||||
# verbose:BOOL=<> OFF: Be as quiet as possible (default)
|
||||
# ON : Describe each step
|
||||
# build_configuration:STRING=<> Build configuration. Defaults to Debug.
|
||||
# generated_file:STRING=<> File to generate. Mandatory argument.
|
||||
|
||||
if(NOT build_configuration)
|
||||
set(build_configuration Debug)
|
||||
endif()
|
||||
if(NOT generated_file)
|
||||
message(FATAL_ERROR "You must specify generated_file on the command line")
|
||||
endif()
|
||||
|
||||
# Set these up as variables to make reading the generated file easier
|
||||
set(HIP_HIPCC_EXECUTABLE "@HIP_HIPCC_EXECUTABLE@") # path
|
||||
set(HIP_HIPCONFIG_EXECUTABLE "@HIP_HIPCONFIG_EXECUTABLE@") #path
|
||||
set(HIP_HOST_COMPILER "@HIP_HOST_COMPILER@") # path
|
||||
set(CMAKE_COMMAND "@CMAKE_COMMAND@") # path
|
||||
set(HIP_run_make2cmake "@HIP_run_make2cmake@") # path
|
||||
set(HCC_HOME "@HCC_HOME@") #path
|
||||
set(HIP_CLANG_PATH "@HIP_CLANG_PATH@") #path
|
||||
set(HIP_CLANG_PARALLEL_BUILD_COMPILE_OPTIONS "@HIP_CLANG_PARALLEL_BUILD_COMPILE_OPTIONS@")
|
||||
|
||||
@HIP_HOST_FLAGS@
|
||||
@_HIP_HIPCC_FLAGS@
|
||||
@_HIP_HCC_FLAGS@
|
||||
@_HIP_CLANG_FLAGS@
|
||||
@_HIP_NVCC_FLAGS@
|
||||
#Needed to bring the HIP_HIPCC_INCLUDE_ARGS variable in scope
|
||||
set(HIP_HIPCC_INCLUDE_ARGS @HIP_HIPCC_INCLUDE_ARGS@) # list
|
||||
|
||||
set(cmake_dependency_file "@cmake_dependency_file@") # path
|
||||
set(source_file "@source_file@") # path
|
||||
set(host_flag "@host_flag@") # bool
|
||||
|
||||
# Determine compiler and compiler flags
|
||||
execute_process(COMMAND ${HIP_HIPCONFIG_EXECUTABLE} --platform OUTPUT_VARIABLE HIP_PLATFORM OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
execute_process(COMMAND ${HIP_HIPCONFIG_EXECUTABLE} --compiler OUTPUT_VARIABLE HIP_COMPILER OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
execute_process(COMMAND ${HIP_HIPCONFIG_EXECUTABLE} --runtime OUTPUT_VARIABLE HIP_RUNTIME OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
if(NOT host_flag)
|
||||
set(__CC ${HIP_HIPCC_EXECUTABLE})
|
||||
if("${HIP_PLATFORM}" STREQUAL "amd")
|
||||
if("${HIP_COMPILER}" STREQUAL "hcc")
|
||||
if(NOT "x${HCC_HOME}" STREQUAL "x")
|
||||
set(ENV{HCC_HOME} ${HCC_HOME})
|
||||
endif()
|
||||
set(__CC_FLAGS ${HIP_HIPCC_FLAGS} ${HIP_HCC_FLAGS} ${HIP_HIPCC_FLAGS_${build_configuration}} ${HIP_HCC_FLAGS_${build_configuration}})
|
||||
elseif("${HIP_COMPILER}" STREQUAL "clang")
|
||||
if(NOT "x${HIP_CLANG_PATH}" STREQUAL "x")
|
||||
set(ENV{HIP_CLANG_PATH} ${HIP_CLANG_PATH})
|
||||
endif()
|
||||
# Temporarily include HIP_HCC_FLAGS for HIP-Clang for PyTorch builds
|
||||
set(__CC_FLAGS ${HIP_CLANG_PARALLEL_BUILD_COMPILE_OPTIONS} ${HIP_HIPCC_FLAGS} ${HIP_HCC_FLAGS} ${HIP_CLANG_FLAGS} ${HIP_HIPCC_FLAGS_${build_configuration}} ${HIP_HCC_FLAGS_${build_configuration}} ${HIP_CLANG_FLAGS_${build_configuration}})
|
||||
endif()
|
||||
else()
|
||||
set(__CC_FLAGS ${HIP_HIPCC_FLAGS} ${HIP_NVCC_FLAGS} ${HIP_HIPCC_FLAGS_${build_configuration}} ${HIP_NVCC_FLAGS_${build_configuration}})
|
||||
endif()
|
||||
else()
|
||||
set(__CC ${HIP_HOST_COMPILER})
|
||||
set(__CC_FLAGS ${CMAKE_HOST_FLAGS} ${CMAKE_HOST_FLAGS_${build_configuration}})
|
||||
endif()
|
||||
set(__CC_INCLUDES ${HIP_HIPCC_INCLUDE_ARGS})
|
||||
|
||||
# hip_execute_process - Executes a command with optional command echo and status message.
|
||||
# status - Status message to print if verbose is true
|
||||
# command - COMMAND argument from the usual execute_process argument structure
|
||||
# ARGN - Remaining arguments are the command with arguments
|
||||
# HIP_result - Return value from running the command
|
||||
macro(hip_execute_process status command)
|
||||
set(_command ${command})
|
||||
if(NOT "x${_command}" STREQUAL "xCOMMAND")
|
||||
message(FATAL_ERROR "Malformed call to hip_execute_process. Missing COMMAND as second argument. (command = ${command})")
|
||||
endif()
|
||||
if(verbose)
|
||||
execute_process(COMMAND "${CMAKE_COMMAND}" -E echo -- ${status})
|
||||
# Build command string to print
|
||||
set(hip_execute_process_string)
|
||||
foreach(arg ${ARGN})
|
||||
# Escape quotes if any
|
||||
string(REPLACE "\"" "\\\"" arg ${arg})
|
||||
# Surround args with spaces with quotes
|
||||
if(arg MATCHES " ")
|
||||
list(APPEND hip_execute_process_string "\"${arg}\"")
|
||||
else()
|
||||
list(APPEND hip_execute_process_string ${arg})
|
||||
endif()
|
||||
endforeach()
|
||||
# Echo the command
|
||||
execute_process(COMMAND ${CMAKE_COMMAND} -E echo ${hip_execute_process_string})
|
||||
endif()
|
||||
# Run the command
|
||||
execute_process(COMMAND ${ARGN} RESULT_VARIABLE HIP_result)
|
||||
endmacro()
|
||||
|
||||
# Delete the target file
|
||||
hip_execute_process(
|
||||
"Removing ${generated_file}"
|
||||
COMMAND "${CMAKE_COMMAND}" -E remove "${generated_file}"
|
||||
)
|
||||
|
||||
# Generate the dependency file
|
||||
hip_execute_process(
|
||||
"Generating dependency file: ${cmake_dependency_file}.pre"
|
||||
COMMAND "${__CC}"
|
||||
-M
|
||||
"${source_file}"
|
||||
-o "${cmake_dependency_file}.pre"
|
||||
${__CC_FLAGS}
|
||||
${__CC_INCLUDES}
|
||||
)
|
||||
|
||||
if(HIP_result)
|
||||
message(FATAL_ERROR "Error generating ${generated_file}")
|
||||
endif()
|
||||
|
||||
# Generate the cmake readable dependency file to a temp file
|
||||
hip_execute_process(
|
||||
"Generating temporary cmake readable file: ${cmake_dependency_file}.tmp"
|
||||
COMMAND "${CMAKE_COMMAND}"
|
||||
-D "input_file:FILEPATH=${cmake_dependency_file}.pre"
|
||||
-D "output_file:FILEPATH=${cmake_dependency_file}.tmp"
|
||||
-D "verbose=${verbose}"
|
||||
-P "${HIP_run_make2cmake}"
|
||||
)
|
||||
|
||||
if(HIP_result)
|
||||
message(FATAL_ERROR "Error generating ${generated_file}")
|
||||
endif()
|
||||
|
||||
# Copy the file if it is different
|
||||
hip_execute_process(
|
||||
"Copy if different ${cmake_dependency_file}.tmp to ${cmake_dependency_file}"
|
||||
COMMAND "${CMAKE_COMMAND}" -E copy_if_different "${cmake_dependency_file}.tmp" "${cmake_dependency_file}"
|
||||
)
|
||||
|
||||
if(HIP_result)
|
||||
message(FATAL_ERROR "Error generating ${generated_file}")
|
||||
endif()
|
||||
|
||||
# Delete the temporary file
|
||||
hip_execute_process(
|
||||
"Removing ${cmake_dependency_file}.tmp and ${cmake_dependency_file}.pre"
|
||||
COMMAND "${CMAKE_COMMAND}" -E remove "${cmake_dependency_file}.tmp" "${cmake_dependency_file}.pre"
|
||||
)
|
||||
|
||||
if(HIP_result)
|
||||
message(FATAL_ERROR "Error generating ${generated_file}")
|
||||
endif()
|
||||
|
||||
# Generate the output file
|
||||
hip_execute_process(
|
||||
"Generating ${generated_file}"
|
||||
COMMAND "${__CC}"
|
||||
-c
|
||||
"${source_file}"
|
||||
-o "${generated_file}"
|
||||
${__CC_FLAGS}
|
||||
${__CC_INCLUDES}
|
||||
)
|
||||
|
||||
if(HIP_result)
|
||||
# Make sure that we delete the output file
|
||||
hip_execute_process(
|
||||
"Removing ${generated_file}"
|
||||
COMMAND "${CMAKE_COMMAND}" -E remove "${generated_file}"
|
||||
)
|
||||
message(FATAL_ERROR "Error generating file ${generated_file}")
|
||||
else()
|
||||
if(verbose)
|
||||
message("Generated ${generated_file} successfully.")
|
||||
endif()
|
||||
endif()
|
||||
# vim: ts=4:sw=4:expandtab:smartindent
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user