Compare commits

..
200 changed files with 7851 additions and 12870 deletions
-7
View File
@@ -19,15 +19,9 @@ CMakeFiles/
# Clangd server cache
*.cache*
#vscode settings
/.vscode/
# Backup files
*~
# clangd index
/.cache/
# Default install location
/mfem/
@@ -85,7 +79,6 @@ examples/sol_u.*
examples/sol_p.*
examples/sol_r.*
examples/sol_i.*
examples/sol_z.*
examples/ex6p-checkpoint.*
examples/order.*
examples/ex9.mesh
+70 -529
View File
@@ -9,550 +9,91 @@
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# DESCRIPTION:
###############################################################################
# General GitLab pipelines configurations for supercomputers and Linux clusters
# at Lawrence Livermore National Laboratory (LLNL).
# This entire pipeline is LLNL-specific
#
# Important note: This file is a template provided by llnl/radiuss-shared-ci.
# Remains to set variable values, change the reference to the radiuss-shared-ci
# repo, opt-in and out optional features. The project can then extend it with
# additional stages.
#
# In addition, each project should copy over and complete:
# - .gitlab/custom-jobs-and-variables.yml
# - .gitlab/subscribed-pipelines.yml
#
# The jobs should be specified in a file local to the project,
# - .gitlab/jobs/${CI_MACHINE}.yml
# or generated (see LLNL/Umpire for an example).
###############################################################################
# MAP OF GITLAB CI
#######################
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
# File dependencies: direct, through jobs, through variables
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
# .gitlab-ci.yml
# ├── .build-and-test [job]
# │ ├── .gitlab/custom-jobs-and-variables.yml
# │ │ ├── .custom_job [job]
# │ │ ├── .reproducer_vars [job]
# │ │ ├── .report_job_success [job]
# │ │ │ └── .gitlab/scripts/report_build_and_test [script]
# │ │ │ ├── .gitlab/scripts/safe_create_rundir [script]
# │ │ │ └── .gitlab/scripts/git_try_to_push [script]
# │ │ ├── .report_job_failure [job]
# │ │ │ └── .gitlab/scripts/report_build_and_test [script]
# │ │ │ ├── .gitlab/scripts/safe_create_rundir [script]
# │ │ │ └── .gitlab/scripts/git_try_to_push [script]
# │ │ └── JOB_CMD [var]
# │ │ └── tests/gitlab/build_and_test [script]
# │ │ └── tests/gitlab/get_mfem_uberenv [script]
# │ ├── <radiuss-shared-ci>/pipelines/matrix.yml [conditional]
# │ │ ├── .on_matrix [job]
# │ │ ├── .matrix_reproducer_init [job]
# │ │ ├── .matrix_reproducer_vars [job]
# │ │ ├── .matrix_reproducer_job [job]
# │ │ ├── .matrix_job_command [job]
# │ │ └── .job_on_matrix [job]
# │ ├── <radiuss-shared-ci>/pipelines/dane.yml [conditional]
# │ │ ├── .on_dane [job]
# │ │ ├── .dane_reproducer_init [job]
# │ │ ├── .dane_reproducer_vars [job]
# │ │ ├── .dane_reproducer_job [job]
# │ │ ├── .dane_job_command [job]
# │ │ ├── .job_on_dane [job]
# │ │ ├── allocate_resources [job]
# │ │ └── release_resources [job]
# │ ├── <radiuss-shared-ci>/pipelines/tioga.yml [conditional]
# │ │ ├── .on_tioga [job]
# │ │ ├── .tioga_reproducer_init [job]
# │ │ ├── .tioga_reproducer_vars [job]
# │ │ ├── .tioga_reproducer_job [job]
# │ │ ├── .tioga_job_command [job]
# │ │ ├── .job_on_tioga [job]
# │ │ ├── allocate_resources [job]
# │ │ └── release_resources [job]
# │ ├── <artifact>/matrix-jobs.yml [conditional, from 'generate-job-lists']
# │ │ ├── .gitlab/jobs/matrix.yml
# │ │ │ ├── .matrix_reproducer_vars [job]
# │ │ │ ├── setup [job]
# │ │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
# │ │ │ ├── opt_mpi_cuda_gcc [job]
# │ │ │ └── opt_mpi_cuda_hypre_cuda_gcc [job]
# │ │ └── .gitlab/jobs/matrix-reports.yml [used conditionally]
# │ │ ├── report_job_success
# │ │ └── report_job_failure
# │ ├── <artifact>/dane-jobs.yml [conditional, from 'generate-job-lists']
# │ │ ├── .gitlab/jobs/dane.yml
# │ │ │ ├── .dane_reproducer_vars [job]
# │ │ │ ├── setup [job]
# │ │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
# │ │ │ ├── debug_ser_gcc_10 [job]
# │ │ │ ├── debug_par_gcc_10 [job]
# │ │ │ ├── opt_ser_gcc_10 [job]
# │ │ │ ├── opt_par_gcc_10 [job]
# │ │ │ ├── opt_par_gcc_10_sundials [job]
# │ │ │ ├── opt_par_gcc_10_petsc [job]
# │ │ │ └── opt_par_gcc_10_pumi [job]
# │ │ └── .gitlab/jobs/dane-reports.yml [used conditionally]
# │ │ ├── report_job_success
# │ │ └── report_job_failure
# │ └── <artifact>/tioga-jobs.yml [conditional, from 'generate-job-lists']
# │ ├── .gitlab/jobs/tioga.yml
# │ │ ├── .tioga_reproducer_vars [job]
# │ │ ├── setup [job]
# │ │ │ └── ./tests/gitlab/build_and_test_setup [script]
# │ │ └── cce_16_0_1 [job]
# │ └── .gitlab/jobs/tioga-reports.yml [used conditionally]
# │ ├── report_job_success
# │ └── report_job_failure
# └── .gitlab/subscribed-pipelines.yml
# ├── .machine-check [job]
# ├── generate-job-lists [job]
# ├── dane-up-check [job]
# ├── dane-build-and-test [job]
# ├── dane-baseline [job]
# │ └── .gitlab/dane-baseline.yml
# │ ├── .on_dane [job]
# │ ├── baselinecheck_mfem_intel_dane [job]
# │ │ └── .gitlab/scripts/baseline [script]
# │ ├── cleanup [job]
# │ ├── report_baseline [job]
# │ │ ├── .gitlab/scripts/safe_create_rundir [script]
# │ │ └── .gitlab/scripts/git_try_to_push [script]
# │ ├── baselinepublish_mfem_dane [job]
# │ │ └── .gitlab/scripts/rebaseline [script]
# │ ├── .gitlab/custom-jobs-and-variables.yml
# │ │ └── <same as above: see .gitlab-ci.yml/.build-and-test>
# │ └── .gitlab/configs/setup-baseline.yml
# │ └── setup_baseline [job]
# ├── tioga-up-check [job]
# ├── tioga-build-and-test [job]
# ├── matrix-up-check [job]
# └── matrix-build-and-test [job]
#
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
# File tree hierarchy with file contents highlights
#~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
# In addition to the files in the MFEM repo, the Gitlab CI uses files from the
# radiuss/radiuss-shared-ci project, see below, after the <mfem-root> tree.
#
# <mfem root>
# ├── .gitlab-ci.yml [this file]
# │ ├── <jobs>
# │ │ └── .build-and-test
# │ ├── <included files>
# │ │ ├── .gitlab/subscribed-pipelines.yml
# │ │ ├── .gitlab/custom-jobs-and-variables.yml [by ".build-and-test"]
# │ │ ├── <artifact> [by ".build-and-test"]
# │ │ │ ├── artifact: '${CI_MACHINE}-jobs.yml'
# │ │ │ └── job: 'generate-job-lists'
# │ │ └── <external> [by ".build-and-test"]
# │ │ ├── project: 'radiuss/radiuss-shared-ci'
# │ │ ├── ref: 'v2025.09.1'
# │ │ └── file: 'pipelines/${CI_MACHINE}.yml'
# │ └── <defined variables>
# │ ├── CUSTOM_CI_BUILDS_DIR
# │ ├── USER_CI_TOP_DIR
# │ ├── SHARED_REPOS_DIR
# │ ├── AUTOTEST_ROOT
# │ ├── MFEM_DATA_DIR
# │ ├── AUTOTEST
# │ ├── AUTOTEST_COMMIT
# │ ├── REBASELINE
# │ ├── GITHUB_PROJECT_NAME
# │ └── GITHUB_PROJECT_ORG
# ├── .gitlab
# │ ├── configs
# │ │ └── setup-baseline.yml
# │ │ ├── <jobs>
# │ │ │ └── setup_baseline
# │ │ └── <used variables>
# │ │ ├── MACHINE_NAME
# │ │ ├── REBASELINE
# │ │ ├── AUTOTEST
# │ │ ├── AUTOTEST_COMMIT
# │ │ ├── BUILD_ROOT
# │ │ ├── TPLS_REPO
# │ │ ├── TESTS_REPO
# │ │ ├── AUTOTEST_ROOT
# │ │ └── AUTOTEST_REPO
# │ ├── jobs
# │ │ ├── matrix-reports.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── report_job_success
# │ │ │ │ └── report_job_failure
# │ │ │ └── <used jobs>
# │ │ │ ├── .on_matrix
# │ │ │ ├── .report_job_success
# │ │ │ └── .report_job_failure
# │ │ ├── matrix.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── .matrix_reproducer_vars
# │ │ │ │ ├── setup
# │ │ │ │ ├── opt_mpi_cuda_gcc
# │ │ │ │ └── opt_mpi_cuda_hypre_cuda_gcc
# │ │ │ ├── <used jobs>
# │ │ │ │ ├── .reproducer_vars
# │ │ │ │ ├── .on_matrix
# │ │ │ │ └── .job_on_matrix
# │ │ │ ├── <included and used files>
# │ │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
# │ │ │ └── <defined variables>
# │ │ │ └── SPEC
# │ │ ├── dane-reports.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── report_job_success
# │ │ │ │ └── report_job_failure
# │ │ │ └── <used jobs>
# │ │ │ ├── .on_dane
# │ │ │ ├── .report_job_success
# │ │ │ └── .report_job_failure
# │ │ ├── dane.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── .dane_reproducer_vars
# │ │ │ │ ├── setup
# │ │ │ │ ├── debug_ser_gcc_10
# │ │ │ │ ├── debug_par_gcc_10
# │ │ │ │ ├── opt_ser_gcc_10
# │ │ │ │ ├── opt_par_gcc_10
# │ │ │ │ ├── opt_par_gcc_10_sundials
# │ │ │ │ ├── opt_par_gcc_10_petsc
# │ │ │ │ └── opt_par_gcc_10_pumi
# │ │ │ ├── <used jobs>
# │ │ │ │ ├── .reproducer_vars
# │ │ │ │ ├── .on_dane
# │ │ │ │ └── .job_on_dane
# │ │ │ ├── <included and used files>
# │ │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
# │ │ │ └── <defined variables>
# │ │ │ ├── SPEC
# │ │ │ └── THREADS
# │ │ ├── tioga-reports.yml
# │ │ │ ├── <jobs>
# │ │ │ │ ├── report_job_success
# │ │ │ │ └── report_job_failure
# │ │ │ └── <used jobs>
# │ │ │ ├── .on_tioga
# │ │ │ ├── .report_job_success
# │ │ │ └── .report_job_failure
# │ │ └── tioga.yml
# │ │ ├── <jobs>
# │ │ │ ├── .tioga_reproducer_vars
# │ │ │ ├── setup
# │ │ │ └── opt_mpi_rocm_hypre_rocm
# │ │ ├── <used jobs>
# │ │ │ ├── .reproducer_vars
# │ │ │ ├── .on_tioga
# │ │ │ └── .job_on_tioga
# │ │ ├── <included and used files>
# │ │ │ └── tests/gitlab/build_and_test_setup [by "setup"]
# │ │ └── <defined variables>
# │ │ ├── SPEC
# │ │ └── THREADS
# │ ├── scripts
# │ │ ├── baseline
# │ │ │ └── <used variables>
# │ │ │ ├── BASELINE_TEST
# │ │ │ ├── SYS_TYPE
# │ │ │ ├── MACHINE_NAME
# │ │ │ ├── CI_PROJECT_DIR
# │ │ │ ├── ARTIFACTS_DIR
# │ │ │ ├── BUILD_ROOT
# │ │ │ └── TPLS_DIR
# │ │ ├── git_try_to_push
# │ │ ├── rebaseline
# │ │ │ └── <used variables>
# │ │ │ ├── CI_PROJECT_DIR
# │ │ │ ├── ARTIFACTS_DIR
# │ │ │ ├── SYS_TYPE
# │ │ │ ├── BUILD_ROOT
# │ │ │ ├── MACHINE_NAME
# │ │ │ └── CI_PIPELINE_ID
# │ │ ├── report_build_and_test
# │ │ │ ├── <used files>
# │ │ │ │ ├── .gitlab/scripts/safe_create_rundir
# │ │ │ │ └── .gitlab/scripts/git_try_to_push
# │ │ │ └── <used variables>
# │ │ │ ├── AUTOTEST_ROOT
# │ │ │ ├── CI_COMMIT_REF_SLUG
# │ │ │ ├── CI_PROJECT_DIR
# │ │ │ ├── CI_PIPELINE_URL
# │ │ │ ├── AUTOTEST_COMMIT
# │ │ │ └── CI_MACHINE
# │ │ └── safe_create_rundir
# │ ├── custom-jobs-and-variables.yml
# │ │ ├── <jobs>
# │ │ │ ├── .custom_job
# │ │ │ ├── .reproducer_vars
# │ │ │ ├── .report_job_success
# │ │ │ └── .report_job_failure
# │ │ ├── <used files>
# │ │ │ ├── tests/gitlab/build_and_test [in JOB_CMD]
# │ │ │ └── .gitlab/scripts/report_build_and_test [by .report_job_*]
# │ │ ├── <defined variables>
# │ │ │ ├── JOB_CMD
# │ │ │ ├── BUILD_ROOT
# │ │ │ ├── ALLOC_NAME
# │ │ │ ├── TPLS_REPO
# │ │ │ ├── TESTS_REPO
# │ │ │ ├── AUTOTEST_REPO
# │ │ │ ├── MFEM_DATA_REPO
# │ │ │ ├── ARTIFACTS_DIR: artifacts
# │ │ │ ├── SLURM_OVERLAP: 1
# │ │ │ ├── DANE_SHARED_ALLOC
# │ │ │ ├── DANE_JOB_ALLOC
# │ │ │ ├── TIOGA_SHARED_ALLOC
# │ │ │ ├── TIOGA_JOB_ALLOC
# │ │ │ └── MATRIX_JOB_ALLOC
# │ │ └── <used variables>
# │ │ ├── SPEC
# │ │ ├── BUILD_ROOT
# │ │ └── ...
# │ ├── dane-baseline.yml
# │ │ ├── <jobs>
# │ │ │ ├── .on_dane
# │ │ │ ├── baselinecheck_mfem_intel_dane
# │ │ │ ├── cleanup
# │ │ │ ├── report_baseline
# │ │ │ └── baselinepublish_mfem_dane
# │ │ ├── <included and used files>
# │ │ │ ├── .gitlab/custom-jobs-and-variables.yml
# │ │ │ ├── .gitlab/configs/setup-baseline.yml
# │ │ │ ├── .gitlab/scripts/rebaseline
# │ │ │ ├── .gitlab/scripts/baseline
# │ │ │ └── .gitlab/scripts/git_try_to_push
# │ │ ├── <defined variables>
# │ │ │ ├── BASELINE_TEST: baseline
# │ │ │ ├── MACHINE_NAME: dane
# │ │ │ ├── TPLS_DIR
# │ │ │ └── export MFEM_TEST_NP
# │ │ └── <used variables>
# │ │ ├── ON_DANE
# │ │ ├── AUTOTEST [defined by .gitlab-ci.yml]
# │ │ ├── BUILD_ROOT [defined by custom-jobs-and-variables.yml]
# │ │ ├── TPLS_DIR [defined by this file]
# │ │ ├── ARTIFACTS_DIR [defined by custom-jobs-and-variables.yml]
# │ │ ├── MACHINE_NAME [defined by this file]
# │ │ ├── AUTOTEST_COMMIT [defined by .gitlab-ci.yml]
# │ │ ├── AUTOTEST_ROOT [defined by .gitlab-ci.yml]
# │ │ ├── BASELINE_TEST [defined by this file]
# │ │ └── REBASELINE [defined by .gitlab-ci.yml]
# │ └── subscribed-pipelines.yml
# │ ├── <jobs>
# │ │ ├── .machine-check
# │ │ ├── generate-job-lists
# │ │ ├── dane-up-check
# │ │ ├── dane-build-and-test
# │ │ ├── dane-baseline
# │ │ ├── tioga-up-check
# │ │ ├── tioga-build-and-test
# │ │ ├── matrix-up-check
# │ │ └── matrix-build-and-test
# │ ├── <used jobs>
# │ │ └── .build-and-test [from ".gitlab-ci.yml"]
# │ ├── <included files>
# │ │ └── .gitlab/dane-baseline.yml [by "dane-baseline"]
# │ └── <used variables>
# │ ├── GITHUB_PROJECT_ORG
# │ ├── GITHUB_PROJECT_NAME
# │ ├── AUTOTEST
# │ ├── AUTOTEST_COMMIT
# │ └── REBASELINE
# └── tests
# ├── gitlab
# │ ├── build_and_test
# │ │ ├── <builds and tests a given MFEM spec with uberenv>
# │ │ ├── <used files>
# │ │ │ ├── tests/uberenv/uberenv.py [deps mode, cloned]
# │ │ │ └── tests/gitlab/get_mfem_uberenv [deps mode]
# │ │ └── <used variables>
# │ │ ├── SYS_TYPE
# │ │ ├── THREADS [num. parallel jobs to build MFEM]
# │ │ ├── MODULE_LIST [modules to load]
# │ │ ├── CI_JOB_ID
# │ │ ├── USE_DEV_SHM
# │ │ ├── SPACK_DEBUG
# │ │ ├── DEBUG_MODE
# │ │ ├── REGISTRY_TOKEN
# │ │ ├── CI_REGISTRY_USER (defined by Gitlab)
# │ │ ├── USER
# │ │ ├── CI_REGISTRY_IMAGE (defined by Gitlab)
# │ │ └── CI_JOB_TOKEN (defined by Gitlab)
# │ ├── build_and_test_setup
# │ │ ├── <updates MFEM_DATA_REPO and AUTOTEST_REPO using locks>
# │ │ └── <used variables>
# │ │ ├── MFEM_DATA_REPO
# │ │ ├── SHARED_REPOS_DIR
# │ │ ├── AUTOTEST_REPO
# │ │ └── AUTOTEST_ROOT
# │ └── get_mfem_uberenv
# │ ├── <github.com/mfem/mfem-uberenv.git -> tests/uberenv>
# │ └── <defines the uberenv hash to use>
# └── uberenv [cloned by tests/gitlab/get_mfem_uberenv]
# └── uberenv.py
#
# <root of radiuss/radiuss-shared-ci, ref: 'v2025.09.1'>
# └── pipelines
# ├── matrix.yml
# │ ├── <jobs>
# │ │ ├── .on_matrix
# │ │ ├── .matrix_reproducer_init
# │ │ ├── .matrix_reproducer_vars
# │ │ ├── .matrix_reproducer_job
# │ │ ├── .matrix_job_command
# │ │ └── .job_on_matrix
# │ ├── <used jobs>
# │ │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
# │ └── <used variables>
# │ ├── ON_MATRIX
# │ ├── ADVANCED_JOB
# │ ├── ALL_TARGETS
# │ ├── SYS_TYPE
# │ ├── LLNL_SERVICE_USER
# │ ├── USER
# │ ├── GITHUB_PROJECT_NAME
# │ ├── GITHUB_PROJECT_ORG
# │ ├── MATRIX_JOB_ALLOC
# │ └── JOB_CMD
# ├── dane.yml
# │ ├── <jobs>
# │ │ ├── .on_dane
# │ │ ├── .dane_reproducer_init
# │ │ ├── .dane_reproducer_vars
# │ │ ├── .dane_reproducer_job
# │ │ ├── .dane_job_command
# │ │ ├── .job_on_dane
# │ │ ├── allocate_resources
# │ │ └── release_resources
# │ ├── <used jobs>
# │ │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
# │ ├── <defined variables>
# │ │ └── export JOBID
# │ └── <used variables>
# │ ├── ON_DANE
# │ ├── ADVANCED_JOB
# │ ├── ALL_TARGETS
# │ ├── SYS_TYPE
# │ ├── LLNL_SERVICE_USER
# │ ├── USER
# │ ├── GITHUB_PROJECT_NAME
# │ ├── GITHUB_PROJECT_ORG
# │ ├── DANE_JOB_ALLOC
# │ ├── JOB_CMD
# │ ├── JOBID
# │ ├── ALLOC_NAME
# │ └── DANE_SHARED_ALLOC
# └── tioga.yml
# ├── <jobs>
# │ ├── .on_tioga
# │ ├── .tioga_reproducer_init
# │ ├── .tioga_reproducer_vars
# │ ├── .tioga_reproducer_job
# │ ├── .tioga_job_command
# │ ├── .job_on_tioga
# │ ├── allocate_resources
# │ └── release_resources
# ├── <used jobs>
# │ └── .custom_job [from .gitlab/custom-jobs-and-variables.yml]
# ├── <defined variables>
# │ └── PROXY
# └── <used variables>
# ├── ON_TIOGA
# ├── ADVANCED_JOB
# ├── ALL_TARGETS
# ├── SYS_TYPE
# ├── LLNL_SERVICE_USER
# ├── USER
# ├── GITHUB_PROJECT_NAME
# ├── GITHUB_PROJECT_ORG
# ├── TIOGA_JOB_ALLOC
# ├── JOB_CMD
# ├── PROXY
# ├── ALLOC_NAME
# └── TIOGA_SHARED_ALLOC
# at Lawrence Livermore National Laboratory (LLNL). This entire pipeline is
# LLNL-specific!
include:
- project: 'lc-templates/id_tokens'
file: 'id_tokens.yml'
# The pipeline is divided into stages. Usually, jobs in a given stage wait for
# the preceding stages to complete before to start. However, we sometimes use
# the "needs" keyword and express the DAG of jobs for more efficiency.
# - We use setup and setup_baseline phases to download content outside of mfem
# directory.
# - Allocate/Release is where Dane resource are allocated/released once for all.
# - Build and Test is where we build and MFEM for multiple toolchains.
# - Baseline_checks gathers baseline-type test suites execution
# - Baseline_publish, only available on master, allows to update baseline
# results
stages:
- sub-pipelines
###############################################################################
# We define the following GitLab pipeline variables:
variables:
##### LC GITLAB CONFIGURATION
CUSTOM_CI_BUILDS_DIR: "/usr/workspace/mfem/gitlab-runner"
##### PROJECT VARIABLES
USER_CI_TOP_DIR: "${CUSTOM_CI_BUILDS_DIR}/${GITLAB_USER_LOGIN}"
SHARED_REPOS_DIR: "${USER_CI_TOP_DIR}/repos"
AUTOTEST_ROOT: "${SHARED_REPOS_DIR}"
# MFEM_DATA_DIR is setup in '.gitlab/configs/setup-build-and-test.yml' and
# used in '.gitlab/configs/<machine>-config.yml':
MFEM_DATA_DIR: "${SHARED_REPOS_DIR}/mfem-data"
# AUTOTEST: enable (ON/YES) or disable (any other value) test reporting. See
# also AUTOTEST_COMMIT.
AUTOTEST: "OFF"
# AUTOTEST_COMMIT: used only when AUTOTEST is set to ON/YES.
# * If AUTOTEST_COMMIT is set to ON/YES, reporting jobs will commit their
# files to the MFEM/autotest repo.
# * If AUTOTEST_COMMIT is NOT set to ON/YES, reporting jobs will NOT commit
# their files to the MFEM/autotest repo. Instead they will just show the
# contents of the report files and remove them.
AUTOTEST_COMMIT: "ON"
# REBASELINE:
# Defines the default choice for updating the saved baseline results. By default
# the baseline can only be updated from the master branch. This variable offers
# the option to manually ask for rebaselining from another branch if necessary.
REBASELINE: "OFF"
REBASELINE: "NO"
AUTOTEST: "NO"
# AUTOTEST_COMMIT: used only when AUTOTEST is set to YES.
# * If AUTOTEST_COMMIT is NOT set to NO, reporting jobs will commit their
# files to the MFEM/autotest repo.
# * If AUTOTEST_COMMIT is set to NO, reporting jobs will NOT commit their
# files to the MFEM/autotest repo. Instead they will just show the contents
# of the report files and remove them.
AUTOTEST_COMMIT: "YES"
##### SHARED_CI CONFIGURATION
# Required information about GitHub repository
GITHUB_PROJECT_NAME: "mfem"
GITHUB_PROJECT_ORG: "MFEM"
# Override the pattern describing branches that will skip the "draft PR filter
# test". Add protected branches here. See default value in
# preliminary-ignore-draft-pr.yml.
# ALWAYS_RUN_PATTERN: ""
###############################################################################
##### High level stages
# We organize the test-pipelines stage with sub-pipelines. Each sub-pipeline
# corresponds to a test batch on a given machine.
stages:
- prerequisites
- test-pipelines
###############################################################################
# Template for jobs triggering a build-and-test sub-pipeline:
.build-and-test:
stage: test-pipelines
# Trigger subpipelines:
dane-build-and-test:
stage: sub-pipelines
variables:
# Explicitly pass down values that are not always propagated to child
# pipelines, e.g. when a variable is set in the "Settings -> CI" web
# interface (project variables).
# Note: in some cases, this does not work as expected, e.g. when the
# variable is not re-defined in the web interface; in such cases, the child
# pipeline gets a definition like '${AUTOTEST}', i.e. it behaves as if
# AUTOTEST is undefined, even though there is a default value in
# .gitlab-ci.yml.
# Explicitly pass down values that we want to be able to set when triggering
# pipelines manually or using scheduling
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include:
- local: '.gitlab/custom-jobs-and-variables.yml'
- project: 'radiuss/radiuss-shared-ci'
ref: 'v2025.09.1'
file: 'pipelines/${CI_MACHINE}.yml'
- artifact: '${CI_MACHINE}-jobs.yml'
job: 'generate-job-lists'
include: .gitlab/dane-build-and-test.yml
strategy: depend
forward:
pipeline_variables: true
###############################################################################
include:
# Sets ID tokens for every job using `default:`
- project: 'lc-templates/id_tokens'
file: 'id_tokens.yml'
# [Optional] checks preliminary to running the actual CI test
#- project: 'radiuss/radiuss-shared-ci'
# ref: 'v2025.09.1'
# file: 'preliminary-ignore-draft-pr.yml'
# pipelines subscribed by the project
- local: '.gitlab/subscribed-pipelines.yml'
dane-baseline:
stage: sub-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
# pipelines manually or using scheduling
REBASELINE: "${REBASELINE}"
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/dane-baseline.yml
strategy: depend
lassen-build-and-test:
stage: sub-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
# pipelines manually or using scheduling
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/lassen-build-and-test.yml
strategy: depend
corona-build-and-test:
stage: sub-pipelines
variables:
# Explicitly pass down values that we want to be able to set when triggering
# pipelines manually or using scheduling
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/corona-build-and-test.yml
strategy: depend
+12 -36
View File
@@ -8,8 +8,6 @@
https://mfem.org
FIXME: this file needs to be updated
This directory contains most of the GitLab CI configuration. MFEM runs both PR
and nightly testing on GitLab.
@@ -17,9 +15,9 @@ and nightly testing on GitLab.
## Top level
The root configuration file is `.gitlab-ci.yml` at the root of MFEM repo. This
file only defines three stages, a prerequisites one, and two main stages in
which we trigger several sub-pipelines.
The root configuration file is `.gitlab-ci.yml` at the root of MFEM repo.
This file only defines one stage, in which we trigger several
sub-pipelines.
We use sub-pipelines to isolate the test for one combination of `machine`
and `test type`.
@@ -27,8 +25,8 @@ and `test type`.
Machines typically include:
* Dane: Intel Sapphire Rapids
* Matrix: Intel Sapphire Rapids + Nvidia H100 GPU
* Tioga: AMD MI250X GPU
* Lassen: Power9 + Nvidia GPU
* Corona: AMD GPU
Test types include:
@@ -41,31 +39,9 @@ altering the scheduling, execution and displaying of the others.
## Sub-pipelines
### build-and-test
The build-and-test sub-pipelines leverage RADIUSS Shared CI to share most of
the CI implementation. RADIUSS Shared CI provides a shared CI infrastructure
vetted on most LC systems of interest and efficiently leveraging each machine
scheduler to increase CI throughput. The maintenance of RADIUSS Shared CI is
shared among several RADIUSS projects.
Jobs for the build-and-test sub-pipelines are defined in the jobs directory.
Because build-and-test jobs leverage Uberenv and Spack to build the
dependencies automatically, the jobs essentially consists in a `spack spec`
defined in the jobs files, and some scheduling parameters defined in the
`.gitlab/custom-jobs-and-variables.yml` file.
Build-and-test jobs all run the `tests/gitlab/build_and_test` script.
The build-and-test pipelines are controlled by the
`.gitlab/subscribed-pipelines.yml` which defines which machines to run on and
implements additional features like machine availability check, and job list
generation.
### baseline
Baseline sub-pipelines are described by files with names reflecting the
machine it runs on, e.g. `dane-baseline`.
Each file is this directory is the root configuration file for one
sub-pipeline. The naming reflects the corresponding couple (`machine`,
`test_type`).
Those files define the *stages* and the *jobs* for the sub-pipeline. They
also contain any configuration that cannot be shared. For the most part
@@ -87,11 +63,11 @@ usage function. This should be improved.
# More testing
## Adding a new target to a build-and-test pipeline
## Adding a new target to a build_and_test pipeline
`build-and-test` pipelines rely on Spack to install dependencies. Spack is
`build_and_test` pipelines rely on Spack to install dependencies. Spack is
driven by Uberenv which helps freezing Spack configuration: the goal being to
point to a specific commit in Spack and isolate its configuration so that it is
point to specific commit in Spack and isolate its configuration so that it is
not influenced by the user environment. More documentation about this can be
found in `tests/gitlab`.
@@ -106,7 +82,7 @@ spack spec to use. Adding a job on Dane for example resumes to:
<job_name>:
variables:
SPEC: "<spack_spec>"
extends: .job_on_dane
extends: .build_and_test_on_dane
```
The remaining and non trivial work is to make sure this spec is working. To
+40
View File
@@ -0,0 +1,40 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
include:
- project: 'lc-templates/id_tokens'
file: 'id_tokens.yml'
# We define the following GitLab pipeline variables:
variables:
# The path to the shared resource between all jobs. For example, external
# repositories like 'tests' and 'tpls' are cloned here. Also, 'tpls' is built
# once for all targets, so that build happen here. The BUILD_ROOT is unique to
# the pipeline, preventing any form of concurrency with other pipelines. This
# also means that the BUILD_ROOT directory will never be cleaned.
# TODO: add a clean-up mechanism
BUILD_ROOT: ${USER_CI_TOP_DIR}/${CI_PROJECT_NAME}-${MACHINE_NAME}-pipeline-${CI_PIPELINE_ID}
# On LLNL's Dane, there is only one allocation shared among jobs in order to
# save time and resource. This allocation has to be uniquely named so that we
# are sure to retrieve it.
ALLOC_NAME: ${CI_PROJECT_NAME}_ci_${CI_PIPELINE_ID}
# Git repositories used in the pipeline
TPLS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tpls.git
TESTS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tests.git
AUTOTEST_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/autotest.git
MFEM_DATA_REPO: https://github.com/mfem/data.git
# Directory used to place artifacts.
ARTIFACTS_DIR: artifacts
SLURM_OVERLAP: 1
+59
View File
@@ -0,0 +1,59 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# GitLab pipeline configuration for the Corona machine at LLNL
variables:
MACHINE_NAME: corona
.on_corona:
tags:
- shell
- corona
rules:
# Don't run corona jobs if...
# Note: This makes corona an "opt-in" machine. To activate builds on corona
# for a given GitLab clone of MFEM, go to Setting/CI-CD/variables, and set
# "ON_CORONA" to "ON". An LC account on for corona is required to trigger a
# pipeline there.
- if: '$CI_COMMIT_BRANCH =~ /_cnone/ || $ON_CORONA != "ON"'
when: never
# Don't run autotest update if...
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
when: never
# Report success on success status
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
when: on_success
# Report failure on failure status
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
when: on_failure
# Always release resource
- if: '$CI_JOB_NAME =~ /release_resource/'
when: always
# Always cleanup
- if: '$CI_JOB_NAME =~ /cleanup/'
when: always
# Default is to run if previous stage succeeded
- when: on_success
# Spack helped builds
# Generic corona build job, extending build script
.build_and_test_on_corona:
extends: [.on_corona]
stage: build_and_test
script:
# THREADS is used by 'tests/gitlab/build_and_test', run below
- export THREADS=12
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
- echo ${MFEM_DATA_DIR}
- echo ${SPEC}
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) -t 15 -N 1 tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
+56
View File
@@ -0,0 +1,56 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# GitLab pipelines configurations for the Dane machine at LLNL
variables:
MACHINE_NAME: dane
.on_dane:
tags:
- shell
- dane
rules:
# Don't run dane jobs if...
- if: '$CI_COMMIT_BRANCH =~ /_qnone/ || $ON_DANE == "OFF"'
when: never
# Don't run autotest update if...
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
when: never
# Report success on success status
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
when: on_success
# Report failure on failure status
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
when: on_failure
# Always release resource
- if: '$CI_JOB_NAME =~ /release_resource/'
when: always
# Always cleanup
- if: '$CI_JOB_NAME =~ /cleanup/'
when: always
# Default is to run if previous stage succeeded
- when: on_success
# Spack helped builds
# Generic dane build job, extending build script
.build_and_test_on_dane:
extends: [.on_dane]
stage: build_and_test
script:
# THREADS is used by 'tests/gitlab/build_and_test', run below
# Dane has 224 threads/node and we run 7 separate jobs: 224=7*32
- export THREADS=28
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
- echo ${MFEM_DATA_DIR}
- echo ${SPEC}
- srun $( [[ -n "${JOBID}" ]] && echo "--jobid=${JOBID}" ) --reservation=ci -t 60 -N 1 tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
+48
View File
@@ -0,0 +1,48 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# GitLab pipelines configurations for the Lassen machine at LLNL
variables:
MACHINE_NAME: lassen
.on_lassen:
tags:
- shell
- lassen
rules:
- if: '$CI_COMMIT_BRANCH =~ /_lnone/ || $ON_LASSEN == "OFF"' #run except if ...
when: never
# Don't run autotest update if...
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "YES"'
when: never
# Report success on success status
- if: '$CI_JOB_NAME =~ /report_job_success/ && $AUTOTEST == "YES"'
when: on_success
# Report failure on failure status
- if: '$CI_JOB_NAME =~ /report_job_failure/ && $AUTOTEST == "YES"'
when: on_failure
# Always cleanup
- if: '$CI_JOB_NAME =~ /cleanup/'
when: always
- when: on_success
# Lassen uses a different job scheduler (spectrum lsf) that does not allow
# pre-allocation the same way slurm does. We use the pci queue on lassen
# to speed-up the allocation.
.build_and_test_on_lassen:
extends: [.on_lassen]
stage: build_and_test
script:
- echo ${MFEM_DATA_DIR}
- echo ${SPEC}
# Next script uses 'THREADS': leaving it empty --> it uses 'make all -j'
- lalloc 1 -W 45 -q pci --atsdisable tests/gitlab/build_and_test --spec "${SPEC}" --data-dir "${MFEM_DATA_DIR}" --data
needs: [setup]
+77
View File
@@ -0,0 +1,77 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Jobs report
.report_job_success:
script:
- echo ${MACHINE_NAME}
- echo ${AUTOTEST}
- echo ${AUTOTEST_COMMIT}
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
- cd ${AUTOTEST_ROOT}
- |
(
date
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
# every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do
sleep 5
done
echo "Acquired lock on '$PWD/autotest.lock'"
date
# Report SUCCESS while holding the file lock on 'autotest.lock'.
# The next script uses the following environment variables:
# - MACHINE_NAME, AUTOTEST_ROOT, AUTOTEST_COMMIT
# - CI_COMMIT_REF_SLUG, CI_PROJECT_DIR, CI_PIPELINE_URL
# It also calls the script '.gitlab/scripts/safe_create_rundir'.
${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test_success
err=$?
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> autotest.lock
.report_job_failure:
script:
- echo ${MACHINE_NAME}
- echo ${AUTOTEST}
- echo ${AUTOTEST_COMMIT}
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
- cd ${AUTOTEST_ROOT}
- |
(
date
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
# every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do
sleep 5
done
echo "Acquired lock on '$PWD/autotest.lock'"
date
# Report FAILURE while holding the file lock on 'autotest.lock'.
# The next script uses the following environment variables:
# - MACHINE_NAME, AUTOTEST_ROOT, AUTOTEST_COMMIT
# - CI_COMMIT_REF_SLUG, CI_PROJECT_DIR, CI_PIPELINE_URL
# It also calls the script '.gitlab/scripts/safe_create_rundir'.
${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test_failure
err=$?
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> autotest.lock
+90
View File
@@ -0,0 +1,90 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
# script then symlinks the repo to the parent directory of the MFEM source
# directory. Unit tests that depend on the mfem/data repo will then detect that
# this directory is present and be enabled.
setup:
tags:
- shell
- dane
stage: setup
variables:
GIT_STRATEGY: none
script:
#
# Setup MFEM_DATA_DIR=${SHARED_REPOS_DIR}/mfem-data, see '.gitlab-ci.yml'
# and '.gitlab/configs/<machine>-config.yml'
#
- echo "MACHINE_NAME = ${MACHINE_NAME}"
- echo "AUTOTEST = ${AUTOTEST}"
- echo "AUTOTEST_COMMIT = ${AUTOTEST_COMMIT}"
- echo "SHARED_REPOS_DIR ${SHARED_REPOS_DIR}"
- mkdir -p ${SHARED_REPOS_DIR} && cd ${SHARED_REPOS_DIR}
- command -v flock || echo "Required command 'flock' not found"
- |
(
date
echo "Waiting to acquire lock on '$PWD/mfem-data.lock' ..."
# try to get an exclusive lock on fd 9 (mfem-data.lock) repeating the
# try every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do
sleep 5
done
echo "Acquired lock on '$PWD/mfem-data.lock'"
date
# clone/update the mfem/data repo while holding the file lock on
# 'mfem-data.lock'
err=0
if [[ ! -d "mfem-data" ]]; then
git clone ${MFEM_DATA_REPO} "mfem-data"
else
cd "mfem-data" && git pull && cd ..
fi || err=1
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> mfem-data.lock
#
# Setup ${AUTOTEST_ROOT}/autotest:
#
- echo "AUTOTEST_ROOT ${AUTOTEST_ROOT}"
- mkdir -p ${AUTOTEST_ROOT} && cd ${AUTOTEST_ROOT}
- |
(
date
echo "Waiting to acquire lock on '$PWD/autotest.lock' ..."
# try to get an exclusive lock on fd 9 (autotest.lock) repeating the try
# every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do
sleep 5
done
echo "Acquired lock on '$PWD/autotest.lock'"
date
# clone/update the autotest repo while holding the file lock on
# 'autotest.lock'
err=0
if [[ ! -d "autotest" ]]; then
git clone ${AUTOTEST_REPO}
else
cd autotest && git pull && cd ..
fi || err=1
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> autotest.lock
+67
View File
@@ -0,0 +1,67 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
stages:
- setup
- allocate_resource
- build_and_test
- release_resource_and_report
# Slurm shared allocation
allocate_resource:
variables:
GIT_STRATEGY: none
extends: .on_corona
stage: allocate_resource
script:
- echo ${ALLOC_NAME}
- salloc --exclusive --nodes=1 --partition=mi60 --time=45 --no-shell --job-name=${ALLOC_NAME}
timeout: 6h
needs: [setup]
# Build and test jobs, simply provide a spec
rocm_gcc_8.3.1:
variables:
SPEC: "@develop%gcc@8.3.1+rocm amdgpu_target=gfx906"
extends: .build_and_test_on_corona
needs: [allocate_resource]
# Release slurm allocation
release_resource:
variables:
GIT_STRATEGY: none
extends: .on_corona
stage: release_resource_and_report
script:
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
- ([[ -n "${JOBID}" ]] && scancel ${JOBID})
needs: [rocm_gcc_8.3.1]
# Jobs report
report_job_success:
stage: release_resource_and_report
extends:
- .on_corona
- .report_job_success
report_job_failure:
stage: release_resource_and_report
extends:
- .on_corona
- .report_job_failure
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/corona-config.yml
- local: .gitlab/configs/setup-build-and-test.yml
- local: .gitlab/configs/report-build-and-test.yml
-132
View File
@@ -1,132 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
include:
- project: 'lc-templates/id_tokens'
file: 'id_tokens.yml'
# We define the following GitLab pipeline variables:
variables:
# Set the build-and-test command.
# Nested variables are allowed and useful to customize the job command. We
# protect variables with quotes so that their value may remain a string even if
# they contain whitespaces.
JOB_CMD:
value: tests/gitlab/build_and_test --spec \"${SPEC}\" --data-dir ${MFEM_DATA_DIR} --data
# The path to the shared resource between all jobs in the 'dane-baseline'
# pipeline. For example, external repositories like 'tests' and 'tpls' are
# cloned here. Also, 'tpls' is built once for all targets, so that build happens
# here. The BUILD_ROOT is unique to the pipeline, preventing any form of
# concurrency with other pipelines. This directory is removed by the 'cleanup'
# stage in the 'dane-baseline' pipeline.
BUILD_ROOT: ${USER_CI_TOP_DIR}/${CI_PROJECT_NAME}-${CI_MACHINE}-pipeline-${CI_PIPELINE_ID}
# On LLNL's dane and tioga, the 'build-and-test' pipelines creates only one
# allocation shared among jobs in the pipeline in order to save time and
# resources. This allocation has to be uniquely named so that we are sure to
# retrieve it and avoid collisions.
ALLOC_NAME: ${CI_PROJECT_NAME}_ci_${CI_PIPELINE_ID}
# Git repositories used in the pipelines:
# - TPLS_REPO and TESTS_REPO are used only by the 'dane-baseline' pipeline
# - AUTOTEST_REPO is used by all pipelines
# - MFEM_DATA_REPO is used only by the 'build-and-test' pipelines
TPLS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tpls.git
TESTS_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/tests.git
AUTOTEST_REPO: ssh://git@mybitbucket.llnl.gov:7999/mfem/autotest.git
MFEM_DATA_REPO: https://github.com/mfem/data.git
# Directory used to place artifacts:
# - ARTIFACTS_DIR is only used by the 'dane-baseline' pipeline
ARTIFACTS_DIR: artifacts
SLURM_OVERLAP: 1
# Dane
# Arguments for top level allocation
DANE_SHARED_ALLOC: "--exclusive --reservation=ci --time=60 --nodes=1"
# Arguments for job level allocation
# Note: We repeat the reservation, necessary when jobs are manually re-triggered.
DANE_JOB_ALLOC: "--reservation=ci --overlap --nodes=1"
# Tioga
# Arguments for top level allocation
TIOGA_SHARED_ALLOC: "--queue=pci --exclusive --time-limit=45m --nodes=1"
# Arguments for job level allocation
TIOGA_JOB_ALLOC: "--nodes=1 --begin-time=+5s"
# Matrix
# Arguments for top level allocation
MATRIX_SHARED_ALLOC: "-p pdebug --exclusive --time=45 --nodes=1 -G 4"
# Arguments for job level allocation
# Note: We repeat the reservation, necessary when jobs are manually re-triggered.
MATRIX_JOB_ALLOC: "--overlap --nodes=1"
# Configuration shared by build and test jobs specific to this project.
# Not all configuration can be shared. Here projects can fine tune the
# CI behavior.
# See Umpire for an example (export junit test reports).
.custom_job:
artifacts:
reports:
# Note: this part is not used by the 'dane-baseline' pipeline.
# FIXME: BUILD_ROOT, TPLS_REPO, TESTS_REPO are not needed here.
# Also, the definition of SHARED_REPOS_DIR is wrong.
.reproducer_vars:
script:
- |
echo -e "
# Variables \n
export SPEC=\"${SPEC//\"/\\\"}\" \n
# Directories \n
export BUILD_ROOT=\"\${working_dir}\" \n
export SHARED_REPOS_DIR=\"\${BUILD_ROOT}/..\" \n
export MFEM_DATA_DIR=\"\${SHARED_REPOS_DIR}/mfem-data\" \n
# Repositories \n
export TPLS_REPO=\"${TPLS_REPO//\"/\\\"}\" \n
export TESTS_REPO=\"${TESTS_REPO//\"/\\\"}\" \n
export AUTOTEST_REPO=\"${AUTOTEST_REPO//\"/\\\"}\" \n
export MFEM_DATA_REPO=\"${MFEM_DATA_REPO//\"/\\\"}\" \n
# Setup directories \n
./tests/gitlab/build_and_test_setup \n
# Using the CI build cache is optional and requires a token. Set it like so: \n
# export REGISTRY_TOKEN=\"<your token here>\" \n"
#
# Jobs report
.report_job_success:
script:
- ${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test SUCCESS
rules:
- when: on_success
.report_job_failure:
script:
- ${CI_PROJECT_DIR}/.gitlab/scripts/report_build_and_test FAILURE
rules:
- when: on_failure
# Keep the following for debugging purposes: renaming this job from
# '.show_variables' to 'show_variables' will insert this debug job at the
# beginning of all child pipelines.
.show_variables:
tags: [shell, oslic]
variables:
GIT_STRATEGY: none
stage: .pre
script:
- |
echo "~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~"
echo "AUTOTEST=${AUTOTEST}"
echo "AUTOTEST_COMMIT=${AUTOTEST_COMMIT}"
echo "~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~"
# Fail the job on purpose to prevent the rest of the pipeline from running
false
+5 -33
View File
@@ -11,7 +11,6 @@
variables:
BASELINE_TEST: baseline
MACHINE_NAME: dane
stages:
- setup
@@ -20,25 +19,6 @@ stages:
- cleanup
- baseline_publish
.on_dane:
tags:
- shell
- dane
rules:
# Don't run dane jobs if...
- if: '$ON_DANE == "OFF"'
when: never
# Don't run autotest update if...
# Note: in some cases, the content of AUTOTEST can be '${AUTOTEST}', so we
# need to treat that value as the default value of 'OFF'.
- if: '$CI_JOB_NAME =~ /report/ && $AUTOTEST != "ON" && $AUTOTEST != "YES"'
when: never
# Always cleanup
- if: '$CI_JOB_NAME =~ /cleanup/'
when: always
# Default is to run if previous stage succeeded
- when: on_success
baselinecheck_mfem_intel_dane:
extends: [.on_dane]
stage: baseline_check
@@ -49,9 +29,6 @@ baselinecheck_mfem_intel_dane:
# .gitlab/configs/setup-baseline.yml.
TPLS_DIR: ${BUILD_ROOT}/tpls
script:
- echo "AUTOTEST=$AUTOTEST"
- echo "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
- echo "AUTOTEST_ROOT=$AUTOTEST_ROOT"
- echo ${BUILD_ROOT}
- echo ${TPLS_DIR}
# Used by the tests in MFEM/tests, dane has 224 threads/node:
@@ -112,13 +89,7 @@ report_baseline:
cp ${rundir}/pipeline.txt ${rundir}/autotest-email.html
fi
msg="GitLab CI log for ${BASELINE_TEST} on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
# Note: in some cases, the content of AUTOTEST_COMMIT can be
# '${AUTOTEST_COMMIT}', so we need to treat that value as the default
# value of 'ON'.
if [[ "$AUTOTEST_COMMIT" == '${AUTOTEST_COMMIT}' ]]; then
AUTOTEST_COMMIT="ON"
fi
if [[ "$AUTOTEST_COMMIT" == "ON" || "$AUTOTEST_COMMIT" == "YES" ]]; then
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
git pull && \
git add ${rundir} && \
git commit -m "${msg}" && \
@@ -146,8 +117,8 @@ baselinepublish_mfem_dane:
extends: [.on_dane]
stage: baseline_publish
rules:
# - if: '$CI_COMMIT_BRANCH == "master" || $REBASELINE == "ON"'
- if: '$REBASELINE == "ON"'
# - if: '$CI_COMMIT_BRANCH == "master" || $REBASELINE == "YES"'
- if: '$REBASELINE == "YES"'
when: manual
script:
- echo ${BUILD_ROOT}
@@ -157,5 +128,6 @@ baselinepublish_mfem_dane:
- .gitlab/scripts/rebaseline
include:
- local: .gitlab/custom-jobs-and-variables.yml
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/dane-config.yml
- local: .gitlab/configs/setup-baseline.yml
+94
View File
@@ -0,0 +1,94 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
stages:
- setup
- allocate_resource
- build_and_test
- release_resource_and_report
# Allocate
allocate_resource:
variables:
GIT_STRATEGY: none
extends: .on_dane
stage: allocate_resource
script:
- echo ${ALLOC_NAME}
- salloc --exclusive --nodes=1 --reservation=ci --time=60 --no-shell --job-name=${ALLOC_NAME}
timeout: 6h
# GitLab jobs for the Dane machine at LLNL
debug_ser_gcc_10:
variables:
SPEC: "%gcc@10.3.1 +debug~mpi"
extends: .build_and_test_on_dane
debug_par_gcc_10:
variables:
SPEC: "%gcc@10.3.1 +debug+mpi"
extends: .build_and_test_on_dane
opt_ser_gcc_10:
variables:
SPEC: "%gcc@10.3.1 ~mpi"
extends: .build_and_test_on_dane
opt_par_gcc_10:
variables:
SPEC: "%gcc@10.3.1"
extends: .build_and_test_on_dane
opt_par_gcc_10_sundials:
variables:
SPEC: "%gcc@10.3.1 +sundials"
extends: .build_and_test_on_dane
opt_par_gcc_10_petsc:
variables:
SPEC: "%gcc@10.3.1 +petsc ^petsc+mumps~superlu-dist"
extends: .build_and_test_on_dane
opt_par_gcc_10_pumi:
variables:
SPEC: "%gcc@10.3.1 +pumi"
extends: .build_and_test_on_dane
# Release
release_resource:
variables:
GIT_STRATEGY: none
extends: .on_dane
stage: release_resource_and_report
script:
- echo ${ALLOC_NAME}
- export JOBID=$(squeue -h --name=${ALLOC_NAME} --format=%A)
- echo ${JOBID}
- ([[ -n "${JOBID}" ]] && scancel ${JOBID})
# Jobs report
report_job_success:
stage: release_resource_and_report
extends:
- .on_dane
- .report_job_success
report_job_failure:
stage: release_resource_and_report
extends:
- .on_dane
- .report_job_failure
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/dane-config.yml
- local: .gitlab/configs/setup-build-and-test.yml
- local: .gitlab/configs/report-build-and-test.yml
-19
View File
@@ -1,19 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Jobs report
report_job_success:
extends: [.on_dane, .report_job_success]
stage: jobs-stage-3
report_job_failure:
extends: [.on_dane, .report_job_failure]
stage: jobs-stage-3
-87
View File
@@ -1,87 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Override reproducer section to define MFEM specific variables.
.dane_reproducer_vars:
script:
- !reference [.reproducer_vars, script]
# TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY
# cannot be "none" anymore).
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
# script then symlinks the repo to the parent directory of the MFEM source
# directory. Unit tests that depend on the mfem/data repo will then detect that
# this directory is present and be enabled.
setup:
extends: .on_dane
stage: jobs-stage-1
script:
- ./tests/gitlab/build_and_test_setup
########################
# Overridden shared jobs
########################
# When using shared jobs, we can duplicate them here to override description and
# add necessary changes.
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
# the comparison with the original job is easier.
############
# Extra jobs
############
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
# describe the spec here.
.mfem_job_on_dane:
extends: .job_on_dane
stage: jobs-stage-2
variables:
# Dane has 224 threads/node and we run 7 separate jobs: 224=7*32
THREADS: 28
debug_ser_gcc_10:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +debug~mpi"
debug_par_gcc_10:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +debug+mpi"
opt_ser_gcc_10:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 ~mpi"
opt_par_gcc_10:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1"
opt_par_gcc_10_sundials:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +sundials"
opt_par_gcc_10_petsc:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +petsc ^petsc+mumps~superlu-dist"
opt_par_gcc_10_pumi:
extends: .mfem_job_on_dane
variables:
SPEC: "%gcc@10.3.1 +pumi"
-19
View File
@@ -1,19 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Jobs report
report_job_success:
extends: [.on_matrix, .report_job_success]
stage: jobs-stage-3
report_job_failure:
extends: [.on_matrix, .report_job_failure]
stage: jobs-stage-3
-65
View File
@@ -1,65 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Override reproducer section to define UMPIRE specific variables.
.matrix_reproducer_vars:
script:
- !reference [.reproducer_vars, script]
#TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY cannot be "none" anymore).
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
# script then symlinks the repo to the parent directory of the MFEM source
# directory. Unit tests that depend on the mfem/data repo will then detect that
# this directory is present and be enabled.
setup:
extends: .on_matrix
stage: jobs-stage-1
script:
- ./tests/gitlab/build_and_test_setup
########################
# Overridden shared jobs
########################
# When using shared jobs , we can duplicate them here to override description and add necessary changes.
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
# the comparison with the original job is easier.
############
# Extra jobs
############
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
# describe the spec here.
.mfem_job_on_matrix:
extends: .job_on_matrix
stage: jobs-stage-2
variables:
# We run 2 jobs on 1 node that has 112 threads
THREADS: 48
# These modules need to be consistent with the uberenv configurations:
MODULE_LIST: "gcc/10.3.1-magic cuda/12.9.1"
allocate_resources:
timeout: 4h
opt_mpi_cuda_gcc:
extends: .mfem_job_on_matrix
variables:
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90"
opt_mpi_cuda_hypre_cuda_gcc:
extends: .mfem_job_on_matrix
variables:
SPEC: "%gcc@10.3.1 +mpi +cuda cuda_arch=90 ^hypre+cuda"
-20
View File
@@ -1,20 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Jobs report
report_job_success:
extends: [.on_tioga, .report_job_success]
stage: jobs-stage-3
report_job_failure:
extends: [.on_tioga, .report_job_failure]
stage: jobs-stage-3
-70
View File
@@ -1,70 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Override reproducer section to define UMPIRE specific variables.
.tioga_reproducer_vars:
script:
- !reference [.reproducer_vars, script]
#TODO: Setup script should be defined as a bash script (but then GIT_STRATEGY cannot be "none" anymore).
# Setup clones the mfem/data repo in ${SHARED_REPOS_DIR}. The build_and_test
# script then symlinks the repo to the parent directory of the MFEM source
# directory. Unit tests that depend on the mfem/data repo will then detect that
# this directory is present and be enabled.
setup:
extends: .on_tioga
stage: jobs-stage-1
script:
- ./tests/gitlab/build_and_test_setup
########################
# Overridden shared jobs
########################
# When using shared jobs , we can duplicate them here to override description and add necessary changes.
# We keep ${PROJECT_<MACHINE>_VARIANTS} and ${PROJECT_<MACHINE>_DEPS} So that
# the comparison with the original job is easier.
############
# Extra jobs
############
# We do not recommend using ${PROJECT_<MACHINE>_VARIANTS} and
# ${PROJECT_<MACHINE>_DEPS} in the extra jobs. There is not reason not to fully
# describe the spec here.
# Build and test jobs, simply provide a spec
#.tioga_job_command:
# script:
# - echo PROXY="${PROXY}"
# - echo TIOGA_JOB_ALLOC="${TIOGA_JOB_ALLOC}"
# - "printf '#!/bin/bash\n%s\n' \"${JOB_CMD}\" > flux_script.sh"
# - cat flux_script.sh
# - ${PROXY} flux watch $( ${PROXY} flux batch -o output.stdout.type=kvs ${TIOGA_JOB_ALLOC} flux_script.sh )
# - rm -f flux_script.sh
.mfem_job_on_tioga:
extends: .job_on_tioga
stage: jobs-stage-2
variables:
# We run 1 job on 1 node that has 64 threads
THREADS: 64
opt_mpi_rocm_hypre_rocm:
extends: .mfem_job_on_tioga
variables:
SPEC: "%rocmcc@=6.3.1 +rocm amdgpu_target=gfx90a ^hypre+rocm"
# cce_16_0_1:
# extends: .mfem_job_on_tioga
# variables:
# SPEC: "%cce@=16.0.1"
+44
View File
@@ -0,0 +1,44 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
stages:
- setup
- build_and_test
- report
opt_mpi_cuda_gcc:
variables:
SPEC: "%gcc@8.3.1 +mpi +cuda cuda_arch=70"
extends: .build_and_test_on_lassen
opt_mpi_cuda_hypre_cuda_gcc:
variables:
SPEC: "%gcc@8.3.1 +mpi +cuda cuda_arch=70 ^hypre+cuda~shared cuda_arch=70"
extends: .build_and_test_on_lassen
# Jobs report
report_job_success:
stage: report
extends:
- .on_lassen
- .report_job_success
report_job_failure:
stage: report
extends:
- .on_lassen
- .report_job_failure
include:
- local: .gitlab/configs/common.yml
- local: .gitlab/configs/lassen-config.yml
- local: .gitlab/configs/setup-build-and-test.yml
- local: .gitlab/configs/report-build-and-test.yml
+2
View File
@@ -35,6 +35,8 @@ if [[ "${MACHINE_NAME}" == "dane" ]]; then
salloc --nodes=1 --exclusive --reservation=ci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "corona" ]]; then
salloc --nodes=1 -t 60 -p pbatch ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
elif [[ ${MACHINE_NAME} == "lassen" ]]; then
lalloc 1 -q pci ../runtest ../../mfem "${BASELINE_TEST} ${TPLS_DIR}"
else
echo "Unknown machine: MACHINE_NAME=$MACHINE_NAME"
exit 1
-118
View File
@@ -1,118 +0,0 @@
#!/bin/bash
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
function info_msg ()
{
echo "[Information:] ${1}"
}
function error_msg ()
{
echo "[Error:] ${1}"
}
# Perform a report while holding a lock file to prevent concurrency on
# the destination.
# Usage:
# locked_clone <report_function> <lock_name>
function locked_report ()
{
if ! command -v flock
then
error_msg "Required command 'flock' not found"
exit 1
fi
info_msg "Will report ${1} while holding a lock in ${2}"
( date; info_msg "Waiting to acquire lock on '${PWD}/${2}.lock' ..."
# try to get an exclusive lock on fd 9 (mfem-data.lock) repeating the
# try every 5 seconds; we may want to add a counter for the number of
# retries to interrupt a potential infinite loop
while ! flock -n 9; do sleep 5; done
date; info_msg "Acquired lock on '${PWD}/${2}.lock'"
report ${1}
err=$?
# sleep for a period to allow NFS to propagate the above changes;
# clearly, there is no guarantee that other NFS clients will see the
# changes even after the timeout
sleep 10
exit $err
) 9> ${2}.lock
}
function report ()
{
if [[ "${1}" == "SUCCESS" ]]
then
info_msg "All the ${MACHINE_NAME} jobs passed"
status_msg="The 'build-and-test' jobs on ${MACHINE_NAME} were SUCCESSFUL."
elif [[ "${1}" == "FAILURE" ]]
then
info_msg "At least one failure on ${MACHINE_NAME}"
status_msg="Some 'build-and-test' jobs on ${MACHINE_NAME} FAILED."
else
error_msg "Unknown status: ${1} ... aborting"
exit 1
fi
cd ${AUTOTEST_ROOT}/autotest || \
{ error_msg "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
mkdir -p ${MACHINE_NAME}
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
printf "%s\n" "${status_msg}" \
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.err
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
if [[ "${1}" == "FAILURE" ]]
then
# Create 'autotest-email.html' to indicate failure:
cp ${rundir}/gitlab.err ${rundir}/autotest-email.html
fi
# Note: in some cases, the content of AUTOTEST_COMMIT can be
# '${AUTOTEST_COMMIT}', so we need to treat that value as the default
# value of 'ON'.
if [[ "$AUTOTEST_COMMIT" == '${AUTOTEST_COMMIT}' ]]; then
AUTOTEST_COMMIT="ON"
fi
if [[ "$AUTOTEST_COMMIT" == "ON" || "$AUTOTEST_COMMIT" == "YES" ]]; then
git pull && \
git add ${rundir} && \
git commit -m "${msg}" && \
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
else
for file in ${rundir}/*; do
echo "------------------------------"
echo "Content of '$file'"
echo "******************************"
cat $file
echo "******************************"
done
rm -rf ${rundir} || true
fi
}
export MACHINE_NAME=${CI_MACHINE}
info_msg "MACHINE_NAME is ${MACHINE_NAME}"
info_msg "AUTOTEST_ROOT is ${AUTOTEST_ROOT}"
info_msg "AUTOTEST=$AUTOTEST"
info_msg "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
cd ${AUTOTEST_ROOT} && locked_report ${1} autotest
+45
View File
@@ -0,0 +1,45 @@
#!/bin/bash
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
echo "Runs if there was at least one failure on ${MACHINE_NAME}"
cd ${AUTOTEST_ROOT}/autotest || \
{ echo "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
mkdir -p ${MACHINE_NAME}
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
printf "%s\n" "Some 'build-and-test' jobs on ${MACHINE_NAME} FAILED." \
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.err
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
# Create 'autotest-email.html' to indicate failure:
cp ${rundir}/gitlab.err ${rundir}/autotest-email.html
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
git pull && \
git add ${rundir} && \
git commit -m "${msg}" && \
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
else
for file in ${rundir}/*; do
echo "------------------------------"
echo "Content of '$file'"
echo "******************************"
cat $file
echo "******************************"
done
rm -rf ${rundir} || true
fi
+42
View File
@@ -0,0 +1,42 @@
#!/bin/bash
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
echo "Can only run if all the ${MACHINE_NAME} jobs passed"
cd ${AUTOTEST_ROOT}/autotest || \
{ echo "Invalid 'autotest' dir: ${AUTOTEST_ROOT}/autotest"; exit 1; }
mkdir -p ${MACHINE_NAME}
rundir="${MACHINE_NAME}/$(date +%Y-%m-%d)-gitlab-ci-${CI_COMMIT_REF_SLUG}"
rundir=$(${CI_PROJECT_DIR}/.gitlab/scripts/safe_create_rundir $rundir)
printf "%s\n" "The 'build-and-test' jobs on ${MACHINE_NAME} were SUCCESSFUL." \
"Pipeline URL:" "$CI_PIPELINE_URL" > ${rundir}/gitlab.out
msg="GitLab CI log for build-and-test on ${MACHINE_NAME} ($(date +%Y-%m-%d))"
if [[ "$AUTOTEST_COMMIT" != "NO" ]]; then
git pull && \
git add ${rundir} && \
git commit -m "${msg}" && \
${CI_PROJECT_DIR}/.gitlab/scripts/git_try_to_push
else
for file in ${rundir}/*; do
echo "------------------------------"
echo "Content of '$file'"
echo "******************************"
cat $file
echo "******************************"
done
rm -rf ${rundir} || true
fi
-130
View File
@@ -1,130 +0,0 @@
# Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
# at the Lawrence Livermore National Laboratory. All Rights reserved. See files
# LICENSE and NOTICE for details. LLNL-CODE-806117.
#
# This file is part of the MFEM library. For more information and source code
# availability visit https://mfem.org.
#
# MFEM is free software; you can redistribute it and/or modify it under the
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# The template job to test whether a machine is up.
# Expects CI_MACHINE defined to machine name.
.machine-check:
stage: prerequisites
tags: [shell, oslic]
variables:
GIT_STRATEGY: none
script:
- |
if [[ $(jq '.[env.CI_MACHINE].total_nodes_up' /usr/global/tools/lorenz/data/loginnodeStatus) == 0 ]]
then
echo -e "\e[31mNo node available on ${CI_MACHINE}\e[0m"
false && \
curl --url "https://api.github.com/repos/${GITHUB_PROJECT_ORG}/${GITHUB_PROJECT_NAME}/statuses/${CI_COMMIT_SHA}" \
--header 'Content-Type: application/json' \
--header "authorization: Bearer ${GITHUB_TOKEN}" \
--data "{ \"state\": \"failure\", \"target_url\": \"${CI_PIPELINE_URL}\", \"description\": \"GitLab ${CI_MACHINE} down\", \"context\": \"ci/gitlab/${CI_MACHINE}\" }"
exit 1
fi
###
# Trigger a build-and-test pipeline for a machine.
# Comment the jobs for machines you dont need.
###
# One job to generate the job list for all the subpipelines
generate-job-lists:
stage: prerequisites
tags: [shell, oslic]
variables:
LOCAL_JOBS_PATH: ".gitlab/jobs"
script:
- |
echo "AUTOTEST=$AUTOTEST"
echo "AUTOTEST_COMMIT=$AUTOTEST_COMMIT"
echo "AUTOTEST_ROOT=$AUTOTEST_ROOT"
- |
cat ${LOCAL_JOBS_PATH}/dane.yml > dane-jobs.yml
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
then
cat ${LOCAL_JOBS_PATH}/dane-reports.yml >> dane-jobs.yml
fi
- |
cat ${LOCAL_JOBS_PATH}/matrix.yml > matrix-jobs.yml
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
then
cat ${LOCAL_JOBS_PATH}/matrix-reports.yml >> matrix-jobs.yml
fi
- |
cat ${LOCAL_JOBS_PATH}/tioga.yml > tioga-jobs.yml
if [[ ${AUTOTEST} == "ON" || ${AUTOTEST} == "YES" ]]
then
cat ${LOCAL_JOBS_PATH}/tioga-reports.yml >> tioga-jobs.yml
fi
artifacts:
paths:
- dane-jobs.yml
- matrix-jobs.yml
- tioga-jobs.yml
# DANE
dane-up-check:
variables:
CI_MACHINE: "dane"
extends: [.machine-check]
dane-build-and-test:
variables:
CI_MACHINE: "dane"
needs: [dane-up-check, generate-job-lists]
extends: [.build-and-test]
# DANE, MFEM Specific
dane-baseline:
stage: test-pipelines
variables:
# Explicitly pass down values that are not always propagated to child
# pipelines, e.g. when a variable is set in the "Settings -> CI" web
# interface (project variables).
# Note: in some cases, this does not work as expected, e.g. when the
# variable is not re-defined in the web interface; in such cases, the child
# pipeline gets a definition like '${AUTOTEST}', i.e. it behaves as if
# AUTOTEST is undefined, even though there is a default value in
# .gitlab-ci.yml.
AUTOTEST: "${AUTOTEST}"
AUTOTEST_COMMIT: "${AUTOTEST_COMMIT}"
trigger:
include: .gitlab/dane-baseline.yml
strategy: depend
forward:
pipeline_variables: true
needs: [dane-up-check]
# TIOGA
tioga-up-check:
variables:
CI_MACHINE: "tioga"
extends: [.machine-check]
tioga-build-and-test:
variables:
CI_MACHINE: "tioga"
needs: [tioga-up-check, generate-job-lists]
extends: [.build-and-test]
# Matrix
matrix-up-check:
variables:
CI_MACHINE: "matrix"
extends: [.machine-check]
matrix-build-and-test:
variables:
CI_MACHINE: "matrix"
needs: [matrix-up-check, generate-job-lists]
extends: [.build-and-test]
-16
View File
@@ -43,13 +43,6 @@ Discretization improvements
Meshing improvements
--------------------
- The TMOP kernel hierarchy has been restructured to reduce compilation time.
Most large kernels have been split into smaller, specific ones, with kernels
for each metric. The directory structure has been updated with assemble,
metrics, mult and tools subdirectories. The new kernel dispatch and
specialization system has also been integrated.
Unit tests have been revised to ensure --all tests pass.
- Introduced NC-patch NURBS meshes, which are conforming element-wise but allow
for nonconforming patch topology. This new mesh format supports element
spacing formulas for refinement, as well as local refinement factors for a
@@ -83,15 +76,6 @@ GPU computing
- Added GPU support in GradientGridFunctionCoefficient and
InnerProductCoefficient by implementing their Project methods.
Linear and nonlinear solvers
----------------------------
- Added `FilteredSolver`: a base class for solvers with filtering. It handles cases
where a solver performs well except in small subspaces, by adding a filtering step
formulated as a subspace correction.
- Added `AMGFSolver`: a derived class of `FilteredSolver`, specialized for
AMG with Filtering (AMGF), providing robust preconditioning for linear systems
arising in constrained optimization problems such as frictionless contact.
New and updated examples and miniapps
-------------------------------------
- Added miniapps to demonstrate an implementation of the absolute-value
+24 -57
View File
@@ -133,49 +133,33 @@ if (MFEM_USE_CUDA)
if (NOT CMAKE_CUDA_HOST_COMPILER)
set(CMAKE_CUDA_HOST_COMPILER ${CMAKE_CXX_COMPILER})
endif()
if (NOT CMAKE_CUDA_ARCHITECTURES)
# make CUDA_ARCH resemble the same form as CMAKE_CUDA_ARCHITECTURES
string(REPLACE "sm_" "" CUDA_ARCH_TMP "${CUDA_ARCH}")
string(REPLACE "," ";" CUDA_ARCH "${CUDA_ARCH_TMP}")
set(CMAKE_CUDA_ARCHITECTURES "${CUDA_ARCH}")
if (CMAKE_VERSION VERSION_LESS 3.18.0)
set(CUDA_FLAGS "-arch=${CUDA_ARCH} ${CUDA_FLAGS}")
elseif (NOT CMAKE_CUDA_ARCHITECTURES)
string(REGEX REPLACE "^sm_" "" ARCH_NUMBER "${CUDA_ARCH}")
if ("${CUDA_ARCH}" STREQUAL "sm_${ARCH_NUMBER}")
set(CMAKE_CUDA_ARCHITECTURES "${ARCH_NUMBER}")
else()
message(FATAL_ERROR "Unknown CUDA_ARCH: ${CUDA_ARCH}")
endif()
else()
set(CUDA_ARCH "CMAKE_CUDA_ARCHITECTURES: ${CMAKE_CUDA_ARCHITECTURES}")
endif()
message(STATUS "Using CUDA architecture: ${CUDA_ARCH}")
enable_language(CUDA)
if (CMAKE_VERSION VERSION_LESS 3.18.0)
# backup try to detect if this is clang or nvcc
if(CMAKE_CUDA_COMPILER MATCHES "nvcc$")
# nvcc
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
if ("all" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
OR "native" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}"
OR "all-major" STREQUAL "${CMAKE_CUDA_ARCHITECTURES}")
set(CUDA_FLAGS "-arch=${CMAKE_CUDA_ARCHITECTURES} ${CUDA_FLAGS}")
else()
# build -gencode sequence for multiple architectures
foreach(ENTRY IN LISTS CMAKE_CUDA_ARCHITECTURES)
set(CUDA_FLAGS
"-gencode arch=compute_${ENTRY},code=sm_${ENTRY} ${CUDA_FLAGS}")
endforeach()
endif()
else()
# build cuda-gpu-arch sequence for multiple architectures
# does not support all/all-major/native
foreach(ENTRY IN LISTS CMAKE_CUDA_ARCHITECTURES)
set(CUDA_FLAGS "-cuda-gpu-arch=sm_${ENTRY} ${CUDA_FLAGS}")
endforeach()
endif()
# backup try to detect if this is clang or nvcc
if(CMAKE_CUDA_COMPILER MATCHES "nvcc$")
# nvcc
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
endif()
else()
# TODO: all, native, all-major require CMake 3.24+
# backport support for CMake 3.18 to 3.24
if (CMAKE_CUDA_COMPILER_ID STREQUAL "NVIDIA")
# nvcc
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
set(CUDA_FLAGS
"${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
endif()
if (CMAKE_CUDA_COMPILER_ID STREQUAL "NVIDIA")
# nvcc
set(MFEM_CUDA_COMPILER_IS_NVCC ON)
set(CUDA_FLAGS "${CUDA_FLAGS} --expt-extended-lambda --expt-relaxed-constexpr")
endif()
endif()
set(CMAKE_CUDA_STANDARD ${CMAKE_CXX_STANDARD} CACHE STRING
"CUDA standard to use.")
@@ -258,16 +242,10 @@ endif()
# AMD HIP
if (MFEM_USE_HIP)
if (NOT CMAKE_HIP_ARCHITECTURES)
if (HIP_ARCH)
set(CMAKE_HIP_ARCHITECTURES CACHE STRING "HIP targets to compile for" "${HIP_ARCH}")
set(GPU_TARGETS "${HIP_ARCH}" CACHE STRING "HIP targets to compile for" FORCE)
endif()
else()
set(HIP_ARCH CACHE STRING "HIP targets to compile for" "${CMAKE_HIP_ARCHITECTURES}")
set(GPU_TARGETS "${CMAKE_HIP_ARCHITECTURES}" CACHE STRING "HIP targets to compile for" FORCE)
if (HIP_ARCH)
message(STATUS "Using HIP architecture: ${HIP_ARCH}")
set(GPU_TARGETS "${HIP_ARCH}" CACHE STRING "HIP targets to compile for")
endif()
message(STATUS "Using HIP architecture: ${CMAKE_HIP_ARCHITECTURES}")
if (ROCM_PATH)
list(INSERT CMAKE_PREFIX_PATH 0 ${ROCM_PATH})
endif()
@@ -300,19 +278,8 @@ if (MFEM_USE_OPENMP OR MFEM_USE_LEGACY_OPENMP)
endif()
endif()
# Warn user if deprecated FETCH_TPLS is provided
if (DEFINED FETCH_TPLS)
message(STATUS "Setting MFEM_FETCH_TPLS to user-provided value of FETCH_TPLS (i.e., MFEM_FETCH_TPLS=${FETCH_TPLS})")
set (MFEM_FETCH_TPLS FETCH_TPLS)
message(DEPRECATION "The use of FETCH_TPLS is deprecated and will be removed in future verison. Please use MFEM_FETCH_TPLS instead.")
endif()
# Umpire (must be included before hypre, so hypre can use it if needed)
# Umpire (must be included before hypre, so hypre can use it if needed)
if (MFEM_USE_UMPIRE)
# umpire uses FindCUDA, which needs CMP0146=OLD in CMake >= 3.27
if (CMAKE_VERSION VERSION_GREATER_EQUAL 3.27.0)
cmake_policy(SET CMP0146 OLD)
endif()
find_package(UMPIRE REQUIRED)
endif()
+4 -6
View File
@@ -123,7 +123,7 @@ Parallel build:
Parallel build with fetching of hypre and METIS:
mkdir <mfem-buil-dir> ; cd <mfem-build-dir>
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DMFEM_FETCH_TPLS=YES
cmake <mfem-source-dir> -DMFEM_USE_MPI=YES -DFETCH_TPLS=YES
make -j 4
CUDA build:
@@ -1081,10 +1081,9 @@ The following options are CMake specific:
MFEM_ENABLE_TESTING - Enable the ctest framework for testing.
MFEM_ENABLE_EXAMPLES - Build all of the examples by default.
MFEM_ENABLE_MINIAPPS - Build all of the miniapps by default.
MFEM_FETCH_TPLS - Enable fetching of all supported third-party libraries.
MFEM_FETCH_GSLIB - Enable fetching of gslib.
MFEM_FETCH_HYPRE - Enable fetching of hypre.
MFEM_FETCH_METIS - Enable fetching of metis.
FETCH_TPLS - Enable fetching of all supported third-party libraries.
HYPRE_FETCH - Enable fetching of hypre.
METIS_FETCH - Enable fetching of metis.
External libraries (CMake):
---------------------------
@@ -1150,7 +1149,6 @@ The MFEM CMake build system also provides fetching (automated building) for the
packages/libraries listed below. Note that when fetching is enabled, any related
auto-detection functionality is disabled.
- GSLIB
- HYPRE
- METIS
+1 -38
View File
@@ -9,47 +9,10 @@
# terms of the BSD-3 license. We welcome feedback and contributions, see file
# CONTRIBUTING.md for details.
# Defines the following variables if fetching of TPLs is disabled (default):
# Defines the following variables:
# - GSLIB_FOUND
# - GSLIB_LIBRARIES
# - GSLIB_INCLUDE_DIRS
# otherwise, the following are defined:
# - GSLIB (imported library target)
if (MFEM_FETCH_GSLIB OR MFEM_FETCH_TPLS)
enable_language(C)
string(TOUPPER "${CMAKE_BUILD_TYPE}" BUILD_TYPE)
set(GSLIB_FETCH_VERSION 1.0.9)
set(GSLIB_C_FLAGS ${CMAKE_C_FLAGS_${BUILD_TYPE}})
if (CMAKE_C_FLAGS)
set(GSLIB_C_FLAGS "${CMAKE_C_FLAGS} ${CMAKE_C_FLAGS_${BUILD_TYPE}}")
endif()
if (BUILD_SHARED_LIBS)
set(GSLIB_C_FLAGS "${GSLIB_C_FLAGS} -fPIC")
endif()
add_library(GSLIB STATIC IMPORTED)
# define external project and create future include directory so it is present
# to pass CMake checks at end of MFEM configuration step
message(STATUS "Will fetch GSLIB ${GSLIB_FETCH_VERSION} to be built with ${GSLIB_C_FLAGS}")
set(PREFIX ${CMAKE_BINARY_DIR}/fetch/gslib)
include(ExternalProject)
ExternalProject_Add(gslib
GIT_REPOSITORY https://github.com/Nek5000/gslib
GIT_TAG v${GSLIB_FETCH_VERSION}
GIT_SHALLOW TRUE
UPDATE_DISCONNECTED TRUE
PREFIX ${PREFIX}
CONFIGURE_COMMAND ""
BUILD_COMMAND cd ${PREFIX}/src/gslib && $(MAKE) clean && $(MAKE) DESTDIR=${PREFIX} MPI=$<BOOL:${MFEM_USE_MPI}> "CFLAGS= ${GSLIB_C_FLAGS}"
INSTALL_COMMAND "")
file(MAKE_DIRECTORY ${PREFIX}/include)
# set imported library target properties
add_dependencies(GSLIB gslib)
set_target_properties(GSLIB PROPERTIES
IMPORTED_LOCATION ${PREFIX}/lib/libgs.a
INTERFACE_INCLUDE_DIRECTORIES ${PREFIX}/include)
return()
endif()
include(MfemCmakeUtilities)
mfem_find_package(GSLIB GSLIB GSLIB_DIR "include" gslib.h "lib" gs
+8 -8
View File
@@ -37,21 +37,21 @@ if (HYPRE_FOUND OR TARGET HYPRE)
endif()
endif()
if (MFEM_FETCH_HYPRE OR MFEM_FETCH_TPLS)
set(HYPRE_FETCH_VERSION 2.33.0)
set(HYPRE_FETCH_TAG "v${HYPRE_FETCH_VERSION}" CACHE STRING "Tag, branch, or commit for HYPRE")
add_library(HYPRE STATIC IMPORTED)
# set options and associated dependencies
if (HYPRE_FETCH OR FETCH_TPLS)
# Collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
set(HYPRE_CMAKE_OPTIONS "")
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
# collect all HYPRE_ENABLE variables and pass them to hypre, assuming they are BOOL.
get_cmake_property(all_vars VARIABLES)
foreach(var ${all_vars})
if(var MATCHES "^HYPRE_ENABLE")
list(APPEND HYPRE_CMAKE_OPTIONS "-D${var}:BOOL=${${var}}")
endif()
endforeach()
# process all MFEM_USE variables that impact hypre
set(HYPRE_FETCH_VERSION 2.33.0)
set(HYPRE_FETCH_TAG "v${HYPRE_FETCH_VERSION}" CACHE STRING "Tag, branch, or commit for HYPRE")
add_library(HYPRE STATIC IMPORTED)
# set options and associated dependencies
list(APPEND HYPRE_CMAKE_OPTIONS -DCMAKE_BUILD_TYPE:STRING=${CMAKE_BUILD_TYPE})
if (MFEM_USE_CUDA)
list(APPEND HYPRE_CMAKE_OPTIONS -DHYPRE_ENABLE_CUDA:BOOL=ON -DCMAKE_CUDA_ARCHITECTURES:STRING=${CMAKE_CUDA_ARCHITECTURES})
find_package(CUDAToolkit REQUIRED)
+1 -1
View File
@@ -18,7 +18,7 @@
# - METIS (imported library target)
# - METIS_VERSION_5 (cache variable)
if (MFEM_FETCH_METIS OR MFEM_FETCH_TPLS)
if (METIS_FETCH OR FETCH_TPLS)
set(METIS_FETCH_VERSION 4.0.3)
add_library(METIS STATIC IMPORTED)
# define external project
+3 -4
View File
@@ -91,10 +91,9 @@ option(MFEM_ENABLE_BENCHMARKS "Build all of the benchmarks" OFF)
# Allow a user to specify fetching of certain third-party libraries instead of
# searching for existing installations.
option(MFEM_FETCH_TPLS "Enable fetching of all supported third-party libraries" OFF)
option(MFEM_FETCH_GSLIB "Enable fetching of GSLIB" OFF)
option(MFEM_FETCH_HYPRE "Enable fetching of hypre" OFF)
option(MFEM_FETCH_METIS "Enable fetching of METIS" OFF)
option(FETCH_TPLS "Enable fetching of all supported third-party libraries" OFF)
option(HYPRE_FETCH "Enable fetching of hypre" OFF)
option(METIS_FETCH "Enable fetching of METIS" OFF)
# Setting CXX/MPICXX on the command line or in user.cmake will overwrite the
# autodetected C++ compiler.
-3
View File
@@ -471,13 +471,10 @@ int main(int argc, char *argv[])
ofstream sol_r_ofs("sol_r.gf");
ofstream sol_i_ofs("sol_i.gf");
ofstream sol_z_ofs("sol_z.gf");
sol_r_ofs.precision(8);
sol_i_ofs.precision(8);
sol_z_ofs.precision(8);
u.real().Save(sol_r_ofs);
u.imag().Save(sol_i_ofs);
u.Save(sol_z_ofs);
}
// 14. Send the solution by socket to a GLVis server.
+1 -5
View File
@@ -507,11 +507,10 @@ int main(int argc, char *argv[])
// 15. Save the refined mesh and the solution in parallel. This output can be
// viewed later using GLVis: "glvis -np <np> -m mesh -g sol".
{
ostringstream mesh_name, sol_r_name, sol_i_name, sol_z_name;
ostringstream mesh_name, sol_r_name, sol_i_name;
mesh_name << "mesh." << setfill('0') << setw(6) << myid;
sol_r_name << "sol_r." << setfill('0') << setw(6) << myid;
sol_i_name << "sol_i." << setfill('0') << setw(6) << myid;
sol_z_name << "sol_z." << setfill('0') << setw(6) << myid;
ofstream mesh_ofs(mesh_name.str().c_str());
mesh_ofs.precision(8);
@@ -519,13 +518,10 @@ int main(int argc, char *argv[])
ofstream sol_r_ofs(sol_r_name.str().c_str());
ofstream sol_i_ofs(sol_i_name.str().c_str());
ofstream sol_z_ofs(sol_z_name.str().c_str());
sol_r_ofs.precision(8);
sol_i_ofs.precision(8);
sol_z_ofs.precision(8);
u.real().Save(sol_r_ofs);
u.imag().Save(sol_i_ofs);
u.Save(sol_z_ofs);
}
// 16. Send the solution by socket to a GLVis server.
+2 -2
View File
@@ -195,8 +195,8 @@ clean-build:
clean-exec:
@rm -f refined.mesh displaced.mesh mesh.* ex5.mesh ex6p-checkpoint.*
@rm -rf Example5* Example9* Example15* Example16* Example23* ParaView
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.* sol_z.*
@rm -f order.* ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
@rm -f sphere_refined.* sol.* sol_u.* sol_p.* sol_r.* sol_i.* order.*
@rm -f ex9.mesh ex9-mesh.* ex9-init.* ex9-final.*
@rm -f deformed.* velocity.* elastic_energy.* mode_* mode_deriv_* flux.*
@rm -f ex5-p-*.bp ex9-p-*.bp ex12-p-*.bp ex16-p-*.bp
@rm -f ex16.mesh ex16-mesh.* ex16-init.* ex16-final.*
+119 -66
View File
@@ -28,10 +28,8 @@
//
// The example demonstrates the use of nonlinear operators (the
// class ConductionOperator defining C(u)), as well as their
// implicit time integration. Note that implementing the method
// ConductionOperator::ImplicitSolve is the only requirement for
// high-order implicit (SDIRK) time integration. By default, this
// example uses the SUNDIALS ODE solvers from CVODE and ARKODE.
// implicit time integration. By default, this example uses the
// SUNDIALS ODE solvers from CVODE and ARKODE.
//
// We recommend viewing examples 2, 9 and 10 before viewing this
// example.
@@ -51,15 +49,16 @@ using namespace mfem;
* and K(u) is the diffusion operator with diffusivity depending on u:
* (\kappa + \alpha u).
*
* Class ConductionOperatorOperator represents the above ODE operator in the
* general form F(u, k, t) = G(u, t) where
* Class ConductionOperator represents the above ODE operator as a
* TimeDependentOperator for use with native MFEM integrators and CVODE
* integrators, i.e., F(u, k, t) = G(u, t) with F(u, du/dt, t) = du/dt and
* G(u, t) = -K(u) u
*
* 1. F(u, du/dt, t) = du/dt (ODE is expressed in EXPLICIT form)
* G(u, t) = - inv(M) K(u) u
* 2. F(u, du/dt, t) = M du/dt (ODE is expressed in IMPLICIT form)
* G(u, t) = - K(u) u
* Class ConductionOperator represents the above ODE operator as an
* ARKStepODE for use with ARKODE integrators, i.e., either M du/dt = -K(u) u
* (mass form) or du/dt = -inv(M) K(u) u (MFEM form)
*/
class ConductionOperator : public TimeDependentOperator
class ConductionOperator : public TimeDependentOperator, public ARKStepODE
{
FiniteElementSpace &fespace;
Array<int> ess_tdof_list; // this list remains empty for pure Neumann b.c.
@@ -81,50 +80,90 @@ class ConductionOperator : public TimeDependentOperator
mutable Vector z; // auxiliary vector
const bool use_mass_form;
public:
ConductionOperator(FiniteElementSpace &f, const real_t alpha,
const real_t kappa, const Vector &u,
const Type &ode_expression_type);
const bool use_mass_form);
// Compute K(u_n) for use as an approximation in - K(u) u
void SetConductionTensor(const Vector &u);
/** Compute G(u, t) as defined in the IMPLICIT expression form of the ODE
operator, i.e., @a v = - K(u_n) @a u. Note that K(u_n) is an
approximation to K(u). */
void ExplicitMult(const Vector &u, Vector &v) const override;
// ********* methods for MFEM native time integrators *********
/** Solve for k in F(u, k, t) = G(u, t) for either EXPLICIT or IMPLICIT
expression forms of the ODE operator, i.e., @a k = - inv(M) K(u_n) @a u.
/** Solve for k in F(u, k, t) = G(u, t), i.e., @a k = - inv(M) K(u_n) @a u.
Note that K(u_n) is an approximation to K(u). */
void Mult(const Vector &u, Vector &k) const override;
/** Solve for k in F(u + gam*k, k, t) = G(u + gam*k, t) for either EXPLICIT
or IMPLICIT expression forms of the ODE operator, i.e.,
[ M + @a gam K(u_n) ] @a k = - K(u_n) @a u . Note that K(u_n) is an
approximation to K(u). */
/** Solve for k in F(u + gam*k, k, t) = G(u + gam*k, t), i.e.,
[ M + @a gam K(u_n) ] @a k = - K(u_n) @a u .
Note that K(u_n) is an approximation to K(u). */
void ImplicitSolve(const real_t gam, const Vector &u, Vector &k) override;
/** Setup to solve for dk in [dF/dk + gam*dF/du - gam*dG/du] dk = G - F for
either EXPLICIT or IMPLICIT expression forms of the ODE operator, i.e.,
[M - @a gam Jf(u)] dk = G - F, where Jf(u) is an approximation of the
Jacobian of -K(u) u. The approximation chosen here is Jf(u) = -K(u_n). */
int SUNImplicitSetup(const Vector &u, const Vector &fu, int jok, int *jcur,
real_t gam) override;
// ********* methods for ARKODE time integrators *********
// TODO: add comments
int ARKSize() const override;
// TODO: add comments
bool ARKInMassForm() const override;
// TODO: add comments
void ARKEvaluateRHS(const Vector &u, const real_t t, Vector &result) const override;
// TODO: add comments
int ARKImplicitSetup(const Vector &u, const real_t t, const Vector &fu,
int jok, int *jcur, real_t gam) override;
/** Solve for @a dk in the system in SUNImplicitSetup to the given tolerance,
with the residual @a r providing either
1. @a r = G - F = inv(M) f(u) - k (EXPLICIT expression form)
1. @a r = G - F = f(u) - M k (IMPLICIT expression form)
1. @a r = G - F = inv(M) f(u) - k (MFEM form)
1. @a r = G - F = f(u) - M k (mass form)
*/
int SUNImplicitSolve(const Vector &r, Vector &dk, real_t tol) override;
int ARKImplicitSolve(const Vector &r, Vector &dk, real_t tol) override;
int SUNMassSetup() override;
int ARKMassSetup(const real_t t) override;
int SUNMassSolve(const Vector &b, Vector &x, real_t tol) override;
int ARKMassSolve(const Vector &b, Vector &x, real_t tol) override;
int SUNMassMult(const Vector &x, Vector &v) override;
int ARKMassMult(const Vector &x, Vector &v) override;
// ********* methods for CVODE time integrators *********
// note these methods merely call the corresponding ARKStepODE methods until
// the CVODESolver is refactored to use specialized interface like ARKStepODE
/** Setup to solve for dk in [dF/dk + gam*dF/du - gam*dG/du] dk = G - F, i.e.,
[M - @a gam Jf(u)] dk = G - F, where Jf(u) is an approximation of the
Jacobian of -K(u) u. The approximation chosen here is Jf(u) = -K(u_n). */
int SUNImplicitSetup(const Vector &u, const Vector &fu, int jok, int *jcur,
real_t gam) override
{
return ARKImplicitSetup(u, 0.0, fu, jok, jcur, gam); // the ODE is autonomous
}
/** Solve for @a dk in the system in SUNImplicitSetup to the given tolerance,
with the residual @a r providing @a r = G - F = inv(M) f(u) - k. */
int SUNImplicitSolve(const Vector &r, Vector &dk, real_t tol) override
{
return ARKImplicitSolve(r, dk, tol);
}
int SUNMassSetup() override
{
return ARKMassSetup(0.0); // the ODE is autonomous
}
int SUNMassSolve(const Vector &b, Vector &x, real_t tol) override
{
return ARKMassSolve(b, x, tol);
}
int SUNMassMult(const Vector &x, Vector &v) override
{
return ARKMassMult(x, v);
}
};
real_t InitialTemperature(const Vector &x)
@@ -245,16 +284,7 @@ int main(int argc, char *argv[])
u_gf.GetTrueDofs(u);
// 6. Initialize the conduction ODE operator and the visualization.
ConductionOperator::Type ode_expression_type;
if (use_mass_solver)
{
ode_expression_type = ConductionOperator::Type::IMPLICIT;
}
else
{
ode_expression_type = ConductionOperator::Type::EXPLICIT;
}
ConductionOperator oper(fespace, alpha, kappa, u, ode_expression_type);
ConductionOperator oper(fespace, alpha, kappa, u, use_mass_solver);
u_gf.SetFromTrueDofs(u);
{
@@ -352,7 +382,7 @@ int main(int argc, char *argv[])
}
std::unique_ptr<ARKStepSolver> arkode(
new ARKStepSolver(arkode_solver_type));
arkode->Init(oper);
arkode->Init(&oper);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
if (ode_solver_type == 11 || ode_solver_type == 14)
@@ -445,9 +475,10 @@ int main(int argc, char *argv[])
ConductionOperator::ConductionOperator(FiniteElementSpace &fes,
const real_t alpha, const real_t kappa,
const Vector &u,
const Type &ode_expression_type)
: TimeDependentOperator(fes.GetTrueVSize(), 0.0, ode_expression_type),
fespace(fes), M(&fespace), alpha(alpha), kappa(kappa), z(height)
const bool use_mass_form)
: TimeDependentOperator(fes.GetTrueVSize(), 0.0),
fespace(fes), M(&fespace), alpha(alpha), kappa(kappa), z(height),
use_mass_form(use_mass_form)
{
// specify a relative tolerance for all solves with MFEM integrators
const real_t rel_tol = 1e-8;
@@ -474,6 +505,16 @@ ConductionOperator::ConductionOperator(FiniteElementSpace &fes,
SetConductionTensor(u);
}
int ConductionOperator::ARKSize() const
{
return z.Size();
}
bool ConductionOperator::ARKInMassForm() const
{
return use_mass_form;
}
void ConductionOperator::SetConductionTensor(const Vector &u)
{
// Compute K(u_n).
@@ -491,17 +532,27 @@ void ConductionOperator::SetConductionTensor(const Vector &u)
K->FormSystemMatrix(ess_tdof_list, Kmat);
}
void ConductionOperator::ExplicitMult(const Vector &u, Vector &v) const
void ConductionOperator::ARKEvaluateRHS(const Vector &u, const real_t t,
Vector &result) const
{
// Compute - K(u_n) u.
Kmat.Mult(u, v);
v.Neg();
if (use_mass_form) // compute -K(u_n) u.
{
Kmat.Mult(u, result);
result.Neg();
}
else // compute -inv(M) K(u_n) u
{
Kmat.Mult(u, z);
z.Neg();
M_solver.Mult(z, result);
}
}
void ConductionOperator::Mult(const Vector &u, Vector &k) const
{
// Compute - inv(M) K(u_n) u.
ExplicitMult(u, z);
Kmat.Mult(u, z);
z.Neg();
M_solver.Mult(z, k);
}
@@ -509,14 +560,16 @@ void ConductionOperator::ImplicitSolve(const real_t gam, const Vector &u,
Vector &k)
{
// Solve for k in M k = - K(u_n) [u + gam*k].
ExplicitMult(u, z);
Kmat.Mult(u, z);
z.Neg();
T = std::unique_ptr<SparseMatrix>(Add(1.0, Mmat, gam, Kmat));
T_solver.SetOperator(*T);
T_solver.Mult(z, k);
}
int ConductionOperator::SUNImplicitSetup(const Vector &u, const Vector &fu,
int jok, int *jcur, real_t gam)
int ConductionOperator::ARKImplicitSetup(const Vector &u, const real_t t,
const Vector &fu, int jok, int *jcur,
real_t gam)
{
// Compute T = M + gamma K(u_n).
T = std::unique_ptr<SparseMatrix>(Add(1.0, Mmat, gam, Kmat));
@@ -525,22 +578,22 @@ int ConductionOperator::SUNImplicitSetup(const Vector &u, const Vector &fu,
return SUN_SUCCESS;
}
int ConductionOperator::SUNImplicitSolve(const Vector &r, Vector &dk,
int ConductionOperator::ARKImplicitSolve(const Vector &r, Vector &dk,
real_t tol)
{
// Solve the system [M + gamma K(u_n)] dk = - K(u_n) u - M k.
// What value r is providing depends on the ODE expression form:
// EXPLICIT form: r = -inv(M) K(u_n) u - k
// IMPLICIT form: r = -K(u_n) u - M k
// MFEM form: r = -inv(M) K(u_n) u - k
// mass form: r = -K(u_n) u - M k
T_solver.SetRelTol(tol);
if (isExplicit())
if (use_mass_form)
{
Mmat.Mult(r, z);
T_solver.Mult(z, dk);
T_solver.Mult(r, dk);
}
else
{
T_solver.Mult(r, dk);
Mmat.Mult(r, z);
T_solver.Mult(z, dk);
}
if (T_solver.GetConverged())
{
@@ -552,13 +605,13 @@ int ConductionOperator::SUNImplicitSolve(const Vector &r, Vector &dk,
}
}
int ConductionOperator::SUNMassSetup()
int ConductionOperator::ARKMassSetup(const real_t t)
{
// Do nothing b/c mass solver was setup in constructor.
return SUN_SUCCESS;
}
int ConductionOperator::SUNMassSolve(const Vector &b, Vector &x, real_t tol)
int ConductionOperator::ARKMassSolve(const Vector &b, Vector &x, real_t tol)
{
// Solve the system M x = b.
M_solver.SetRelTol(tol);
@@ -573,7 +626,7 @@ int ConductionOperator::SUNMassSolve(const Vector &b, Vector &x, real_t tol)
}
}
int ConductionOperator::SUNMassMult(const Vector &x, Vector &v)
int ConductionOperator::ARKMassMult(const Vector &x, Vector &v)
{
// Compute M x.
Mmat.Mult(x, v);
+119 -66
View File
@@ -29,10 +29,8 @@
//
// The example demonstrates the use of nonlinear operators (the
// class ConductionOperator defining C(u)), as well as their
// implicit time integration. Note that implementing the method
// ConductionOperator::ImplicitSolve is the only requirement for
// high-order implicit (SDIRK) time integration. By default, this
// example uses the SUNDIALS ODE solvers from CVODE and ARKODE.
// implicit time integration. By default, this example uses the
// SUNDIALS ODE solvers from CVODE and ARKODE.
//
// We recommend viewing examples 2, 9 and 10 before viewing this
// example.
@@ -52,15 +50,16 @@ using namespace mfem;
* and K(u) is the diffusion operator with diffusivity depending on u:
* (\kappa + \alpha u).
*
* Class ConductionOperatorOperator represents the above ODE operator in the
* general form F(u, k, t) = G(u, t) where either
* Class ConductionOperator represents the above ODE operator as a
* TimeDependentOperator for use with native MFEM integrators and CVODE
* integrators, i.e., F(u, k, t) = G(u, t) with F(u, du/dt, t) = du/dt and
* G(u, t) = -K(u) u
*
* 1. F(u, du/dt, t) = du/dt (ODE is expressed in EXPLICIT form)
* G(u, t) = - inv(M) K(u) u
* 2. F(u, du/dt, t) = M du/dt (ODE is expressed in IMPLICIT form)
* G(u, t) = - K(u) u
* Class ConductionOperator represents the above ODE operator as an
* ARKStepODE for use with ARKODE integrators, i.e., either M du/dt = -K(u) u
* (mass form) or du/dt = -inv(M) K(u) u (MFEM form)
*/
class ConductionOperator : public TimeDependentOperator
class ConductionOperator : public TimeDependentOperator, public ARKStepODE
{
ParFiniteElementSpace &fespace;
Array<int> ess_tdof_list; // this list remains empty for pure Neumann b.c.
@@ -82,50 +81,90 @@ class ConductionOperator : public TimeDependentOperator
mutable Vector z; // auxiliary vector
const bool use_mass_form;
public:
ConductionOperator(ParFiniteElementSpace &f, const real_t alpha,
const real_t kappa, const Vector &u,
const Type &ode_expression_type);
const bool use_mass_form);
// Compute K(u_n) for use as an approximation in - K(u) u
void SetConductionTensor(const Vector &u);
/** Compute G(u, t) as defined in the IMPLICIT expression form of the ODE
operator, i.e., @a v = - K(u_n) @a u. Note that K(u_n) is an
approximation to K(u). */
void ExplicitMult(const Vector &u, Vector &v) const override;
// ********* methods for MFEM native time integrators *********
/** Solve for k in F(u, k, t) = G(u, t) for either EXPLICIT or IMPLICIT
expression forms of the ODE operator, i.e., @a k = - inv(M) K(u_n) @a u.
/** Solve for k in F(u, k, t) = G(u, t), i.e., @a k = - inv(M) K(u_n) @a u.
Note that K(u_n) is an approximation to K(u). */
void Mult(const Vector &u, Vector &k) const override;
/** Solve for k in F(u + gam*k, k, t) = G(u + gam*k, t) for either EXPLICIT
or IMPLICIT expression forms of the ODE operator, i.e.,
[ M + @a gam K(u_n) ] @a k = - K(u_n) @a u . Note that K(u_n) is an
approximation to K(u). */
/** Solve for k in F(u + gam*k, k, t) = G(u + gam*k, t), i.e.,
[ M + @a gam K(u_n) ] @a k = - K(u_n) @a u .
Note that K(u_n) is an approximation to K(u). */
void ImplicitSolve(const real_t gam, const Vector &u, Vector &k) override;
/** Setup to solve for dk in [dF/dk + gam*dF/du - gam*dG/du] dk = G - F for
either EXPLICIT or IMPLICIT expression forms of the ODE operator, i.e.,
[M - @a gam Jf(u)] dk = G - F, where Jf(u) is an approximation of the
Jacobian of -K(u) u. The approximation chosen here is Jf(u) = -K(u_n). */
int SUNImplicitSetup(const Vector &u, const Vector &fu, int jok, int *jcur,
real_t gam) override;
// ********* methods for ARKODE time integrators *********
// TODO: add comments
int ARKSize() const override;
// TODO: add comments
bool ARKInMassForm() const override;
// TODO: add comments
void ARKEvaluateRHS(const Vector &u, const real_t t, Vector &result) const override;
// TODO: add comments
int ARKImplicitSetup(const Vector &u, const real_t t, const Vector &fu,
int jok, int *jcur, real_t gam) override;
/** Solve for @a dk in the system in SUNImplicitSetup to the given tolerance,
with the residual @a r providing either
1. @a r = G - F = inv(M) f(u) - k (EXPLICIT expression form)
1. @a r = G - F = f(u) - M k (IMPLICIT expression form)
1. @a r = G - F = inv(M) f(u) - k (MFEM form)
1. @a r = G - F = f(u) - M k (mass form)
*/
int SUNImplicitSolve(const Vector &r, Vector &dk, real_t tol) override;
int ARKImplicitSolve(const Vector &r, Vector &dk, real_t tol) override;
int SUNMassSetup() override;
int ARKMassSetup(const real_t t) override;
int SUNMassSolve(const Vector &b, Vector &x, real_t tol) override;
int ARKMassSolve(const Vector &b, Vector &x, real_t tol) override;
int SUNMassMult(const Vector &x, Vector &v) override;
int ARKMassMult(const Vector &x, Vector &v) override;
// ********* methods for CVODE time integrators *********
// note these methods merely call the corresponding ARKStepODE methods until
// the CVODESolver is refactored to use specialized interface like ARKStepODE
/** Setup to solve for dk in [dF/dk + gam*dF/du - gam*dG/du] dk = G - F, i.e.,
[M - @a gam Jf(u)] dk = G - F, where Jf(u) is an approximation of the
Jacobian of -K(u) u. The approximation chosen here is Jf(u) = -K(u_n). */
int SUNImplicitSetup(const Vector &u, const Vector &fu, int jok, int *jcur,
real_t gam) override
{
return ARKImplicitSetup(u, 0.0, fu, jok, jcur, gam); // the ODE is autonomous
}
/** Solve for @a dk in the system in SUNImplicitSetup to the given tolerance,
with the residual @a r providing @a r = G - F = inv(M) f(u) - k. */
int SUNImplicitSolve(const Vector &r, Vector &dk, real_t tol) override
{
return ARKImplicitSolve(r, dk, tol);
}
int SUNMassSetup() override
{
return ARKMassSetup(0.0); // the ODE is autonomous
}
int SUNMassSolve(const Vector &b, Vector &x, real_t tol) override
{
return ARKMassSolve(b, x, tol);
}
int SUNMassMult(const Vector &x, Vector &v) override
{
return ARKMassMult(x, v);
}
};
real_t InitialTemperature(const Vector &x)
@@ -273,16 +312,7 @@ int main(int argc, char *argv[])
u_gf.GetTrueDofs(u);
// 8. Initialize the conduction ODE operator and the visualization.
ConductionOperator::Type ode_expression_type;
if (use_mass_solver)
{
ode_expression_type = ConductionOperator::Type::IMPLICIT;
}
else
{
ode_expression_type = ConductionOperator::Type::EXPLICIT;
}
ConductionOperator oper(fespace, alpha, kappa, u, ode_expression_type);
ConductionOperator oper(fespace, alpha, kappa, u, use_mass_solver);
u_gf.SetFromTrueDofs(u);
{
@@ -394,7 +424,7 @@ int main(int argc, char *argv[])
}
std::unique_ptr<ARKStepSolver> arkode(
new ARKStepSolver(MPI_COMM_WORLD, arkode_solver_type));
arkode->Init(oper);
arkode->Init(&oper);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
if (ode_solver_type == 11 || ode_solver_type == 14)
@@ -497,10 +527,11 @@ int main(int argc, char *argv[])
ConductionOperator::ConductionOperator(ParFiniteElementSpace &fes,
const real_t alpha, const real_t kappa,
const Vector &u,
const Type &ode_expression_type)
: TimeDependentOperator(fes.GetTrueVSize(), 0.0, ode_expression_type),
const bool use_mass_form)
: TimeDependentOperator(fes.GetTrueVSize(), 0.0),
fespace(fes), M(&fespace), alpha(alpha), kappa(kappa),
M_solver(fes.GetComm()), T_solver(fes.GetComm()), z(height)
M_solver(fes.GetComm()), T_solver(fes.GetComm()), z(height),
use_mass_form(use_mass_form)
{
// specify a relative tolerance for all solves with MFEM integrators
const real_t rel_tol = 1e-8;
@@ -528,6 +559,16 @@ ConductionOperator::ConductionOperator(ParFiniteElementSpace &fes,
SetConductionTensor(u);
}
int ConductionOperator::ARKSize() const
{
return z.Size();
}
bool ConductionOperator::ARKInMassForm() const
{
return use_mass_form;
}
void ConductionOperator::SetConductionTensor(const Vector &u)
{
// Compute K(u_n).
@@ -545,17 +586,27 @@ void ConductionOperator::SetConductionTensor(const Vector &u)
K->FormSystemMatrix(ess_tdof_list, Kmat);
}
void ConductionOperator::ExplicitMult(const Vector &u, Vector &v) const
void ConductionOperator::ARKEvaluateRHS(const Vector &u, const real_t t,
Vector &result) const
{
// Compute - K(u_n) u.
Kmat.Mult(u, v);
v.Neg();
if (use_mass_form) // compute -K(u_n) u.
{
Kmat.Mult(u, result);
result.Neg();
}
else // compute -inv(M) K(u_n) u
{
Kmat.Mult(u, z);
z.Neg();
M_solver.Mult(z, result);
}
}
void ConductionOperator::Mult(const Vector &u, Vector &k) const
{
// Compute - inv(M) K(u_n) u.
ExplicitMult(u, z);
Kmat.Mult(u, z);
z.Neg();
M_solver.Mult(z, k);
}
@@ -563,14 +614,16 @@ void ConductionOperator::ImplicitSolve(const real_t gam, const Vector &u,
Vector &k)
{
// Solve for k in M k = - K(u_n) [u + gam*k].
ExplicitMult(u, z);
Kmat.Mult(u, z);
z.Neg();
T = std::unique_ptr<HypreParMatrix>(Add(1.0, Mmat, gam, Kmat));
T_solver.SetOperator(*T);
T_solver.Mult(z, k);
}
int ConductionOperator::SUNImplicitSetup(const Vector &u, const Vector &fu,
int jok, int *jcur, real_t gam)
int ConductionOperator::ARKImplicitSetup(const Vector &u, const real_t t,
const Vector &fu, int jok, int *jcur,
real_t gam)
{
// Compute T = M + gamma K(u_n).
T = std::unique_ptr<HypreParMatrix>(Add(1.0, Mmat, gam, Kmat));
@@ -579,22 +632,22 @@ int ConductionOperator::SUNImplicitSetup(const Vector &u, const Vector &fu,
return SUN_SUCCESS;
}
int ConductionOperator::SUNImplicitSolve(const Vector &r, Vector &dk,
int ConductionOperator::ARKImplicitSolve(const Vector &r, Vector &dk,
real_t tol)
{
// Solve the system [M + gamma K(u_n)] dk = - K(u_n) u - M k.
// What value r is providing depends on the ODE expression form:
// EXPLICIT form: r = -inv(M) K(u_n) u - k
// IMPLICIT form: r = -K(u_n) u - M k
// MFEM form: r = -inv(M) K(u_n) u - k
// mass form: r = -K(u_n) u - M k
T_solver.SetRelTol(tol);
if (isExplicit())
if (use_mass_form)
{
Mmat.Mult(r, z);
T_solver.Mult(z, dk);
T_solver.Mult(r, dk);
}
else
{
T_solver.Mult(r, dk);
Mmat.Mult(r, z);
T_solver.Mult(z, dk);
}
if (T_solver.GetConverged())
{
@@ -606,13 +659,13 @@ int ConductionOperator::SUNImplicitSolve(const Vector &r, Vector &dk,
}
}
int ConductionOperator::SUNMassSetup()
int ConductionOperator::ARKMassSetup(const real_t t)
{
// Do nothing b/c mass solver was setup in constructor.
return SUN_SUCCESS;
}
int ConductionOperator::SUNMassSolve(const Vector &b, Vector &x, real_t tol)
int ConductionOperator::ARKMassSolve(const Vector &b, Vector &x, real_t tol)
{
// Solve the system M x = b.
M_solver.SetRelTol(tol);
@@ -627,7 +680,7 @@ int ConductionOperator::SUNMassSolve(const Vector &b, Vector &x, real_t tol)
}
}
int ConductionOperator::SUNMassMult(const Vector &x, Vector &v)
int ConductionOperator::ARKMassMult(const Vector &x, Vector &v)
{
// Compute M x.
Mmat.Mult(x, v);
+21 -3
View File
@@ -119,7 +119,7 @@ public:
and advection matrices, and b describes the flow on the boundary. This can
be written as a general ODE, du/dt = M^{-1} (K u + b), and this class is
used to evaluate the right-hand side. */
class FE_Evolution : public TimeDependentOperator
class FE_Evolution : public TimeDependentOperator, public ARKStepODE
{
private:
BilinearForm &M, &K;
@@ -133,9 +133,14 @@ private:
public:
FE_Evolution(BilinearForm &M_, BilinearForm &K_, const Vector &b_);
// TimeDependentOperator methods for MFEM native and CVODE time integrators
virtual void Mult(const Vector &x, Vector &y) const;
virtual void ImplicitSolve(const double dt, const Vector &x, Vector &k);
// ARKStepODE methods for ARKODE time integrators
int ARKSize() const override;
void ARKEvaluateRHS(const Vector &u, const real_t t, Vector& result) const override;
virtual ~FE_Evolution();
};
@@ -404,14 +409,14 @@ int main(int argc, char *argv[])
ode_solver = cvode; break;
case 8:
arkode = new ARKStepSolver(ARKStepSolver::EXPLICIT);
arkode->Init(adv);
arkode->Init(&adv);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
arkode->SetOrder(4);
ode_solver = arkode; break;
case 9:
arkode = new ARKStepSolver(ARKStepSolver::EXPLICIT);
arkode->Init(adv);
arkode->Init(&adv);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
arkode->SetERKTableNum(ARKODE_FEHLBERG_13_7_8);
@@ -520,6 +525,19 @@ void FE_Evolution::ImplicitSolve(const double dt, const Vector &x, Vector &k)
dg_solver->Mult(z, k);
}
int FE_Evolution::ARKSize() const
{
return z.Size();
}
void FE_Evolution::ARKEvaluateRHS(const Vector &u, const real_t t, Vector &result) const
{
// y = M^{-1} (K x + b)
K.Mult(u, z);
z += b;
M_solver.Mult(z, result);
}
FE_Evolution::~FE_Evolution()
{
delete M_prec;
+20 -2
View File
@@ -206,7 +206,7 @@ public:
and advection matrices, and b describes the flow on the boundary. This can
be written as a general ODE, du/dt = M^{-1} (K u + b), and this class is
used to evaluate the right-hand side. */
class FE_Evolution : public TimeDependentOperator
class FE_Evolution : public TimeDependentOperator, public ARKStepODE
{
private:
OperatorHandle M, K;
@@ -221,9 +221,14 @@ public:
FE_Evolution(ParBilinearForm &M_, ParBilinearForm &K_, const Vector &b_,
PrecType prec_type);
// TimeDependentOperator methods for MFEM native and CVODE time integrators
virtual void Mult(const Vector &x, Vector &y) const;
virtual void ImplicitSolve(const double dt, const Vector &x, Vector &k);
// ARKStepODE methods for ARKODE time integrators
int ARKSize() const override;
void ARKEvaluateRHS(const Vector &u, const real_t t, Vector& result) const override;
virtual ~FE_Evolution();
};
@@ -575,7 +580,7 @@ int main(int argc, char *argv[])
case 8:
case 9:
arkode = new ARKStepSolver(MPI_COMM_WORLD, ARKStepSolver::EXPLICIT);
arkode->Init(adv);
arkode->Init(&adv);
arkode->SetSStolerances(reltol, abstol);
arkode->SetMaxStep(dt);
if (ode_solver_type == 9)
@@ -743,6 +748,19 @@ void FE_Evolution::Mult(const Vector &x, Vector &y) const
M_solver.Mult(z, y);
}
int FE_Evolution::ARKSize() const
{
return z.Size();
}
void FE_Evolution::ARKEvaluateRHS(const Vector &u, const real_t t, Vector &result) const
{
// y = M^{-1} (K x + b)
K->Mult(u, z);
z += b;
M_solver.Mult(z, result);
}
FE_Evolution::~FE_Evolution()
{
delete M_prec;
+27 -48
View File
@@ -128,46 +128,32 @@ set(SRCS
normal_deriv_restriction.cpp
staticcond.cpp
tmop.cpp
tmop/pa.cpp
tmop/assemble/diag2_limit.cpp
tmop/assemble/diag2.cpp
tmop/assemble/grad2_limit.cpp
tmop/assemble/grad2.cpp
tmop/assemble/diag3_limit.cpp
tmop/assemble/diag3.cpp
tmop/assemble/grad3_limit.cpp
tmop/assemble/grad3.cpp
tmop/metrics/001.cpp
tmop/metrics/002.cpp
tmop/metrics/007.cpp
tmop/metrics/056.cpp
tmop/metrics/077.cpp
tmop/metrics/080.cpp
tmop/metrics/094.cpp
tmop/metrics/302.cpp
tmop/metrics/303.cpp
tmop/metrics/315.cpp
tmop/metrics/318.cpp
tmop/metrics/321.cpp
tmop/metrics/332.cpp
tmop/metrics/338.cpp
tmop/mult/grad2_limit.cpp
tmop/mult/grad2.cpp
tmop/mult/mult2_limit.cpp
tmop/mult/mult2.cpp
tmop/mult/grad3_limit.cpp
tmop/mult/grad3.cpp
tmop/mult/mult3_limit.cpp
tmop/mult/mult3.cpp
tmop/tools/det2_jpr.cpp
tmop/tools/det3_jpr.cpp
tmop/tools/discrete.cpp
tmop/tools/energy2_limit.cpp
tmop/tools/energy2.cpp
tmop/tools/energy3_limit.cpp
tmop/tools/energy3.cpp
tmop/tools/target2.cpp
tmop/tools/target3.cpp
tmop/tmop_pa.cpp
tmop/tmop_pa_da3.cpp
tmop/tmop_pa_h2d.cpp
tmop/tmop_pa_h2d_c0.cpp
tmop/tmop_pa_h2m.cpp
tmop/tmop_pa_h2m_c0.cpp
tmop/tmop_pa_h2s.cpp
tmop/tmop_pa_h2s_c0.cpp
tmop/tmop_pa_h3d.cpp
tmop/tmop_pa_h3d_c0.cpp
tmop/tmop_pa_h3m.cpp
tmop/tmop_pa_h3m_c0.cpp
tmop/tmop_pa_h3s.cpp
tmop/tmop_pa_h3s_c0.cpp
tmop/tmop_pa_jp2.cpp
tmop/tmop_pa_jp3.cpp
tmop/tmop_pa_p2.cpp
tmop/tmop_pa_p2_c0.cpp
tmop/tmop_pa_p3.cpp
tmop/tmop_pa_p3_c0.cpp
tmop/tmop_pa_tc2.cpp
tmop/tmop_pa_tc3.cpp
tmop/tmop_pa_w2.cpp
tmop/tmop_pa_w2_c0.cpp
tmop/tmop_pa_w3.cpp
tmop/tmop_pa_w3_c0.cpp
tmop_tools.cpp
tmop_amr.cpp
gslib.cpp
@@ -196,8 +182,6 @@ set(HDRS
integ/bilininteg_hdiv_kernels.hpp
integ/bilininteg_hcurlhdiv_kernels.hpp
integ/bilininteg_mass_kernels.hpp
integ/bilininteg_vecdiffusion_pa.hpp
integ/bilininteg_vecmass_pa.hpp
coefficient.hpp
complex_fem.hpp
convergence.hpp
@@ -295,12 +279,7 @@ set(HDRS
tfespace.hpp
tintrules.hpp
tmop.hpp
tmop/pa.hpp
tmop/assemble/grad2.hpp
tmop/assemble/grad2.hpp
tmop/mult/mult2.hpp
tmop/mult/mult3.hpp
tmop/tools/energy2.hpp
tmop/tmop_pa.hpp
tmop_tools.hpp
tmop_amr.hpp
gslib.hpp
+1
View File
@@ -3066,6 +3066,7 @@ void VectorDiffusionIntegrator::AssembleElementMatrix(
for (int i = 0; i < ir -> GetNPoints(); i++)
{
const IntegrationPoint &ip = ir->IntPoint(i);
el.CalcDShape(ip, dshape);
+41 -44
View File
@@ -2596,40 +2596,41 @@ public:
by scalar FE through standard transformation. */
class VectorMassIntegrator: public BilinearFormIntegrator
{
int vdim = -1, Q_order = 0;
private:
int vdim;
Vector shape, te_shape, vec;
DenseMatrix partelmat;
DenseMatrix mcoeff;
int Q_order;
protected:
Coefficient *Q = nullptr;
VectorCoefficient *VQ = nullptr;
MatrixCoefficient *MQ = nullptr;
Coefficient *Q;
VectorCoefficient *VQ;
MatrixCoefficient *MQ;
// PA extension
Vector pa_data;
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int ne, dim, dofs1D, quad1D, coeff_vdim;
Vector pa_data;
int dim, ne, nq, dofs1D, quad1D;
public:
/// Construct an integrator with coefficient 1.0
VectorMassIntegrator() = default;
VectorMassIntegrator()
: vdim(-1), Q_order(0), Q(NULL), VQ(NULL), MQ(NULL) { }
/** Construct an integrator with scalar coefficient q. If possible, save
memory by using a scalar integrator since the resulting matrix is block
diagonal with the same diagonal block repeated. */
VectorMassIntegrator(Coefficient &q, int qo = 0): Q_order(qo), Q(&q) { }
VectorMassIntegrator(Coefficient &q, const IntegrationRule *ir):
BilinearFormIntegrator(ir), Q(&q) { }
VectorMassIntegrator(Coefficient &q, int qo = 0)
: vdim(-1), Q_order(qo), Q(&q), VQ(NULL), MQ(NULL) { }
VectorMassIntegrator(Coefficient &q, const IntegrationRule *ir)
: BilinearFormIntegrator(ir), vdim(-1), Q_order(0), Q(&q), VQ(NULL),
MQ(NULL) { }
/// Construct an integrator with diagonal coefficient q
VectorMassIntegrator(VectorCoefficient &q, int qo = 0):
vdim(q.GetVDim()), Q_order(qo), VQ(&q) { }
VectorMassIntegrator(VectorCoefficient &q, int qo = 0)
: vdim(q.GetVDim()), Q_order(qo), Q(NULL), VQ(&q), MQ(NULL) { }
/// Construct an integrator with matrix coefficient q
VectorMassIntegrator(MatrixCoefficient &q, int qo = 0):
vdim(q.GetVDim()), Q_order(qo), MQ(&q) { }
VectorMassIntegrator(MatrixCoefficient &q, int qo = 0)
: vdim(q.GetVDim()), Q_order(qo), Q(NULL), VQ(NULL), MQ(&q) { }
int GetVDim() const { return vdim; }
void SetVDim(int vdim_) { vdim = vdim_; }
@@ -2641,7 +2642,6 @@ public:
const FiniteElement &test_fe,
ElementTransformation &Trans,
DenseMatrix &elmat) override;
using BilinearFormIntegrator::AssemblePA;
void AssemblePA(const FiniteElementSpace &fes) override;
void AssembleMF(const FiniteElementSpace &fes) override;
@@ -2650,15 +2650,6 @@ public:
void AddMultPA(const Vector &x, Vector &y) const override;
void AddMultMF(const Vector &x, Vector &y) const override;
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
using VectorMassAddMultPAType =
void(*)(const int, const int,
const Array<real_t>&, const Vector&,
const Vector&, Vector&, const int, const int);
MFEM_REGISTER_KERNELS(VectorMassAddMultPA,
VectorMassAddMultPAType,
(int, int, int));
};
@@ -3129,21 +3120,23 @@ public:
to be the spatial dimension (i.e. 2-dimension or 3-dimension). */
class VectorDiffusionIntegrator : public BilinearFormIntegrator
{
int vdim = -1;
DenseMatrix dshape, dshapedxt, pelmat;
DenseMatrix mcoeff;
Vector vcoeff;
protected:
Coefficient *Q = nullptr;
VectorCoefficient *VQ = nullptr;
MatrixCoefficient *MQ = nullptr;
Coefficient *Q = NULL;
VectorCoefficient *VQ = NULL;
MatrixCoefficient *MQ = NULL;
// PA extension
const DofToQuad *maps; ///< Not owned
const GeometricFactors *geom; ///< Not owned
int ne, dim, sdim, dofs1D, quad1D, coeff_vdim;
int dim, sdim, ne, dofs1D, quad1D;
Vector pa_data;
private:
DenseMatrix dshape, dshapedxt, pelmat;
int vdim = -1;
DenseMatrix mcoeff;
Vector vcoeff;
public:
VectorDiffusionIntegrator(const IntegrationRule *ir = nullptr);
@@ -3196,7 +3189,6 @@ public:
void AssembleElementVector(const FiniteElement &el,
ElementTransformation &Tr,
const Vector &elfun, Vector &elvect) override;
using BilinearFormIntegrator::AssemblePA;
void AssemblePA(const FiniteElementSpace &fes) override;
void AssembleMF(const FiniteElementSpace &fes) override;
@@ -3206,11 +3198,13 @@ public:
void AddMultMF(const Vector &x, Vector &y) const override;
bool SupportsCeed() const override { return DeviceCanUseCeed(); }
/// arguments: ne, coeff_vdim, B, G, pa_data, x, y, d1d, q1d, vdim
using ApplyKernelType = void (*)(const int, const int,
const Array<real_t> &, const Array<real_t> &,
const Vector &, const Vector &, Vector &,
const int, const int, const int);
/// arguments: ne, B, G, Bt, Gt, pa_data, x, y, d1d, q1d, vdim
using ApplyKernelType = void (*)(const int, const Array<real_t> &,
const Array<real_t> &,
const Array<real_t> &,
const Array<real_t> &, const Vector &,
const Vector &, Vector &, const int,
const int, const int);
/// arguments: dim, vdim, d1d, q1d
MFEM_REGISTER_KERNELS(ApplyPAKernels, ApplyKernelType, (int, int, int, int));
@@ -3221,7 +3215,10 @@ public:
ApplyPAKernels::Specialization<DIM, VDIM, D1D, Q1D>::Add();
}
// struct Kernels { Kernels(); };
struct Kernels
{
Kernels();
};
};
/** Integrator for the linear elasticity form:
-72
View File
@@ -1085,29 +1085,6 @@ void SumCoefficient::SetTime(real_t t)
this->Coefficient::SetTime(t);
}
void SumCoefficient::Project(QuadratureFunction &qf)
{
if (a == nullptr)
{
// qf = alpha*aConst + beta * b
const real_t d_alpha_a = aConst*alpha;
const real_t d_beta = beta;
b->Project(qf);
auto d_qf = qf.ReadWrite();
mfem::forall(qf.Size(), [=] MFEM_HOST_DEVICE (int i)
{
d_qf[i] = d_alpha_a + d_beta*d_qf[i];
});
}
else
{
a->Project(qf);
QuadratureFunction qf_b(*qf.GetSpace());
b->Project(qf_b);
add(alpha, qf, beta, qf_b, qf);
}
}
void ProductCoefficient::SetTime(real_t t)
{
if (a) { a->SetTime(t); }
@@ -1115,23 +1092,6 @@ void ProductCoefficient::SetTime(real_t t)
this->Coefficient::SetTime(t);
}
void ProductCoefficient::Project(QuadratureFunction &qf)
{
if (a == nullptr)
{
// qf = aConst * b
b->Project(qf);
qf *= aConst;
}
else
{
a->Project(qf);
QuadratureFunction qf_b(qf.GetSpace());
b->Project(qf_b);
qf *= qf_b;
}
}
void RatioCoefficient::SetTime(real_t t)
{
if (a) { a->SetTime(t); }
@@ -1139,38 +1099,6 @@ void RatioCoefficient::SetTime(real_t t)
this->Coefficient::SetTime(t);
}
void RatioCoefficient::Project(QuadratureFunction &qf)
{
if (b == nullptr)
{
if (a == nullptr)
{
qf = aConst / bConst;
}
else
{
a->Project(qf);
qf *= 1.0/bConst;
}
}
else
{
if (a == nullptr)
{
b->Project(qf);
qf.Reciprocal();
qf *= aConst;
}
else
{
a->Project(qf);
QuadratureFunction qf_b(qf.GetSpace());
b->Project(qf_b);
qf /= qf_b;
}
}
}
void PowerCoefficient::SetTime(real_t t)
{
if (a) { a->SetTime(t); }
-9
View File
@@ -1456,9 +1456,6 @@ public:
/// Set the time for internally stored coefficients
void SetTime(real_t t) override;
/// @copydoc Coefficient::Project(QuadratureFunction &)
void Project(QuadratureFunction &qf) override;
/// Reset the first term in the linear combination as a constant
void SetAConst(real_t A) { a = NULL; aConst = A; }
/// Return the first term in the linear combination
@@ -1640,9 +1637,6 @@ public:
/// Set the time for internally stored coefficients
void SetTime(real_t t) override;
/// @copydoc Coefficient::Project(QuadratureFunction &)
void Project(QuadratureFunction &qf) override;
/// Reset the first term in the product as a constant
void SetAConst(real_t A) { a = NULL; aConst = A; }
/// Return the first term in the product
@@ -1691,9 +1685,6 @@ public:
/// Set the time for internally stored coefficients
void SetTime(real_t t) override;
/// @copydoc Coefficient::Project(QuadratureFunction &)
void Project(QuadratureFunction &qf) override;
/// Reset the numerator in the ratio as a constant
void SetAConst(real_t A) { a = NULL; aConst = A; }
/// Return the numerator of the ratio
+8 -281
View File
@@ -11,15 +11,14 @@
#include "complex_fem.hpp"
#include "../general/forall.hpp"
#include "../general/text.hpp"
using namespace std;
namespace mfem
{
ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *f)
: Vector(2*(f->GetVSize())), fes(f), fec_owned(NULL)
ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *fes)
: Vector(2*(fes->GetVSize()))
{
UseDevice(true);
this->Vector::operator=(0.0);
@@ -29,88 +28,12 @@ ComplexGridFunction::ComplexGridFunction(FiniteElementSpace *f)
gfi = new GridFunction();
gfi->MakeRef(fes, *this, fes->GetVSize());
fes_sequence = fes->GetSequence();
}
ComplexGridFunction::ComplexGridFunction(Mesh *m, std::istream &input)
: Vector(), fes(NULL), fec_owned(NULL)
{
string buff;
// Grid functions are stored on the device
UseDevice(true);
input >> std::ws;
getline(input, buff); // 'ComplexGridFunction'
filter_dos(buff);
if (buff != "ComplexGridFunction")
{
MFEM_ABORT("unrecognized file header: " << buff);
}
fes = new FiniteElementSpace;
fec_owned = fes->Load(m, input);
skip_comment_lines(input, '#');
istream::int_type next_char = input.peek();
if (next_char == 'N') // First letter of "NURBS_patches"
{
getline(input, buff);
filter_dos(buff);
if (buff == "NURBS_patches")
{
MFEM_ABORT("NURBS not yet supported with ComplexGridFunction objects");
}
else
{
MFEM_ABORT("unknown section: " << buff);
}
}
else
{
Vector::Load(input, 2*fes->GetVSize());
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
if (fes->Nonconforming() &&
fes->GetMesh()->ncmesh->IsLegacyLoaded())
{
// LegacyNCReorder();
MFEM_ABORT("LegacyNCReorder not supported for "
"ComplexGridFunction objects");
}
}
gfr = new GridFunction();
gfr->MakeRef(fes, *this, 0);
gfi = new GridFunction();
gfi->MakeRef(fes, *this, fes->GetVSize());
fes_sequence = fes->GetSequence();
}
void ComplexGridFunction::Destroy()
{
delete gfr; delete gfi;
if (fec_owned)
{
delete fes;
delete fec_owned;
fec_owned = NULL;
}
}
void
ComplexGridFunction::Update()
{
if (fes->GetSequence() == fes_sequence)
{
return; // space and grid function are in sync, no-op
}
fes_sequence = fes->GetSequence();
FiniteElementSpace *fes = gfr->FESpace();
const int vsize = fes->GetVSize();
const Operator *T = fes->GetUpdateOperator();
@@ -161,17 +84,6 @@ ComplexGridFunction::Update()
}
}
int ComplexGridFunction::VectorDim() const
{
const FiniteElement *fe = fes->GetTypicalFE();
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
{
return fes->GetVDim();
}
return fes->GetVDim()*std::max(fes->GetMesh()->SpaceDimension(),
fe->GetRangeDim());
}
void
ComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff)
@@ -237,35 +149,6 @@ ComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
gfi->SyncAliasMemory(*this);
}
void ComplexGridFunction::Save(std::ostream &os) const
{
os << "ComplexGridFunction\n";
fes->Save(os);
os << '\n';
if (fes->GetOrdering() == Ordering::byNODES)
{
Vector::Print(os, 1);
}
else
{
Vector::Print(os, fes->GetVDim());
}
os.flush();
}
void ComplexGridFunction::Save(const char *fname, int precision) const
{
ofstream ofs(fname);
ofs.precision(precision);
Save(ofs);
}
std::ostream &operator<<(std::ostream &os, const ComplexGridFunction &sol)
{
sol.Save(os);
return os;
}
ComplexLinearForm::ComplexLinearForm(FiniteElementSpace *fes,
ComplexOperator::Convention convention)
@@ -771,8 +654,8 @@ SesquilinearForm::Update(FiniteElementSpace *nfes)
#ifdef MFEM_USE_MPI
ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pf)
: Vector(2*(pf->GetVSize())), pfes(pf), fec_owned(NULL)
ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pfes)
: Vector(2*(pfes->GetVSize()))
{
UseDevice(true);
this->Vector::operator=(0.0);
@@ -782,105 +665,12 @@ ParComplexGridFunction::ParComplexGridFunction(ParFiniteElementSpace *pf)
pgfi = new ParGridFunction();
pgfi->MakeRef(pfes, *this, pfes->GetVSize());
fes_sequence = pfes->GetSequence();
}
ParComplexGridFunction::ParComplexGridFunction(ParMesh *m, std::istream &input)
: Vector(), pfes(NULL), fec_owned(NULL)
{
string buff;
// Grid functions are stored on the device
UseDevice(true);
input >> std::ws;
getline(input, buff); // 'ParComplexGridFunction'
filter_dos(buff);
if (buff != "ParComplexGridFunction")
{
MFEM_ABORT("unrecognized file header: " << buff);
}
FiniteElementSpace *fes = new FiniteElementSpace;
fec_owned = fes->Load(m, input);
pfes = new ParFiniteElementSpace(m, fec_owned, fes->GetVDim(),
fes->GetOrdering());
delete fes;
skip_comment_lines(input, '#');
istream::int_type next_char = input.peek();
if (next_char == 'N') // First letter of "NURBS_patches"
{
getline(input, buff);
filter_dos(buff);
if (buff == "NURBS_patches")
{
MFEM_ABORT("NURBS not yet supported with ComplexGridFunction objects");
}
else
{
MFEM_ABORT("unknown section: " << buff);
}
}
else
{
int vsize = pfes->GetVSize();
Vector::Load(input, 2*vsize);
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
// if the mesh is a legacy (v1.1) NC mesh, it has old vertex ordering
if (pfes->Nonconforming() &&
pfes->GetMesh()->ncmesh->IsLegacyLoaded())
{
// LegacyNCReorder();
MFEM_ABORT("LegacyNCReorder not supported for "
"ComplexGridFunction objects");
}
}
pgfr = new ParGridFunction();
pgfr->MakeRef(pfes, *this, 0);
pgfi = new ParGridFunction();
pgfi->MakeRef(pfes, *this, pfes->GetVSize());
fes_sequence = pfes->GetSequence();
}
void ParComplexGridFunction::Destroy()
{
delete pgfr; delete pgfi;
if (fec_owned)
{
delete pfes;
delete fec_owned;
fec_owned = NULL;
}
}
void
ParComplexGridFunction::Update()
{
if (pfes->GetSequence() == fes_sequence)
{
return; // space and grid function are in sync, no-op
}
fes_sequence = pfes->GetSequence();
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
const int vsize = pfes->GetVSize();
const Operator *T = pfes->GetUpdateOperator();
@@ -929,17 +719,6 @@ ParComplexGridFunction::Update()
}
}
int ParComplexGridFunction::VectorDim() const
{
const FiniteElement *fe = pfes->GetTypicalFE();
if (!fe || fe->GetRangeType() == FiniteElement::SCALAR)
{
return pfes->GetVDim();
}
return pfes->GetVDim()*std::max(pfes->GetMesh()->SpaceDimension(),
fe->GetRangeDim());
}
void
ParComplexGridFunction::ProjectCoefficient(Coefficient &real_coeff,
Coefficient &imag_coeff)
@@ -1010,6 +789,7 @@ ParComplexGridFunction::ProjectBdrCoefficientTangent(VectorCoefficient
void
ParComplexGridFunction::Distribute(const Vector *tv)
{
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
const int tvsize = pfes->GetTrueVSize();
tv->Read();
@@ -1027,6 +807,7 @@ ParComplexGridFunction::Distribute(const Vector *tv)
void
ParComplexGridFunction::ParallelProject(Vector &tv) const
{
ParFiniteElementSpace *pfes = pgfr->ParFESpace();
const int tvsize = pfes->GetTrueVSize();
tv.Write();
@@ -1044,60 +825,6 @@ ParComplexGridFunction::ParallelProject(Vector &tv) const
tvi.SyncAliasMemory(tv);
}
void ParComplexGridFunction::Save(std::ostream &os) const
{
os << "ParComplexGridFunction\n";
pfes->Save(os);
os << '\n';
int vsize = pfes->GetVSize();
real_t *data_ = const_cast<real_t*>(HostRead());
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
if (pfes->GetOrdering() == Ordering::byNODES)
{
Vector::Print(os, 1);
}
else
{
Vector::Print(os, pfes->GetVDim());
}
for (int i = 0; i < vsize; i++)
{
if (pfes->GetDofSign(i) < 0)
{
data_[i] = -data_[i];
data_[i+vsize] = -data_[i+vsize];
}
}
os.flush();
}
void ParComplexGridFunction::Save(const char *fname, int precision) const
{
int rank = pfes->GetMyRank();
ostringstream fname_with_suffix;
fname_with_suffix << fname << "." << setfill('0') << setw(6) << rank;
ofstream ofs(fname_with_suffix.str().c_str());
ofs.precision(precision);
Save(ofs);
}
std::ostream &operator<<(std::ostream &os, const ParComplexGridFunction &sol)
{
sol.Save(os);
return os;
}
ParComplexLinearForm::ParComplexLinearForm(ParFiniteElementSpace *pfes,
ComplexOperator::Convention
+16 -159
View File
@@ -35,53 +35,15 @@ private:
GridFunction * gfi;
protected:
/// FE space on which the grid function lives. Owned if #fec_owned
/// is not NULL.
FiniteElementSpace *fes;
/** @brief Used when the grid function is read from a file. It can also be
set explicitly, see MakeOwner().
If not NULL, this pointer is owned by the ComplexGridFunction. */
FiniteElementCollection *fec_owned;
long fes_sequence; // see FiniteElementSpace::sequence, Mesh::sequence
void Destroy();
void Destroy() { delete gfr; delete gfi; }
public:
/** @brief Construct a ComplexGridFunction associated with the
FiniteElementSpace @a *f. */
ComplexGridFunction(FiniteElementSpace *f);
/** @brief Construct a ComplexGridFunction on the given Mesh, using the data
from @a input.
The content of @a input should be in the format created by the method
Save(). The reconstructed FiniteElementSpace and FiniteElementCollection
are owned by the ComplexGridFunction. */
ComplexGridFunction(Mesh *m, std::istream &input);
void Update();
/** Return update counter, similar to Mesh::GetSequence(). Used to
check if it is up to date with the space. */
long GetSequence() const { return fes_sequence; }
/// Make the ComplexGridFunction the owner of #fec_owned and #fes.
/** If the new FiniteElementCollection, @a fec_, is NULL, ownership
of #fec_owned and #fes is taken away. */
void MakeOwner(FiniteElementCollection *fec_) { fec_owned = fec_; }
/// Returns a pointer to the FiniteElementCollection used to
/// construct this ComplexGridFunction if this class owns that
/// object. Otherwise this function will return NULL.
FiniteElementCollection *OwnFEC() { return fec_owned; }
/// Shortcut for calling FiniteElementSpace::GetVectorDim() on the
/// underlying #fes
int VectorDim() const;
/// Assign constant values to the ComplexGridFunction data.
ComplexGridFunction &operator=(const std::complex<real_t> & value)
{ *gfr = value.real(); *gfi = value.imag(); return *this; }
@@ -101,8 +63,8 @@ public:
VectorCoefficient &imag_coeff,
Array<int> &attr);
FiniteElementSpace *FESpace() { return fes; }
const FiniteElementSpace *FESpace() const { return fes; }
FiniteElementSpace *FESpace() { return gfr->FESpace(); }
const FiniteElementSpace *FESpace() const { return gfr->FESpace(); }
GridFunction & real() { return *gfr; }
GridFunction & imag() { return *gfi; }
@@ -117,52 +79,11 @@ public:
/// @a gfr and @a gfi to match the ComplexGridFunction.
void SyncAlias() { gfr->SyncAliasMemory(*this); gfi->SyncAliasMemory(*this); }
/// @brief Returns ||u_ex - u_h||_L2 for complex-valued scalar fields
///
/// @see GridFunction::ComputeL2Error(Coefficient &exsol,
/// const IntegrationRule *irs[],
/// const Array<int> *elems) const
/// for more detailed documentation.
virtual real_t ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
const IntegrationRule *irs[] = NULL) const
{
real_t err_r = gfr->ComputeL2Error(exsolr, irs);
real_t err_i = gfi->ComputeL2Error(exsoli, irs);
return sqrt(err_r * err_r + err_i * err_i);
}
/// @brief Returns ||u_ex - u_h||_L2 for complex-valued vector fields
///
/// @see GridFunction::ComputeL2Error(VectorCoefficient &exsol,
/// const IntegrationRule *irs[],
/// const Array<int> *elems) const
/// for more detailed documentation.
virtual real_t ComputeL2Error(VectorCoefficient &exsolr,
VectorCoefficient &exsoli,
const IntegrationRule *irs[] = NULL,
Array<int> *elems = NULL) const
{
real_t err_r = gfr->ComputeL2Error(exsolr, irs, elems);
real_t err_i = gfi->ComputeL2Error(exsoli, irs, elems);
return sqrt(err_r * err_r + err_i * err_i);
}
/// Save the ComplexGridFunction to an output stream.
virtual void Save(std::ostream &out) const;
/// Save the ComplexGridFunction to a file
/** The given @a precision will be used for ASCII output. */
virtual void Save(const char *fname, int precision=16) const;
/// Destroys the grid function.
virtual ~ComplexGridFunction() { Destroy(); }
};
/** Overload operator<< for std::ostream and ComplexGridFunction; not valid
for the class ParComplexGridFunction */
std::ostream &operator<<(std::ostream &out, const ComplexGridFunction &sol);
/** Class for a complex-valued linear form
The @a convention argument in the class's constructor is documented in the
@@ -424,23 +345,12 @@ public:
class ParComplexGridFunction : public Vector
{
private:
ParGridFunction * pgfr;
ParGridFunction * pgfi;
protected:
/// FE space on which the grid function lives. Owned if #fec_owned
/// is not NULL.
ParFiniteElementSpace *pfes;
/** @brief Used when the grid function is read from a file. It can also be
set explicitly, see MakeOwner().
If not NULL, this pointer is owned by the ParComplexGridFunction. */
FiniteElementCollection *fec_owned;
long fes_sequence; // see FiniteElementSpace::sequence, Mesh::sequence
void Destroy();
void Destroy() { delete pgfr; delete pgfi; }
public:
@@ -448,33 +358,8 @@ public:
ParFiniteElementSpace @a *pf. */
ParComplexGridFunction(ParFiniteElementSpace *pf);
/** @brief Construct a ParComplexGridFunction on a given ParMesh,
@a pmesh, reading from an std::istream.
In the process, a ParFiniteElementSpace and a FiniteElementCollection are
constructed. The new ParComplexGridFunction assumes ownership of both. */
ParComplexGridFunction(ParMesh *pmesh, std::istream &input);
void Update();
/** Return update counter, similar to Mesh::GetSequence(). Used to
check if it is up to date with the space. */
long GetSequence() const { return fes_sequence; }
/// Make the ParComplexGridFunction the owner of #fec_owned and #pfes.
/** If the new FiniteElementCollection, @a fec_, is NULL, ownership
of #fec_owned and #pfes is taken away. */
void MakeOwner(FiniteElementCollection *fec_) { fec_owned = fec_; }
/// Returns a pointer to the FiniteElementCollection used to
/// construct this ParComplexGridFunction if this class owns that
/// object. Otherwise this function will return NULL.
FiniteElementCollection *OwnFEC() { return fec_owned; }
/// Shortcut for calling FiniteElementSpace::GetVectorDim() on the
/// underlying #pfes
int VectorDim() const;
/// Assign constant values to the ParComplexGridFunction data.
ParComplexGridFunction &operator=(const std::complex<real_t> & value)
{ *pgfr = value.real(); *pgfi = value.imag(); return *this; }
@@ -500,11 +385,11 @@ public:
/// Returns the vector restricted to the true dofs.
void ParallelProject(Vector &tv) const;
FiniteElementSpace *FESpace() { return pfes; }
const FiniteElementSpace *FESpace() const { return pfes; }
FiniteElementSpace *FESpace() { return pgfr->FESpace(); }
const FiniteElementSpace *FESpace() const { return pgfr->FESpace(); }
ParFiniteElementSpace *ParFESpace() { return pfes; }
const ParFiniteElementSpace *ParFESpace() const { return pfes; }
ParFiniteElementSpace *ParFESpace() { return pgfr->ParFESpace(); }
const ParFiniteElementSpace *ParFESpace() const { return pgfr->ParFESpace(); }
ParGridFunction & real() { return *pgfr; }
ParGridFunction & imag() { return *pgfi; }
@@ -517,32 +402,17 @@ public:
/// Update the alias memory location of the real and imaginary
/// ParGridFunction @a pgfr and @a pgfi to match the ParComplexGridFunction.
void SyncAlias()
{ pgfr->SyncAliasMemory(*this); pgfi->SyncAliasMemory(*this); }
void SyncAlias() { pgfr->SyncAliasMemory(*this); pgfi->SyncAliasMemory(*this); }
/// @brief Returns ||u_ex - u_h||_L2 in parallel for complex-valued
/// scalar fields
///
/// @see GridFunction::ComputeL2Error(Coefficient &exsol,
/// const IntegrationRule *irs[],
/// const Array<int> *elems) const
/// for more detailed documentation.
virtual real_t ComputeL2Error(Coefficient &exsolr, Coefficient &exsoli,
const IntegrationRule *irs[] = NULL,
Array<int> *elems = NULL) const
const IntegrationRule *irs[] = NULL) const
{
real_t err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
real_t err_i = pgfi->ComputeL2Error(exsoli, irs, elems);
return hypot(err_r, err_i);
real_t err_r = pgfr->ComputeL2Error(exsolr, irs);
real_t err_i = pgfi->ComputeL2Error(exsoli, irs);
return sqrt(err_r * err_r + err_i * err_i);
}
/// @brief Returns ||u_ex - u_h||_L2 in parallel for complex-valued
/// vector fields
///
/// @see GridFunction::ComputeL2Error(VectorCoefficient &exsol,
/// const IntegrationRule *irs[],
/// const Array<int> *elems) const
/// for more detailed documentation.
virtual real_t ComputeL2Error(VectorCoefficient &exsolr,
VectorCoefficient &exsoli,
const IntegrationRule *irs[] = NULL,
@@ -550,28 +420,15 @@ public:
{
real_t err_r = pgfr->ComputeL2Error(exsolr, irs, elems);
real_t err_i = pgfi->ComputeL2Error(exsoli, irs, elems);
return hypot(err_r, err_i);
return sqrt(err_r * err_r + err_i * err_i);
}
/// Save the local portion of the ParComplexGridFunction
/** This differs from the serial ComplexGridFunction::Save in that it
takes into account the signs of the local dofs. */
void Save(std::ostream &out) const;
/// Save the ParComplexGridFunction to files
/** Saves one file for each MPI rank. The files will be given suffixes
according to the MPI rank. The given @a precision will be used for ASCII
output. */
void Save(const char *fname, int precision=16) const;
/// Destroys grid function.
virtual ~ParComplexGridFunction() { Destroy(); }
};
/** Overload operator<< for std::ostream and ParComplexGridFunction */
std::ostream &operator<<(std::ostream &out, const ParComplexGridFunction &sol);
/** Class for a complex-valued, parallel linear form
The @a convention argument in the class's constructor is documented in the
+11 -169
View File
@@ -310,9 +310,9 @@ void DataCollection::SaveField(const std::string &field_name)
}
}
void DataCollection::SaveQField(const std::string &field_name)
void DataCollection::SaveQField(const std::string &q_field_name)
{
QFieldMapIterator it = q_field_map.find(field_name);
QFieldMapIterator it = q_field_map.find(q_field_name);
if (it != q_field_map.end())
{
SaveOneQField(it);
@@ -780,11 +780,6 @@ void ParaViewDataCollectionBase::SetHighOrderOutput(bool high_order_output_)
high_order_output = high_order_output_;
}
void ParaViewDataCollectionBase::SetBoundaryOutput(bool bdr_output_)
{
bdr_output = bdr_output_;
}
void ParaViewDataCollectionBase::SetCompressionLevel(int compression_level_)
{
MFEM_ASSERT(compression_level_ >= -1 && compression_level_ <= 9,
@@ -940,19 +935,16 @@ void ParaViewDataCollection::Save()
std::string vtu_prefix = col_path + "/" + GenerateVTUPath() + "/";
// Save the local part of the mesh and grid functions fields to the local
// VTU file. Also save coefficient fields.
// VTU file
{
std::ofstream os(vtu_prefix + GenerateVTUFileName("proc", myid));
os.precision(precision);
SaveDataVTU(os, levels_of_detail);
}
// Save the local part of the quadrature function fields.
// Save the local part of the quadrature function fields
for (const auto &qfield : q_field_map)
{
MFEM_VERIFY(!bdr_output,
"QuadratureFunction output is not supported for "
"ParaViewDataCollection on domain boundary!");
const std::string &field_name = qfield.first;
std::ofstream os(vtu_prefix + GenerateVTUFileName(field_name, myid));
qfield.second->SaveVTU(os, pv_data_format, GetCompressionLevel(), field_name);
@@ -968,7 +960,7 @@ void ParaViewDataCollection::Save()
std::ofstream pvtu_out(vtu_prefix + GeneratePVTUFileName("data"));
WritePVTUHeader(pvtu_out);
// Grid function fields and coefficient fields
// Grid function fields
pvtu_out << "<PPointData>\n";
for (auto &field_it : field_map)
{
@@ -979,24 +971,7 @@ void ParaViewDataCollection::Save()
<< VTKComponentLabels(vec_dim) << " "
<< "format=\"" << GetDataFormatString() << "\" />\n";
}
for (auto &field_it : coeff_field_map)
{
int vec_dim = 1;
pvtu_out << "<PDataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << field_it.first
<< "\" NumberOfComponents=\"" << vec_dim << "\" "
<< "format=\"" << GetDataFormatString() << "\" />\n";
}
for (auto &field_it : vcoeff_field_map)
{
int vec_dim = field_it.second->GetVDim();
pvtu_out << "<PDataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << field_it.first
<< "\" NumberOfComponents=\"" << vec_dim << "\" "
<< "format=\"" << GetDataFormatString() << "\" />\n";
}
pvtu_out << "</PPointData>\n";
// Element attributes
pvtu_out << "<PCellData>\n";
pvtu_out << "\t<PDataArray type=\"Int32\" Name=\"" << "attribute"
@@ -1094,8 +1069,7 @@ void ParaViewDataCollection::SaveDataVTU(std::ostream &os, int ref)
}
os << " version=\"2.2\" byte_order=\"" << VTKByteOrder() << "\">\n";
os << "<UnstructuredGrid>\n";
mesh->PrintVTU(os,ref,pv_data_format,high_order_output,GetCompressionLevel(),
bdr_output);
mesh->PrintVTU(os,ref,pv_data_format,high_order_output,GetCompressionLevel());
// dump out the grid functions as point data
os << "<PointData >\n";
@@ -1103,21 +1077,8 @@ void ParaViewDataCollection::SaveDataVTU(std::ostream &os, int ref)
// iterate over all grid functions
for (FieldMapIterator it=field_map.begin(); it!=field_map.end(); ++it)
{
MFEM_VERIFY(!bdr_output,
"GridFunction output is not supported for "
"ParaViewDataCollection on domain boundary!");
SaveGFieldVTU(os,ref,it);
}
// save the coefficient functions
// iterate over all Coefficient and VectorCoefficient functions
for (const auto &kv : coeff_field_map)
{
SaveCoeffFieldVTU(os, ref, kv.first, *kv.second);
}
for (const auto &kv : vcoeff_field_map)
{
SaveVCoeffFieldVTU(os, ref, kv.first, *kv.second);
}
os << "</PointData>\n";
// close the mesh
os << "</Piece>\n"; // close the piece open in the PrintVTU method
@@ -1140,6 +1101,7 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
<< "format=\"" << GetDataFormatString() << "\" >" << '\n';
if (vec_dim == 1)
{
// scalar data
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
@@ -1169,131 +1131,11 @@ void ParaViewDataCollection::SaveGFieldVTU(std::ostream &os, int ref_,
}
}
}
if (pv_data_format != VTKFormat::ASCII)
if (IsBinaryFormat())
{
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
}
os << "</DataArray>" << std::endl;
}
void ParaViewDataCollection::SaveCoeffFieldVTU(std::ostream &os, int ref_,
const std::string &name, Coefficient &coeff)
{
RefinedGeometry *RefG;
real_t val;
std::vector<char> buf;
int vec_dim = 1;
os << "<DataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << name
<< "\" NumberOfComponents=\"" << vec_dim << "\""
<< " format=\"" << GetDataFormatString() << "\" >" << '\n';
{
// scalar data
if (!bdr_output)
{
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetElementBaseGeometry(i), ref_, 1);
ElementTransformation *eltrans = mesh->GetElementTransformation(i);
const IntegrationRule *ir = &RefG->RefPts;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
eltrans->SetIntPoint(&ip);
val = coeff.Eval(*eltrans, ip);
WriteBinaryOrASCII(os, buf, val, "\n", pv_data_format);
}
}
}
else
{
for (int i = 0; i < mesh->GetNBE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetBdrElementBaseGeometry(i), ref_, 1);
ElementTransformation *eltrans = mesh->GetBdrElementTransformation(i);
const IntegrationRule *ir = &RefG->RefPts;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
eltrans->SetIntPoint(&ip);
val = coeff.Eval(*eltrans, ip);
WriteBinaryOrASCII(os, buf, val, "\n", pv_data_format);
}
}
}
}
if (pv_data_format != VTKFormat::ASCII)
{
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
}
os << "</DataArray>" << std::endl;
}
void ParaViewDataCollection::SaveVCoeffFieldVTU(std::ostream &os, int ref_,
const std::string &name, VectorCoefficient &coeff)
{
RefinedGeometry *RefG;
Vector val;
std::vector<char> buf;
int vec_dim = coeff.GetVDim();
os << "<DataArray type=\"" << GetDataTypeString()
<< "\" Name=\"" << name
<< "\" NumberOfComponents=\"" << vec_dim << "\""
<< " format=\"" << GetDataFormatString() << "\" >" << '\n';
{
// vector data
if (!bdr_output)
{
for (int i = 0; i < mesh->GetNE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetElementBaseGeometry(i), ref_, 1);
ElementTransformation *eltrans = mesh->GetElementTransformation(i);
const IntegrationRule *ir = &RefG->RefPts;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
eltrans->SetIntPoint(&ip);
coeff.Eval(val, *eltrans, ip);
for (int jj = 0; jj < val.Size(); jj++)
{
WriteBinaryOrASCII(os, buf, val(jj), " ", pv_data_format);
}
if (pv_data_format == VTKFormat::ASCII) { os << '\n'; }
}
}
}
else
{
for (int i = 0; i < mesh->GetNBE(); i++)
{
RefG = GlobGeometryRefiner.Refine(
mesh->GetBdrElementBaseGeometry(i), ref_, 1);
ElementTransformation *eltrans = mesh->GetBdrElementTransformation(i);
const IntegrationRule *ir = &RefG->RefPts;
for (int j = 0; j < ir->GetNPoints(); j++)
{
const IntegrationPoint &ip = ir->IntPoint(j);
eltrans->SetIntPoint(&ip);
coeff.Eval(val, *eltrans, ip);
for (int jj = 0; jj < val.Size(); jj++)
{
WriteBinaryOrASCII(os, buf, val(jj), " ", pv_data_format);
}
if (pv_data_format == VTKFormat::ASCII) { os << '\n'; }
}
}
}
}
if (pv_data_format != VTKFormat::ASCII)
{
WriteBase64WithSizeAndClear(os, buf, GetCompressionLevel());
WriteVTKEncodedCompressed(os,buf.data(),buf.size(),GetCompressionLevel());
os << '\n';
}
os << "</DataArray>" << std::endl;
}
+11 -47
View File
@@ -133,7 +133,6 @@ private:
/// A collection of named QuadratureFunctions
typedef NamedFieldsMap<QuadratureFunction> QFieldMap;
public:
typedef GFieldMap::MapType FieldMapType;
typedef GFieldMap::iterator FieldMapIterator;
@@ -250,9 +249,10 @@ public:
{ field_map.Deregister(field_name, own_data); }
/// Add a QuadratureFunction to the collection.
virtual void RegisterQField(const std::string& field_name,
virtual void RegisterQField(const std::string& q_field_name,
QuadratureFunction *qf)
{ q_field_map.Register(field_name, qf, own_data); }
{ q_field_map.Register(q_field_name, qf, own_data); }
/// Remove a QuadratureFunction from the collection
virtual void DeregisterQField(const std::string& field_name)
@@ -280,13 +280,13 @@ public:
#endif
/// Check if a QuadratureFunction with the given name is in the collection.
bool HasQField(const std::string& field_name) const
{ return q_field_map.Has(field_name); }
bool HasQField(const std::string& q_field_name) const
{ return q_field_map.Has(q_field_name); }
/// Get a pointer to a QuadratureFunction in the collection.
/** Returns NULL if @a field_name is not in the collection. */
QuadratureFunction *GetQField(const std::string& field_name)
{ return q_field_map.Get(field_name); }
QuadratureFunction *GetQField(const std::string& q_field_name)
{ return q_field_map.Get(q_field_name); }
/// Get a const reference to the internal field map.
/** The keys in the map are the field names and the values are pointers to
@@ -302,13 +302,11 @@ public:
/// Get a pointer to the mesh in the collection
Mesh *GetMesh() { return mesh; }
/// Set/change the mesh associated with the collection
/** When passed a Mesh, assumes the serial case: MPI rank id is set to 0 and
MPI num_procs is set to 1. When passed a ParMesh, MPI info from the
ParMesh is used to set the DataCollection's MPI rank and num_procs. */
virtual void SetMesh(Mesh *new_mesh);
#ifdef MFEM_USE_MPI
/// Set/change the mesh associated with the collection.
/** For this case, @a comm is used to set the DataCollection's MPI rank id
@@ -371,7 +369,8 @@ public:
/// Save one field, assuming the collection directory already exists.
virtual void SaveField(const std::string &field_name);
/// Save one q-field, assuming the collection directory already exists.
virtual void SaveQField(const std::string &field_name);
virtual void SaveQField(const std::string &q_field_name);
/// Load the collection. Not implemented in the base class DataCollection.
virtual void Load(int cycle_ = 0);
@@ -511,9 +510,7 @@ protected:
int compression_level = -1;
bool high_order_output = false;
bool restart_mode = false;
bool bdr_output = false;
VTKFormat pv_data_format = VTKFormat::BINARY;
public:
ParaViewDataCollectionBase(const std::string &name, Mesh *mesh);
@@ -546,10 +543,6 @@ public:
/// Reading high-order data requires ParaView 5.5 or later.
void SetHighOrderOutput(bool high_order_output_);
/// @brief Configures collection to save only fields evaluated on boundaries of
/// the mesh.
void SetBoundaryOutput(bool bdr_output_);
/// If compression is enabled, return the compression level, else return 0.
int GetCompressionLevel() const;
@@ -571,6 +564,8 @@ public:
///
/// If restart is enabled, new writes will preserve timestep metadata for any
/// solutions prior to the currently defined time.
///
/// Initially, restart mode is disabled.
void UseRestartMode(bool restart_mode_);
};
@@ -580,23 +575,11 @@ class ParaViewDataCollection : public ParaViewDataCollectionBase
private:
std::fstream pvd_stream;
/// A collection of named Coefficients and VectorCoefficients
using CoeffFieldMap = NamedFieldsMap<Coefficient>;
using VCoeffFieldMap = NamedFieldsMap<VectorCoefficient>;
/** A FieldMap mapping registered names to Coefficient and VectorCoefficient
pointers. */
CoeffFieldMap coeff_field_map;
VCoeffFieldMap vcoeff_field_map;
protected:
void WritePVTUHeader(std::ostream &out);
void WritePVTUFooter(std::ostream &out, const std::string &vtu_prefix);
void SaveDataVTU(std::ostream &out, int ref);
void SaveGFieldVTU(std::ostream& out, int ref_, const FieldMapIterator& it);
void SaveCoeffFieldVTU(std::ostream& out, int ref_, const std::string &name,
Coefficient &coeff);
void SaveVCoeffFieldVTU(std::ostream& out, int ref_, const std::string &name,
VectorCoefficient& coeff);
const char *GetDataFormatString() const;
const char *GetDataTypeString() const;
@@ -615,25 +598,6 @@ public:
ParaViewDataCollection(const std::string& collection_name,
Mesh *mesh_ = nullptr);
/// Get a const reference to the internal coefficient-field map.
const typename CoeffFieldMap::MapType &GetCoeffFieldMap() const
{ return coeff_field_map.GetMap(); }
const typename VCoeffFieldMap::MapType &GetVCoeffFieldMap() const
{ return vcoeff_field_map.GetMap(); }
/// Add a Coefficient or VectorCoefficient to the collection.
void RegisterCoeffField(const std::string& field_name, Coefficient *coeff)
{ coeff_field_map.Register(field_name, coeff, own_data); }
void RegisterVCoeffField(const std::string& field_name,
VectorCoefficient *vcoeff)
{ vcoeff_field_map.Register(field_name, vcoeff, own_data); }
/// Remove a Coefficient or VectorCoefficient from the collection
void DeregisterCoeffField(const std::string& field_name)
{ coeff_field_map.Deregister(field_name, own_data); }
void DeregisterVCoeffField(const std::string& field_name)
{ vcoeff_field_map.Deregister(field_name, own_data); }
/// Save the collection - the directory name is constructed based on the
/// cycle value
void Save() override;
+32 -145
View File
@@ -211,8 +211,8 @@ private:
///
/// The operator is constructed with solution fields that it will act on and
/// parameter fields that define coefficients. Quadrature functions are added by
/// e.g. using AddDomainIntegrator() which specify how the operator evaluates
/// those functions and parameters at quadrature points.
/// e.g. using AddDomainIntegrator() which specify how the operator evaluates f
/// those functionas and parameters at quadrature points.
///
/// Derivatives can be computed by obtaining a DerivativeOperator using
/// GetDerivative().
@@ -280,22 +280,6 @@ public:
}
}
/// @brief Add an integrator to the operator.
/// Called only from AddDomainIntegrator() and AddBoundaryIntegrator().
template <
typename entity_t,
typename qfunc_t,
typename input_t,
typename output_t,
typename derivative_ids_t>
void AddIntegrator(
qfunc_t &qfunc,
input_t inputs,
output_t outputs,
const IntegrationRule &integration_rule,
const Array<int> &attributes,
derivative_ids_t derivative_ids);
/// @brief Add a domain integrator to the operator.
///
/// @param qfunc The quadrature function to be added.
@@ -321,31 +305,6 @@ public:
const Array<int> &domain_attributes,
derivative_ids_t derivative_ids = std::make_index_sequence<0> {});
/// @brief Add a boundary integrator to the operator.
///
/// @param qfunc The quadrature function to be added.
/// @param inputs Tuple of FieldOperators for the inputs of the quadrature
/// function.
/// @param outputs Tuple of FieldOperators for the outputs of the quadrature
/// function.
/// @param integration_rule IntegrationRule to use with this integrator.
/// @param boundary_attributes Boundary attributes marker array indicating over
/// which attributes this integrator will integrate over.
/// @param derivative_ids Derivatives to be made available for this
/// integrator.
template <
typename qfunc_t,
typename input_t,
typename output_t,
typename derivative_ids_t = decltype(std::make_index_sequence<0> {})>
void AddBoundaryIntegrator(
qfunc_t &qfunc,
input_t inputs,
output_t outputs,
const IntegrationRule &integration_rule,
const Array<int> &boundary_attributes,
derivative_ids_t derivative_ids = std::make_index_sequence<0> {});
/// @brief Set the parameters for the operator.
///
/// This has to be called before using Mult() or MultTranspose().
@@ -464,52 +423,7 @@ void DifferentiableOperator::AddDomainIntegrator(
const Array<int> &domain_attributes,
derivative_ids_t derivative_ids)
{
AddIntegrator<Entity::Element>(
qfunc, inputs, outputs, integration_rule, domain_attributes, derivative_ids);
}
template <
typename qfunc_t,
typename input_t,
typename output_t,
typename derivative_ids_t>
void DifferentiableOperator::AddBoundaryIntegrator(
qfunc_t &qfunc,
input_t inputs,
output_t outputs,
const IntegrationRule &integration_rule,
const Array<int> &boundary_attributes,
derivative_ids_t derivative_ids)
{
if (mesh.GetNFbyType(FaceType::Boundary) != mesh.GetNBE())
{
MFEM_ABORT("AddBoundaryIntegrator on meshes with interior boundaries is not supported.");
}
AddIntegrator<Entity::BoundaryElement>(
qfunc, inputs, outputs, integration_rule, boundary_attributes, derivative_ids);
}
template <
typename entity_t,
typename qfunc_t,
typename input_t,
typename output_t,
typename derivative_ids_t>
void DifferentiableOperator::AddIntegrator(
qfunc_t &qfunc,
input_t inputs,
output_t outputs,
const IntegrationRule &integration_rule,
const Array<int> &attributes,
derivative_ids_t derivative_ids)
{
if constexpr (!(std::is_same_v<entity_t, Entity::Element> ||
std::is_same_v<entity_t, Entity::BoundaryElement>))
{
static_assert(dfem::always_false<entity_t>,
"entity type not supported in AddIntegrator");
}
using entity_t = Entity::Element;
static constexpr size_t num_inputs =
tuple_size<decltype(inputs)>::value;
@@ -563,44 +477,32 @@ void DifferentiableOperator::AddIntegrator(
auto output_to_field =
create_descriptors_to_fields_map<entity_t>(fields, outputs);
const Array<int> *elem_attributes = nullptr;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
// TODO: factor out
std::vector<int> inputs_vdim(num_inputs);
for_constexpr<num_inputs>([&](auto i)
{
elem_attributes = &mesh.GetElementAttributes();
}
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
inputs_vdim[i] = get<i>(inputs).vdim;
});
Array<int> elem_attributes;
elem_attributes.SetSize(mesh.GetNE());
for (int i = 0; i < mesh.GetNE(); ++i)
{
elem_attributes = &mesh.GetBdrFaceAttributes();
elem_attributes[i] = mesh.GetAttribute(i);
}
const auto output_fop = get<0>(outputs);
test_space_field_idx = FindIdx(output_fop.GetFieldId(), fields);
bool use_sum_factorization = false;
Element::Type entity_element_type;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
auto entity_element_type =
Element::TypeFromGeometry(mesh.GetTypicalElementGeometry());
if ((entity_element_type == Element::QUADRILATERAL ||
entity_element_type == Element::HEXAHEDRON) &&
use_tensor_product_structure == true)
{
entity_element_type =
Element::TypeFromGeometry(mesh.GetTypicalElementGeometry());
if ((entity_element_type == Element::QUADRILATERAL ||
entity_element_type == Element::HEXAHEDRON) &&
use_tensor_product_structure == true)
{
use_sum_factorization = true;
}
}
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
{
entity_element_type =
Element::TypeFromGeometry(mesh.GetTypicalFaceGeometry());
if ((entity_element_type == Element::SEGMENT ||
entity_element_type == Element::QUADRILATERAL) &&
use_tensor_product_structure == true)
{
use_sum_factorization = true;
}
use_sum_factorization = true;
}
ElementDofOrdering element_dof_ordering = ElementDofOrdering::NATIVE;
@@ -638,17 +540,8 @@ void DifferentiableOperator::AddIntegrator(
prolongation_transpose = get_prolongation_transpose(
fields[test_space_field_idx], output_fop, mesh.GetComm());
int dimension;
if constexpr (std::is_same_v<entity_t, Entity::Element>)
{
dimension = mesh.Dimension();
}
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
{
dimension = mesh.Dimension() - 1;
}
[[maybe_unused]] const int num_elements = GetNumEntities<entity_t>(mesh);
const int dimension = mesh.Dimension();
[[maybe_unused]] const int num_elements = GetNumEntities<Entity::Element>(mesh);
const int num_entities = GetNumEntities<entity_t>(mesh);
const int num_qp = integration_rule.GetNPoints();
@@ -722,12 +615,6 @@ void DifferentiableOperator::AddIntegrator(
thread_blocks.z = 1;
}
}
else if (dimension == 1)
{
thread_blocks.x = q1d;
thread_blocks.y = 1;
thread_blocks.z = 1;
}
action_callbacks.push_back(
// Explicitly capture everything we need, so we can make explicit choice
@@ -743,7 +630,7 @@ void DifferentiableOperator::AddIntegrator(
test_vdim, // int (= output_fop.vdim)
test_op_dim, // int (derived from output_fop)
inputs, // mfem::future::tuple
attributes, // Array<int>
domain_attributes, // Array<int>
ir_weights, // DeviceTensor
use_sum_factorization, // bool
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
@@ -776,13 +663,13 @@ void DifferentiableOperator::AddIntegrator(
action_shmem_info.field_sizes,
num_entities);
const bool has_attr = attributes.Size() > 0;
const auto d_attr = attributes.Read();
const auto d_elem_attr = elem_attributes->Read();
const bool has_attr = domain_attributes.Size() > 0;
const auto d_domain_attr = domain_attributes.Read();
const auto d_elem_attr = elem_attributes.Read();
forall([=] MFEM_HOST_DEVICE (int e, void *shmem)
{
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem, input_shmem,
residual_shmem, scratch_shmem] =
@@ -852,7 +739,7 @@ void DifferentiableOperator::AddIntegrator(
test_vdim, // int (= output_fop.vdim)
test_op_dim, // int (derived from output_fop)
inputs, // mfem::future::tuple
attributes, // Array<int>
domain_attributes, // Array<int>
ir_weights, // DeviceTensor
use_sum_factorization, // bool
input_dtq_maps, // std::array<DofToQuadMap, num_fields>
@@ -890,14 +777,14 @@ void DifferentiableOperator::AddIntegrator(
shmem_info.direction_size,
num_entities);
const bool has_attr = attributes.Size() > 0;
const auto d_attr = attributes.Read();
const auto d_elem_attr = elem_attributes->Read();
const auto d_elem_attr = elem_attributes.Read();
const bool has_attr = domain_attributes.Size() > 0;
const auto d_domain_attr = domain_attributes.Read();
derivative_action_e = 0.0;
forall([=] MFEM_HOST_DEVICE (int e, real_t *shmem)
{
if (has_attr && !d_attr[d_elem_attr[e] - 1]) { return; }
if (has_attr && !d_domain_attr[d_elem_attr[e] - 1]) { return; }
auto [input_dtq_shmem, output_dtq_shmem, fields_shmem,
direction_shmem, input_shmem,
+1 -84
View File
@@ -95,85 +95,6 @@ void map_quadrature_data_to_fields_impl(
}
}
template <typename output_t>
MFEM_HOST_DEVICE
void map_quadrature_data_to_fields_tensor_impl_1d(
DeviceTensor<2, real_t> &y,
const DeviceTensor<3, real_t> &f,
const output_t &output,
const DofToQuadMap &dtq,
std::array<DeviceTensor<1>, 6> &scratch_mem)
{
[[maybe_unused]] auto B = dtq.B;
[[maybe_unused]] auto G = dtq.G;
if constexpr (is_value_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
const int vdim = output.vdim;
const int test_dim = output.size_on_qp / vdim;
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d);
auto yd = Reshape(&y(0, 0), d1d, vdim);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
real_t acc = 0.0;
for (int qx = 0; qx < q1d; qx++)
{
acc += fqp(vd, 0, qx) * B(qx, 0, dx);
}
yd(dx, vd) = acc;
}
}
MFEM_SYNC_THREAD;
}
else if constexpr (is_gradient_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = G.GetShape();
const int vdim = output.vdim;
const int test_dim = output.size_on_qp / vdim;
auto fqp = Reshape(&f(0, 0, 0), vdim, test_dim, q1d);
auto yd = Reshape(&y(0, 0), d1d, vdim);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(dx, x, d1d)
{
real_t acc = 0.0;
for (int qx = 0; qx < q1d; qx++)
{
acc += fqp(vd, 0, qx) * G(qx, 0, dx);
}
yd(dx, vd) = acc;
}
}
MFEM_SYNC_THREAD;
}
else if constexpr (is_identity_fop<std::decay_t<output_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
auto fqp = Reshape(&f(0, 0, 0), output.size_on_qp, q1d);
auto yqp = Reshape(&y(0, 0), output.size_on_qp, q1d);
for (int sq = 0; sq < output.size_on_qp; sq++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
yqp(sq, qx) = fqp(sq, qx);
}
MFEM_SYNC_THREAD;
}
}
else
{
MFEM_ABORT("quadrature data mapping to field is not implemented for"
" this field descriptor with sum factorization on tensor product elements");
}
}
template <typename output_t>
MFEM_HOST_DEVICE
void map_quadrature_data_to_fields_tensor_impl_2d(
@@ -510,11 +431,7 @@ void map_quadrature_data_to_fields(
{
if (use_sum_factorization)
{
if (dimension == 1)
{
map_quadrature_data_to_fields_tensor_impl_1d(y, f, output, dtq, scratch_mem);
}
else if (dimension == 2)
if (dimension == 2)
{
map_quadrature_data_to_fields_tensor_impl_2d(y, f, output, dtq, scratch_mem);
}
+5 -109
View File
@@ -338,92 +338,6 @@ void map_field_to_quadrature_data_tensor_product_2d(
}
}
template <typename field_operator_t>
MFEM_HOST_DEVICE inline
void map_field_to_quadrature_data_tensor_product_1d(
DeviceTensor<2> &field_qp,
const DofToQuadMap &dtq,
const DeviceTensor<1> &field_e,
const field_operator_t &input,
const DeviceTensor<1, const real_t> &integration_weights,
const std::array<DeviceTensor<1>, 6> &scratch_mem)
{
[[maybe_unused]] auto B = dtq.B;
[[maybe_unused]] auto G = dtq.G;
if constexpr (is_value_fop<std::decay_t<field_operator_t>>::value)
{
auto [q1d, unused, d1d] = B.GetShape();
const int vdim = input.vdim;
const auto field = Reshape(&field_e[0], d1d, vdim);
auto fqp = Reshape(&field_qp[0], vdim, q1d);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
real_t acc = 0.0;
for (int dx = 0; dx < d1d; dx++)
{
acc += B(qx, 0, dx) * field(dx, vd);
}
fqp(vd, qx) = acc;
}
}
MFEM_SYNC_THREAD;
}
else if constexpr (
is_gradient_fop<std::decay_t<field_operator_t>>::value)
{
const auto [q1d, unused, d1d] = B.GetShape();
const int vdim = input.vdim;
const int dim = input.dim;
const auto field = Reshape(&field_e[0], d1d, vdim);
auto fqp = Reshape(&field_qp[0], vdim, dim, q1d);
for (int vd = 0; vd < vdim; vd++)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
real_t acc = 0.0;
for (int dx = 0; dx < d1d; dx++)
{
acc += G(qx, 0, dx) * field(dx, vd);
}
fqp(vd, 0, qx) = acc;
}
MFEM_SYNC_THREAD;
}
}
// TODO: Create separate function for clarity
else if constexpr (
std::is_same_v<std::decay_t<field_operator_t>, Weight>)
{
const int num_qp = integration_weights.GetShape()[0];
// TODO: eeek
const int q1d = (int)floor(std::pow(num_qp, 1.0/input.dim) + 0.5);
auto w = Reshape(&integration_weights[0], q1d);
auto f = Reshape(&field_qp[0], q1d);
MFEM_FOREACH_THREAD(qx, x, q1d)
{
f(qx) = w(qx);
}
MFEM_SYNC_THREAD;
}
else if constexpr (is_identity_fop<std::decay_t<field_operator_t>>::value)
{
const int q1d = B.GetShape()[0];
auto field = Reshape(&field_e[0], input.size_on_qp, q1d);
field_qp = field;
}
else
{
static_assert(dfem::always_false<std::decay_t<field_operator_t>>,
"can't map field to quadrature data");
}
}
template <typename field_operator_t>
MFEM_HOST_DEVICE
void map_field_to_quadrature_data(
@@ -530,13 +444,7 @@ void map_fields_to_quadrature_data(
if (use_sum_factorization)
{
if (dimension == 1)
{
map_field_to_quadrature_data_tensor_product_1d(
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
integration_weights, scratch_mem);
}
else if (dimension == 2)
if (dimension == 2)
{
map_field_to_quadrature_data_tensor_product_2d(
fields_qp[i], dtqmaps[i], field_e, get<i>(fops),
@@ -581,20 +489,14 @@ void map_field_to_quadrature_data_conditional(
{
if (use_sum_factorization)
{
if (dimension == 1)
if (dimension == 2)
{
map_field_to_quadrature_data_tensor_product_1d(
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
}
else if (dimension == 2)
{
map_field_to_quadrature_data_tensor_product_2d(
map_field_to_quadrature_data_tensor_product_3d(
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
}
else if (dimension == 3)
{
map_field_to_quadrature_data_tensor_product_3d(
map_field_to_quadrature_data_tensor_product_2d(
field_qp, dtqmap, field_e, fop, integration_weights, scratch_mem);
}
}
@@ -645,13 +547,7 @@ void map_direction_to_quadrature_data_conditional(
{
if (use_sum_factorization)
{
if (dimension == 1)
{
map_field_to_quadrature_data_tensor_product_1d(
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
integration_weights, scratch_mem);
}
else if (dimension == 2)
if (dimension == 2)
{
map_field_to_quadrature_data_tensor_product_2d(
directions_qp[i], dtqmaps[i], direction_e, get<i>(fops),
+4 -31
View File
@@ -44,16 +44,7 @@ void call_qfunction(
{
if (use_sum_factorization)
{
if (dimension == 1)
{
MFEM_FOREACH_THREAD(q, x, q1d)
{
auto qf_args = decay_tuple<qf_param_ts> {};
auto r = Reshape(&residual_shmem(0, q), rs_qp);
apply_kernel(r, qfunc, qf_args, input_shmem, q);
}
}
else if (dimension == 2)
if (dimension == 2)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
@@ -132,22 +123,7 @@ void call_qfunction_derivative_action(
{
if (use_sum_factorization)
{
if (dimension == 1)
{
MFEM_FOREACH_THREAD(q, x, q1d)
{
auto r = Reshape(&residual_shmem(0, q), das_qp);
auto qf_args = decay_tuple<qf_param_ts> {};
#ifdef MFEM_USE_ENZYME
auto qf_shadow_args = decay_tuple<qf_param_ts> {};
apply_kernel_fwddiff_enzyme(r, qfunc, qf_args, qf_shadow_args, input_shmem,
shadow_shmem, q);
#else
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
#endif
}
}
else if (dimension == 2)
if (dimension == 2)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
@@ -188,10 +164,7 @@ void call_qfunction_derivative_action(
}
}
}
else
{
MFEM_ABORT_KERNEL("unsupported dimension");
}
MFEM_SYNC_THREAD;
}
else
{
@@ -207,8 +180,8 @@ void call_qfunction_derivative_action(
apply_kernel_native_dual(r, qfunc, qf_args, input_shmem, shadow_shmem, q);
#endif
}
MFEM_SYNC_THREAD;
}
MFEM_SYNC_THREAD;
}
template <typename qfunc_t, typename args_ts, size_t num_args>
+5 -13
View File
@@ -44,7 +44,7 @@ void process_qf_arg(
{
for (int j = 0; j < n; j++)
{
arg(j, i).value = u((i * n) + j);
arg(j, i).value = u((i * m) + j);
}
}
}
@@ -94,8 +94,8 @@ void process_qf_arg(
{
for (int j = 0; j < n; j++)
{
arg(j, i).value = u((i * n) + j);
arg(j, i).gradient = v((i * n) + j);
arg(j, i).value = u((i * m) + j);
arg(j, i).gradient = v((i * m) + j);
}
}
}
@@ -181,14 +181,6 @@ void process_derivative_from_native_dual(
}
}
template <typename T>
MFEM_HOST_DEVICE inline
void process_derivative_from_native_dual(
DeviceTensor<1, T> &r,
const dual<T, T> &x)
{
r(0) = x.gradient;
}
template <typename T0, typename T1>
MFEM_HOST_DEVICE inline
@@ -238,7 +230,7 @@ void process_qf_arg(
{
for (int j = 0; j < n; j++)
{
arg(j, i) = u((i * n) + j);
arg(j, i) = u((i * m) + j);
}
}
}
@@ -338,7 +330,7 @@ void process_qf_arg(
{
for (int j = 0; j < n; j++)
{
arg(j, i) = u((i * n) + j);
arg(j, i) = u((i * m) + j);
}
}
}
+3 -3
View File
@@ -454,7 +454,7 @@ MFEM_HOST_DEVICE constexpr auto operator+=(tuple<T...>& x,
*
* @tparam T the types stored in the tuples x and y
* @tparam i integer sequence used to index the tuples
* @param x tuple of values to be subtracted from
* @param x tuple of values to be subracted from
* @param y tuple of values to subtract from x
*/
template <typename... T, int... i>
@@ -596,7 +596,7 @@ MFEM_HOST_DEVICE constexpr auto div_helper(const real_t a,
* @tparam T the types stored in the tuple y
* @tparam i The integer sequence to i
* @param x tuple of values
* @param a the constant denominator
* @param a the constant denomenator
* @return the returned tuple ratio
*/
template <typename... T, int... i>
@@ -726,7 +726,7 @@ MFEM_HOST_DEVICE constexpr auto operator*(const tuple<T...>& x, const real_t a)
/**
* @tparam T the types stored in the tuple
* @tparam i a list of indices used to access each element of the tuple
* @tparam i a list of indices used to acces each element of the tuple
* @param out the ostream to write the output to
* @param A the tuple of values
* @brief helper used to implement printing a tuple of values
+3 -45
View File
@@ -944,44 +944,7 @@ const Operator *get_element_restriction(const FieldDescriptor &f,
}
else
{
static_assert(dfem::always_false<T>,
"can't use get_element_restriction on type");
}
return nullptr; // Unreachable, but avoids compiler warning
}, f.data);
}
/// @brief Get the face restriction operator for a field descriptor.
///
/// @param f the field descriptor.
/// @param o the face dof ordering.
/// @param ft the face type
/// @param m indicator if single or double valued
/// @returns the face restriction operator for the field descriptor in
/// specified ordering.
inline
const Operator *get_face_restriction(const FieldDescriptor &f,
ElementDofOrdering o,
FaceType ft,
L2FaceValues m)
{
return std::visit([&o, &ft, &m](auto&& arg) -> const Operator*
{
using T = std::decay_t<decltype(arg)>;
if constexpr (std::is_same_v<T, const FiniteElementSpace *> ||
std::is_same_v<T, const ParFiniteElementSpace *>)
{
return arg->GetFaceRestriction(o, ft, m);
}
else if constexpr (std::is_same_v<T, const ParameterSpace *>)
{
// ParameterSpace does not support face restrictions
MFEM_ABORT("internal error");
}
else
{
static_assert(dfem::always_false<T>,
"can't use get_face_restriction on type");
static_assert(dfem::always_false<T>, "can't use GetElementRestriction on type");
}
return nullptr; // Unreachable, but avoids compiler warning
}, f.data);
@@ -1002,11 +965,6 @@ const Operator *get_restriction(const FieldDescriptor &f,
{
return get_element_restriction(f, o);
}
else if constexpr (std::is_same_v<entity_t, Entity::BoundaryElement>)
{
return get_face_restriction(f, o, FaceType::Boundary,
L2FaceValues::SingleValued);
}
MFEM_ABORT("restriction not implemented for Entity");
return nullptr;
}
@@ -1016,7 +974,7 @@ const Operator *get_restriction(const FieldDescriptor &f,
/// @param f the field descriptor.
/// @param o the element dof ordering.
/// @param fop the field operator.
/// @returns a tuple containing a std::function with the transpose
/// @returns a tuple containting a std::function with the transpose
/// restriction callback and it's height.
template <typename entity_t, typename fop_t>
inline std::tuple<std::function<void(const Vector&, Vector&)>, int>
@@ -1431,7 +1389,7 @@ create_descriptors_to_fields_map(
if constexpr (std::is_same_v<std::decay_t<decltype(fop)>, Weight>)
{
// TODO-bug: stealing dimension from the first field
fop.dim = GetDimension<entity_t>(fields[0]);
fop.dim = GetDimension<Entity::Element>(fields[0]);
fop.vdim = 1;
fop.size_on_qp = 1;
map = -1;
+47 -64
View File
@@ -661,78 +661,65 @@ void ScalarFiniteElement::ScalarLocalL2Restriction(
void NodalFiniteElement::CreateLexicographicFullMap(const IntegrationRule &ir)
const
{
// Get the FULL version of the map.
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
//Undo the native ordering which is what FiniteElement::GetDofToQuad returns.
auto *d2q_new = new DofToQuad(d2q);
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
const int nqpt = ir.GetNPoints();
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
const int b_dim = (range_type == VECTOR) ? dim : 1;
for (int i = 0; i < nqpt; i++)
{
// Get the FULL version of the map.
auto &d2q = GetDofToQuad(ir, DofToQuad::FULL);
//Undo the native ordering which is what FiniteElement::GetDofToQuad returns.
auto *d2q_new = new DofToQuad(d2q);
d2q_new->mode = DofToQuad::LEXICOGRAPHIC_FULL;
const int nqpt = ir.GetNPoints();
const int b_dim = (range_type == VECTOR) ? dim : 1;
for (int i = 0; i < nqpt; i++)
for (int d = 0; d < b_dim; d++)
{
for (int d = 0; d < b_dim; d++)
for (int j = 0; j < dof; j++)
{
for (int j = 0; j < dof; j++)
{
const double val = d2q.B[i + nqpt*(d+b_dim*lex_ordering[j])];
d2q_new->B[i+nqpt*(d+b_dim*j)] = val;
d2q_new->Bt[j+dof*(i+nqpt*d)] = val;
}
const double val = d2q.B[i + nqpt*(d+b_dim*lex_ordering[j])];
d2q_new->B[i+nqpt*(d+b_dim*j)] = val;
d2q_new->Bt[j+dof*(i+nqpt*d)] = val;
}
}
const int g_dim = [this]()
{
switch (deriv_type)
{
case GRAD: return dim;
case DIV: return 1;
case CURL: return cdim;
default: return 0;
}
}();
for (int i = 0; i < nqpt; i++)
{
for (int d = 0; d < g_dim; d++)
{
for (int j = 0; j < dof; j++)
{
const double val = d2q.G[i + nqpt*(d+g_dim*lex_ordering[j])];
d2q_new->G[i+nqpt*(d+g_dim*j)] = val;
d2q_new->Gt[j+dof*(i+nqpt*d)] = val;
}
}
}
dof2quad_array.Append(d2q_new);
}
const int g_dim = [this]()
{
switch (deriv_type)
{
case GRAD: return dim;
case DIV: return 1;
case CURL: return cdim;
default: return 0;
}
}();
for (int i = 0; i < nqpt; i++)
{
for (int d = 0; d < g_dim; d++)
{
for (int j = 0; j < dof; j++)
{
const double val = d2q.G[i + nqpt*(d+g_dim*lex_ordering[j])];
d2q_new->G[i+nqpt*(d+g_dim*j)] = val;
d2q_new->Gt[j+dof*(i+nqpt*d)] = val;
}
}
}
dof2quad_array.Append(d2q_new);
}
const DofToQuad &NodalFiniteElement::GetDofToQuad(const IntegrationRule &ir,
DofToQuad::Mode mode) const
{
DofToQuad *d2q = nullptr;
#if defined(MFEM_THREAD_SAFE) && defined(MFEM_USE_OPENMP)
#pragma omp critical (DofToQuad)
#endif
//Should make this loop a function of FiniteElement
for (int i = 0; i < dof2quad_array.Size(); i++)
{
//Should make this loop a function of FiniteElement
for (int i = 0; i < dof2quad_array.Size(); i++)
{
d2q = dof2quad_array[i];
if (d2q->IntRule == &ir && d2q->mode == mode) { break; }
d2q = nullptr;
}
const DofToQuad &d2q = *dof2quad_array[i];
if (d2q.IntRule == &ir && d2q.mode == mode) { return d2q; }
}
if (d2q) { return *d2q; }
if (mode != DofToQuad::LEXICOGRAPHIC_FULL)
{
return FiniteElement::GetDofToQuad(ir, mode);
@@ -2633,12 +2620,8 @@ const DofToQuad &TensorBasisElement::GetTensorDofToQuad(
{
for (int i = 0; i < dof2quad_array.Size(); i++)
{
auto* d2q_ = dof2quad_array[i];
if (d2q_->IntRule == &ir && d2q_->mode == mode)
{
d2q = d2q_;
break;
}
d2q = dof2quad_array[i];
if (d2q->IntRule != &ir || d2q->mode != mode) { d2q = nullptr; }
}
if (!d2q)
{
+8 -32
View File
@@ -308,25 +308,13 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
FiniteElement::INTEGRAL,
BasisType::GetType(name[12]));
}
else if (!strncmp(name, "RT_R1D_", 7))
else if (!strncmp(name, "RT_R1D",6))
{
fec = new RT_R1D_FECollection(atoi(name + 11), atoi(name + 7));
fec = new RT_R1D_FECollection(atoi(name+11),atoi(name + 7));
}
else if (!strncmp(name, "RT_R1D@", 7))
else if (!strncmp(name, "RT_R2D",6))
{
fec = new RT_R1D_FECollection(atoi(name + 14), atoi(name + 10),
BasisType::GetType(name[7]),
BasisType::GetType(name[8]));
}
else if (!strncmp(name, "RT_R2D_", 7))
{
fec = new RT_R2D_FECollection(atoi(name + 11), atoi(name + 7));
}
else if (!strncmp(name, "RT_R2D@", 7))
{
fec = new RT_R2D_FECollection(atoi(name + 14), atoi(name + 10),
BasisType::GetType(name[7]),
BasisType::GetType(name[8]));
fec = new RT_R2D_FECollection(atoi(name+11),atoi(name + 7));
}
else if (!strncmp(name, "RT_", 3))
{
@@ -348,25 +336,13 @@ FiniteElementCollection *FiniteElementCollection::New(const char *name)
BasisType::GetType(name[9]),
BasisType::GetType(name[10]));
}
else if (!strncmp(name, "ND_R1D_", 7))
else if (!strncmp(name, "ND_R1D",6))
{
fec = new ND_R1D_FECollection(atoi(name + 11), atoi(name + 7));
fec = new ND_R1D_FECollection(atoi(name+11),atoi(name + 7));
}
else if (!strncmp(name, "ND_R1D@", 7))
else if (!strncmp(name, "ND_R2D",6))
{
fec = new ND_R1D_FECollection(atoi(name + 14), atoi(name + 10),
BasisType::GetType(name[7]),
BasisType::GetType(name[8]));
}
else if (!strncmp(name, "ND_R2D_", 7))
{
fec = new ND_R2D_FECollection(atoi(name + 11), atoi(name + 7));
}
else if (!strncmp(name, "ND_R2D@", 7))
{
fec = new ND_R2D_FECollection(atoi(name + 14), atoi(name + 10),
BasisType::GetType(name[7]),
BasisType::GetType(name[8]));
fec = new ND_R2D_FECollection(atoi(name+11),atoi(name + 7));
}
else if (!strncmp(name, "ND_", 3))
{
-4
View File
@@ -120,9 +120,7 @@ public:
| ND_Trace_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | H_CURL | H^{1/2}-conforming trace elements for H(curl) defined on the interface between mesh elements (faces) |
| ND_Trace@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | H_CURL | H^{1/2}-conforming trace elements for H(curl) defined on the interface between mesh elements (faces) |
| ND_R1D_[DIM]_[ORDER] | H(curl) | * | 1 / 0 | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 1D. |
| ND_R1D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(curl) | * | * / * | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 1D. |
| ND_R2D_[DIM]_[ORDER] | H(curl) | * | 1 / 0 | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 2D. |
| ND_R2D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(curl) | * | * / * | H_CURL | 3D H(curl)-conforming Nedelec vector elements in 2D. |
| RT_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | Raviart-Thomas vector elements |
| RT@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | Raviart-Thomas vector elements |
| RT_Trace_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | INTEGRAL | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
@@ -130,9 +128,7 @@ public:
| RT_Trace@[BTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | INTEGRAL | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
| RT_ValTrace@[BTYPE]_[DIM]_[ORDER] | H^{1/2} | * | 1 / 0 | VALUE | H^{1/2}-conforming trace elements for H(div) defined on the interface between mesh elements (faces) |
| RT_R1D_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 1D. |
| RT_R1D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 1D. |
| RT_R2D_[DIM]_[ORDER] | H(div) | * | 1 / 0 | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 2D. |
| RT_R2D@[CBTYPE][OBTYPE]_[DIM]_[ORDER] | H(div) | * | * / * | H_DIV | 3D H(div)-conforming Raviart-Thomas vector elements in 2D. |
| L2_[DIM]_[ORDER] | L2 | * | 0 | VALUE | Discontinuous L2 elements |
| L2_T[BTYPE]_[DIM]_[ORDER] | L2 | * | 0 | VALUE | Discontinuous L2 elements |
| L2Int_[DIM]_[ORDER] | L2 | * | 0 | INTEGRAL | Discontinuous L2 elements |
+4 -4
View File
@@ -769,8 +769,8 @@ inline void SmemPADiffusionApply2D(const int NE,
u += Gt[dx][qx] * QQ0[qy][qx];
v += Bt[dx][qx] * QQ1[qy][qx];
}
DQ0[dx][qy] = u;
DQ1[dx][qy] = v;
DQ0[qy][dx] = u;
DQ1[qy][dx] = v;
}
}
MFEM_SYNC_THREAD;
@@ -782,8 +782,8 @@ inline void SmemPADiffusionApply2D(const int NE,
real_t v = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
u += DQ0[dx][qy] * Bt[dy][qy];
v += DQ1[dx][qy] * Gt[dy][qy];
u += DQ0[qy][dx] * Bt[dy][qy];
v += DQ1[qy][dx] * Gt[dy][qy];
}
Y(dx,dy,e) += (u + v);
}
+22 -5
View File
@@ -17,9 +17,11 @@
#include "../ceed/integrators/diffusion/diffusion.hpp"
#include "../gridfunc.hpp"
#include "../qfunction.hpp"
/// \cond DO_NOT_DOCUMENT
namespace mfem::internal
namespace mfem
{
namespace internal
{
// PA Diffusion Apply 2D kernel
@@ -331,8 +333,23 @@ PAVectorDiffusionApply3D(const int NE, const Array<real_t> &b,
}
});
}
} // namespace mfem::internal
} // namespace internal
template <int DIM, int VDIM, int T_D1D, int T_Q1D>
VectorDiffusionIntegrator::ApplyKernelType
VectorDiffusionIntegrator::ApplyPAKernels::Kernel()
{
if constexpr (DIM == 2)
{
return internal::PAVectorDiffusionApply2D<T_D1D, T_Q1D, VDIM>;
}
else if constexpr (DIM == 3)
{
return internal::PAVectorDiffusionApply3D;
}
MFEM_ABORT("");
}
} // namespace mfem
/// \endcond DO_NOT_DOCUMENT
#endif // MFEM_BILININTEG_VECDIFFUSION_KERNELS_HPP
#endif
+186 -248
View File
@@ -9,14 +9,13 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../bilininteg.hpp"
#include "../../general/forall.hpp"
#include "../bilininteg.hpp"
#include "../gridfunc.hpp"
#include "../qfunction.hpp"
#include "../ceed/integrators/diffusion/diffusion.hpp"
#include "./bilininteg_vecdiffusion_pa.hpp" // IWYU pragma: keep
// #include "bilininteg_vecdiffusion_kernels.hpp"
// #include "bilininteg_vecdiffusion_pa.hpp"
#include "bilininteg_vecdiffusion_kernels.hpp"
namespace mfem
{
@@ -24,7 +23,7 @@ namespace mfem
VectorDiffusionIntegrator::VectorDiffusionIntegrator(const IntegrationRule *ir)
: BilinearFormIntegrator(ir)
{
// static Kernels kernels;
static Kernels kernels;
}
VectorDiffusionIntegrator::VectorDiffusionIntegrator(Coefficient &q)
@@ -68,263 +67,210 @@ VectorDiffusionIntegrator::VectorDiffusionIntegrator(MatrixCoefficient &mq)
vdim = mq.GetVDim();
}
// PA Diffusion Assemble 2D kernel
static void PAVectorDiffusionSetup2D(const int Q1D,
const int NE,
const Array<real_t> &w,
const Vector &j,
const Vector &c,
Vector &op)
{
const int NQ = Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 2, 2, NE);
auto y = Reshape(op.Write(), NQ, 3, NE);
const bool const_c = c.Size() == 1;
const auto C = const_c ? Reshape(c.Read(), 1,1) :
Reshape(c.Read(), NQ, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < NQ; ++q)
{
const real_t J11 = J(q,0,0,e);
const real_t J21 = J(q,1,0,e);
const real_t J12 = J(q,0,1,e);
const real_t J22 = J(q,1,1,e);
const real_t C1 = const_c ? C(0,0) : C(q,e);
const real_t c_detJ = W[q] * C1 / ((J11*J22)-(J21*J12));
y(q,0,e) = c_detJ * (J12*J12 + J22*J22); // 1,1
y(q,1,e) = -c_detJ * (J12*J11 + J22*J21); // 1,2
y(q,2,e) = c_detJ * (J11*J11 + J21*J21); // 2,2
}
});
}
// PA Diffusion Assemble 3D kernel
static void PAVectorDiffusionSetup3D(const int Q1D,
const int NE,
const Array<real_t> &w,
const Vector &j,
const Vector &c,
Vector &op)
{
const int NQ = Q1D*Q1D*Q1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, 3, 3, NE);
auto y = Reshape(op.Write(), NQ, 6, NE);
const bool const_c = c.Size() == 1;
const auto C = const_c ? Reshape(c.Read(), 1,1) :
Reshape(c.Read(), NQ,NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
for (int q = 0; q < NQ; ++q)
{
const real_t J11 = J(q,0,0,e);
const real_t J21 = J(q,1,0,e);
const real_t J31 = J(q,2,0,e);
const real_t J12 = J(q,0,1,e);
const real_t J22 = J(q,1,1,e);
const real_t J32 = J(q,2,1,e);
const real_t J13 = J(q,0,2,e);
const real_t J23 = J(q,1,2,e);
const real_t J33 = J(q,2,2,e);
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
J21 * (J12 * J33 - J32 * J13) +
J31 * (J12 * J23 - J22 * J13);
const real_t C1 = const_c ? C(0,0) : C(q,e);
const real_t c_detJ = W[q] * C1 / detJ;
// adj(J)
const real_t A11 = (J22 * J33) - (J23 * J32);
const real_t A12 = (J32 * J13) - (J12 * J33);
const real_t A13 = (J12 * J23) - (J22 * J13);
const real_t A21 = (J31 * J23) - (J21 * J33);
const real_t A22 = (J11 * J33) - (J13 * J31);
const real_t A23 = (J21 * J13) - (J11 * J23);
const real_t A31 = (J21 * J32) - (J31 * J22);
const real_t A32 = (J31 * J12) - (J11 * J32);
const real_t A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
y(q,0,e) = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
y(q,1,e) = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
y(q,2,e) = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
y(q,3,e) = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
y(q,4,e) = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
y(q,5,e) = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
}
});
}
static void PAVectorDiffusionSetup(const int dim,
const int Q1D,
const int NE,
const Array<real_t> &W,
const Vector &J,
const Vector &C,
Vector &op)
{
if (!(dim == 2 || dim == 3))
{
MFEM_ABORT("Dimension not supported.");
}
if (dim == 2)
{
PAVectorDiffusionSetup2D(Q1D, NE, W, J, C, op);
}
if (dim == 3)
{
PAVectorDiffusionSetup3D(Q1D, NE, W, J, C, op);
}
}
void VectorDiffusionIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assumes tensor-product elements
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
const auto *ir = IntRule ? IntRule : &DiffusionIntegrator::GetRule(el, el);
const IntegrationRule *ir
= IntRule ? IntRule : &DiffusionIntegrator::GetRule(el, el);
if (DeviceCanUseCeed())
{
delete ceedOp;
const bool mixed =
mesh->GetNumGeometries(mesh->Dimension()) > 1 || fes.IsVariableOrder();
if (mixed) { ceedOp = new ceed::MixedPADiffusionIntegrator(*this, fes, Q); }
else { ceedOp = new ceed::PADiffusionIntegrator(fes, *ir, Q); }
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
fes.IsVariableOrder();
if (mixed)
{
ceedOp = new ceed::MixedPADiffusionIntegrator(*this, fes, Q);
}
else
{
ceedOp = new ceed::PADiffusionIntegrator(fes, *ir, Q);
}
return;
}
// If vdim is not set, set it to the space dimension
vdim = (vdim == -1) ? fes.GetVDim() : vdim;
MFEM_VERIFY(vdim == fes.GetVDim(), "vdim != fes.GetVDim()");
const MemoryType mt = pa_mt == MemoryType::DEFAULT
? Device::GetDeviceMemoryType()
: pa_mt;
ne = fes.GetNE();
const int dims = el.GetDim();
const int symmDims = (dims * (dims + 1)) / 2; // 1x1: 1, 2x2: 3, 3x3: 6
const int nq = ir->GetNPoints();
dim = mesh->Dimension();
sdim = mesh->SpaceDimension();
const int nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
ne = fes.GetNE();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
const int q1d = quad1D;
pa_data.SetSize(symmDims * nq * ne, Device::GetDeviceMemoryType());
if (!(dim == 2 || dim == 3)) { MFEM_ABORT("Dimension not supported."); }
MFEM_VERIFY(!VQ && !MQ,
"Only scalar coefficient supported for partial assembly for VectorDiffusionIntegrator");
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(qs, CoefficientStorage::FULL);
if (Q)
{
coeff.Project(*Q);
}
else if (VQ)
{
coeff.Project(*VQ);
MFEM_VERIFY(VQ->GetVDim() == vdim, "VQ vdim vs. vdim error");
}
else if (MQ)
{
coeff.ProjectTranspose(*MQ);
MFEM_VERIFY(MQ->GetVDim() == vdim, "MQ dimension vs. vdim error");
MFEM_VERIFY(coeff.Size() == (vdim*vdim) * ne * nq, "MQ size error");
}
else { coeff.SetConstant(1.0); }
coeff_vdim = coeff.GetVDim();
const bool scalar_coeff = coeff_vdim == 1;
const bool vector_coeff = coeff_vdim == vdim;
const bool matrix_coeff = coeff_vdim == vdim * vdim;
MFEM_VERIFY(scalar_coeff + vector_coeff + matrix_coeff == 1, "");
const int pa_size = dim * dim;
pa_data.SetSize(nq * pa_size * vdim * (matrix_coeff ? dim : 1) * ne, mt);
CoefficientVector coeff(Q, qs, CoefficientStorage::COMPRESSED);
const Array<real_t> &w = ir->GetWeights();
const Vector &j = geom->J;
Vector &d = pa_data;
if (dim == 1) { MFEM_ABORT("dim==1 not supported in PAVectorDiffusionSetup"); }
if (dim == 2 && sdim == 3)
{
MFEM_VERIFY(scalar_coeff, "");
const int nc = vdim;
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d);
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
auto D = Reshape(pa_data.Write(), q1d, q1d, pa_size,
vdim * (matrix_coeff ? dim : 1), ne);
constexpr int DIM = 2;
constexpr int SDIM = 3;
const int NQ = quad1D*quad1D;
auto W = w.Read();
auto J = Reshape(j.Read(), NQ, SDIM, DIM, ne);
auto D = Reshape(d.Write(), NQ, SDIM, ne);
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
const bool const_c = coeff.Size() == 1;
const auto C = const_c ? Reshape(coeff.Read(), 1,1) :
Reshape(coeff.Read(), NQ,ne);
mfem::forall(ne, [=] MFEM_HOST_DEVICE (int e)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
for (int i = 0; i < nc; ++i)
{
const real_t wq = W(qx, qy);
const real_t J11 = J(qx, qy, 0, 0, e);
const real_t J21 = J(qx, qy, 1, 0, e);
const real_t J31 = J(qx, qy, 2, 0, e);
const real_t J12 = J(qx, qy, 0, 1, e);
const real_t J22 = J(qx, qy, 1, 1, e);
const real_t J32 = J(qx, qy, 2, 1, e);
const real_t E = J11*J11 + J21*J21 + J31*J31;
const real_t G = J12*J12 + J22*J22 + J32*J32;
const real_t F = J11*J12 + J21*J22 + J31*J32;
const real_t iw = 1.0 / sqrt(E*G - F*F);
const auto C0 = C(0, qx, qy, e);
const real_t alpha = wq * C0 * iw;
D(qx, qy, 0, i, e) = alpha * G; // 1,1
D(qx, qy, 1, i, e) = -alpha * F; // 1,2
D(qx, qy, 2, i, e) = -alpha * F; // 2,1 == 1,2
D(qx, qy, 3, i, e) = alpha * E; // 2,2
}
}
}
});
}
else if (dim == 2 && sdim == 2)
{
const int nc = vdim, cvdim = coeff_vdim;
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d);
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
auto DE = Reshape(pa_data.Write(), q1d, q1d, pa_size,
vdim * (matrix_coeff ? dim : 1), ne);
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
const real_t J11 = J(qx, qy, 0, 0, e);
const real_t J21 = J(qx, qy, 1, 0, e);
const real_t J12 = J(qx, qy, 0, 1, e);
const real_t J22 = J(qx, qy, 1, 1, e);
const real_t w_detJ = W(qx, qy) / ((J11*J22)-(J21*J12));
const real_t D0 = w_detJ * (J12*J12 + J22*J22);
const real_t D1 = -w_detJ * (J12*J11 + J22*J21);
const real_t D2 = w_detJ * (J11*J11 + J21*J21);
const int map[4] = {0, 2, 1, 3};
for (int i = 0; i < (matrix_coeff ? cvdim : nc); ++i)
{
const auto k = matrix_coeff ? map[i] : (vector_coeff ? i : 0);
const auto Cc = C(k, qx, qy, e);
DE(qx, qy, 0, i, e) = D0 * Cc;
DE(qx, qy, 1, i, e) = D1 * Cc;
DE(qx, qy, 2, i, e) = D1 * Cc;
DE(qx, qy, 3, i, e) = D2 * Cc;
}
}
}
});
}
else if (dim == 3 && sdim == 3)
{
const int nc = vdim, cvdim = coeff_vdim;
const auto W = Reshape(ir->GetWeights().Read(), q1d, q1d, q1d);
const auto J = Reshape(geom->J.Read(), q1d, q1d, q1d, sdim, dim, ne);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, q1d, ne);
auto DE = Reshape(pa_data.Write(), q1d, q1d, q1d, pa_size,
vdim * (matrix_coeff ? dim : 1), ne);
mfem::forall_3D(ne, q1d, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
const real_t J11 = J(qx, qy, qz, 0, 0, e);
const real_t J21 = J(qx, qy, qz, 1, 0, e);
const real_t J31 = J(qx, qy, qz, 2, 0, e);
const real_t J12 = J(qx, qy, qz, 0, 1, e);
const real_t J22 = J(qx, qy, qz, 1, 1, e);
const real_t J32 = J(qx, qy, qz, 2, 1, e);
const real_t J13 = J(qx, qy, qz, 0, 2, e);
const real_t J23 = J(qx, qy, qz, 1, 2, e);
const real_t J33 = J(qx, qy, qz, 2, 2, e);
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
J21 * (J12 * J33 - J32 * J13) +
J31 * (J12 * J23 - J22 * J13);
const real_t c_detJ = W(qx, qy, qz) / detJ;
// adj(J)
const real_t A11 = (J22 * J33) - (J23 * J32);
const real_t A12 = (J32 * J13) - (J12 * J33);
const real_t A13 = (J12 * J23) - (J22 * J13);
const real_t A21 = (J31 * J23) - (J21 * J33);
const real_t A22 = (J11 * J33) - (J13 * J31);
const real_t A23 = (J21 * J13) - (J11 * J23);
const real_t A31 = (J21 * J32) - (J31 * J22);
const real_t A32 = (J31 * J12) - (J11 * J32);
const real_t A33 = (J11 * J22) - (J12 * J21);
// detJ J^{-1} J^{-T} = (1/detJ) adj(J) adj(J)^T
const real_t D11 = c_detJ * (A11*A11 + A12*A12 + A13*A13); // 1,1
const real_t D21 = c_detJ * (A11*A21 + A12*A22 + A13*A23); // 2,1
const real_t D31 = c_detJ * (A11*A31 + A12*A32 + A13*A33); // 3,1
const real_t D22 = c_detJ * (A21*A21 + A22*A22 + A23*A23); // 2,2
const real_t D32 = c_detJ * (A21*A31 + A22*A32 + A23*A33); // 3,2
const real_t D33 = c_detJ * (A31*A31 + A32*A32 + A33*A33); // 3,3
const int map[9] = {0, 3, 6, 1, 4, 7, 2, 5, 8};
for (int i = 0; i < (matrix_coeff ? cvdim : nc); ++i)
{
const auto k = matrix_coeff ? map[i] : vector_coeff ? i : 0;
const auto Ck = C(k, qx, qy, qz, e);
DE(qx, qy, qz, 0, i, e) = D11 * Ck;
DE(qx, qy, qz, 1, i, e) = D21 * Ck;
DE(qx, qy, qz, 2, i, e) = D31 * Ck;
DE(qx, qy, qz, 3, i, e) = D22 * Ck;
DE(qx, qy, qz, 4, i, e) = D32 * Ck;
DE(qx, qy, qz, 5, i, e) = D33 * Ck;
}
}
}
const real_t wq = W[q];
const real_t J11 = J(q,0,0,e);
const real_t J21 = J(q,1,0,e);
const real_t J31 = J(q,2,0,e);
const real_t J12 = J(q,0,1,e);
const real_t J22 = J(q,1,1,e);
const real_t J32 = J(q,2,1,e);
const real_t E = J11*J11 + J21*J21 + J31*J31;
const real_t G = J12*J12 + J22*J22 + J32*J32;
const real_t F = J11*J12 + J21*J22 + J31*J32;
const real_t iw = 1.0 / sqrt(E*G - F*F);
const real_t C1 = const_c ? C(0,0) : C(q,e);
const real_t alpha = wq * C1 * iw;
D(q,0,e) = alpha * G; // 1,1
D(q,1,e) = -alpha * F; // 1,2
D(q,2,e) = alpha * E; // 2,2
}
});
}
else
{
MFEM_ABORT("Unknown VectorDiffusionIntegrator::AssemblePA kernel for"
<< " dim:" << dim << ", vdim:" << vdim << ", sdim:" << sdim);
PAVectorDiffusionSetup(dim, quad1D, ne, w, j, coeff, d);
}
}
// PA Diffusion Apply kernel
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
// Use CEED backend if available
if (DeviceCanUseCeed()) { return ceedOp->AddMult(x, y); }
// Add the VectorDiffusionAddMultPA specializations
static const auto vector_diffusion_kernel_specializations =
(
// 2D, SDIM = 2
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 2,2>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 3,3>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 4,4>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 5,5>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 6,6>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 7,7>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 8,8>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,2, 9,9>::Add(),
// 2D, SDIM = 3
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 2,2>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 3,3>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 4,4>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<2,3, 5,5>::Add(),
// 3D, SDIM = 3
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 2,2>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 2,3>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 3,4>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 4,5>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 4,6>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 5,6>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 5,8>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 6,7>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 7,8>::Add(),
VectorDiffusionIntegrator::ApplyPAKernels::Specialization<3,3, 8,9>::Add(),
true);
MFEM_CONTRACT_VAR(vector_diffusion_kernel_specializations);
ApplyPAKernels::Run(dim, sdim, dofs1D, quad1D,
ne, coeff_vdim, maps->B, maps->G, pa_data, x, y,
sdim, dofs1D, quad1D);
}
template<int T_D1D = 0, int T_Q1D = 0>
static void PAVectorDiffusionDiagonal2D(const int NE,
const Array<real_t> &b,
@@ -338,15 +284,12 @@ static void PAVectorDiffusionDiagonal2D(const int NE,
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto G = Reshape(g.Read(), Q1D, D1D);
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
// note the different shape for D, this is a (symmetric) matrix so we only
// store necessary entries
MFEM_VERIFY(d.Size() == Q1D*Q1D*4*2*NE, "");
const auto D = Reshape(d.Read(), Q1D*Q1D, /*3*/4, 2, NE);
auto D = Reshape(d.Read(), Q1D*Q1D, 3, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, 2, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
@@ -367,9 +310,9 @@ static void PAVectorDiffusionDiagonal2D(const int NE,
for (int qy = 0; qy < Q1D; ++qy)
{
const int q = qx + qy * Q1D;
const real_t D0 = D(q,0,0,e);
const real_t D1 = D(q,1,0,e);
const real_t D2 = D(q,3/*2*/,0,e); // size from 3 (symmetric) to 4 (dims x dims)
const real_t D0 = D(q,0,e);
const real_t D1 = D(q,1,e);
const real_t D2 = D(q,2,e);
QD0[qx][dy] += B(qy, dy) * B(qy, dy) * D0;
QD1[qx][dy] += B(qy, dy) * G(qy, dy) * D1;
QD2[qx][dy] += G(qy, dy) * G(qy, dy) * D2;
@@ -413,8 +356,7 @@ static void PAVectorDiffusionDiagonal3D(const int NE,
MFEM_VERIFY(Q1D <= max_q1d, "");
auto B = Reshape(b.Read(), Q1D, D1D);
auto G = Reshape(g.Read(), Q1D, D1D);
MFEM_VERIFY(d.Size() == Q1D*Q1D*Q1D*9*3*NE, "");
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 9/*PA_SIZE:dims*dims*/, 3/*VDIM*/, NE);
auto Q = Reshape(d.Read(), Q1D*Q1D*Q1D, 6, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, 3, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
@@ -442,8 +384,7 @@ static void PAVectorDiffusionDiagonal3D(const int NE,
const int k = j >= i ?
3 - (3-i)*(2-i)/2 + j:
3 - (3-j)*(2-j)/2 + i;
// using 6 symmetric values
const real_t O = Q(q,k,0,e);
const real_t O = Q(q,k,e);
const real_t Bz = B(qz,dz);
const real_t Gz = G(qz,dz);
const real_t L = i==2 ? Gz : Bz;
@@ -527,14 +468,12 @@ void VectorDiffusionIntegrator::AssembleDiagonalPA(Vector &diag)
}
else
{
MFEM_VERIFY(!VQ && !MQ, "VQ and MQ not supported.");
PAVectorDiffusionAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->G,
pa_data, diag);
}
}
/*
// PA Diffusion Apply kernel
void VectorDiffusionIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
@@ -575,6 +514,5 @@ VectorDiffusionIntegrator::Kernels::Kernels()
}
/// \endcond DO_NOT_DOCUMENT
*/
} // namespace mfem
-202
View File
@@ -1,202 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../config/config.hpp"
#include "../../general/array.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "../kernels.hpp"
using mfem::kernels::internal::SetMaxOf;
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
template<int T_SDIM = 0, int T_D1D = 0, int T_Q1D = 0>
void SmemPAVectorDiffusionApply2D(const int NE,
const int coeff_vdim,
const Array<real_t> &b,
const Array<real_t> &g,
const Vector &d,
const Vector &x,
Vector &y,
const int sdim = 0,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int DIM = 2;
const int SDIM = T_SDIM ? T_SDIM : sdim;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int PA_SIZE = DIM*DIM;
const bool matrix_coeff = coeff_vdim == DIM*DIM;
const auto B = b.Read(), G = g.Read();
const auto DE = Reshape(d.Read(), Q1D, Q1D, PA_SIZE,
SDIM * (matrix_coeff ? SDIM : 1), NE);
const auto XE = Reshape(x.Read(), D1D, D1D, SDIM, NE);
auto YE = Reshape(y.ReadWrite(), D1D, D1D, SDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1], smem[MQ1][MQ1];
kernels::internal::vd_regs2d_t<3, DIM, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadMatrix(D1D, Q1D, G, sG);
for (int i = 0; i < SDIM; i++)
{
for (int j = 0; j < (matrix_coeff ? SDIM : 1); j++)
{
kernels::internal::LoadDofs2d(e, D1D, i, XE, r0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1, i);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t gradX = r1[i][0][qy][qx];
const real_t gradY = r1[i][1][qy][qx];
const int k = matrix_coeff ? (j + i * SDIM) : i;
const real_t O11 = DE(qx,qy,0,k,e), O12 = DE(qx,qy,1,k,e);
const real_t O21 = DE(qx,qy,2,k,e), O22 = DE(qx,qy,3,k,e);
r0[i][0][qy][qx] = (O11 * gradX) + (O12 * gradY);
r0[i][1][qy][qx] = (O21 * gradX) + (O22 * gradY);
} // qx
} // qy
MFEM_SYNC_THREAD;
kernels::internal::GradTranspose2d(D1D, Q1D, smem, sB, sG, r0, r1, i);
const int ij = matrix_coeff ? j : i;
kernels::internal::WriteDofs2d(e, D1D, i, ij, r1, YE);
} // j
} // i
});
}
template<int T_SDIM = 0, int T_D1D = 0, int T_Q1D = 0>
void SmemPAVectorDiffusionApply3D(const int NE,
const int coeff_vdim,
const Array<real_t> &b,
const Array<real_t> &g,
const Vector &d,
const Vector &x,
Vector &y,
const int sdim = 0,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int DIM = 3;
const int SDIM = T_SDIM ? T_SDIM : sdim;
MFEM_VERIFY(SDIM == 3, "SDIM must be 3");
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const int PA_SIZE = DIM*DIM;
const bool matrix_coeff = coeff_vdim == DIM*DIM;
const auto B = b.Read(), G = g.Read();
const auto DE = Reshape(d.Read(), Q1D, Q1D, Q1D, PA_SIZE,
SDIM * (matrix_coeff ? SDIM : 1), NE);
const auto XE = Reshape(x.Read(), D1D, D1D, D1D, SDIM, NE);
auto YE = Reshape(y.ReadWrite(), D1D, D1D, D1D, SDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1], smem[MQ1][MQ1];
kernels::internal::vd_regs3d_t<3, DIM, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadMatrix(D1D, Q1D, G, sG);
for (int i = 0; i < SDIM; i++)
{
for (int j = 0; j < (matrix_coeff ? SDIM : 1); j++)
{
kernels::internal::LoadDofs3d(e, D1D, i, XE, r0);
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1, i);
for (int qz = 0; qz < Q1D; qz++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t gradX = r1[i][0][qz][qy][qx];
const real_t gradY = r1[i][1][qz][qy][qx];
const real_t gradZ = r1[i][2][qz][qy][qx];
const int k = matrix_coeff ? (j + i * SDIM) : i;
const real_t O11 = DE(qx,qy,qz,0,k,e), O12 = DE(qx,qy,qz,1,k,e),
O13 = DE(qx,qy,qz,2,k,e);
const real_t O22 = DE(qx,qy,qz,3,k,e), O23 = DE(qx,qy,qz,4,k,e);
const real_t O33 = DE(qx,qy,qz,5,k,e);
r0[i][0][qz][qy][qx] = (O11*gradX)+(O12*gradY)+(O13*gradZ);
r0[i][1][qz][qy][qx] = (O12*gradX)+(O22*gradY)+(O23*gradZ);
r0[i][2][qz][qy][qx] = (O13*gradX)+(O23*gradY)+(O33*gradZ);
} // qx
} // qy
} // qz
MFEM_SYNC_THREAD;
kernels::internal::GradTranspose3d(D1D, Q1D, smem, sB, sG, r0, r1, i);
const int ij = matrix_coeff ? j : i;
kernels::internal::WriteDofs3d(e, D1D, i, ij, r1, YE);
} // j
} // i
});
}
} // namespace internal
template<int DIM, int T_SDIM, int T_D1D, int T_Q1D>
VectorDiffusionIntegrator::ApplyKernelType
VectorDiffusionIntegrator::ApplyPAKernels::Kernel()
{
if (DIM == 2)
{
return internal::SmemPAVectorDiffusionApply2D<T_SDIM, T_D1D, T_Q1D>;
}
else if (DIM == 3)
{
return internal::SmemPAVectorDiffusionApply3D<T_SDIM, T_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorDiffusionIntegrator::ApplyKernelType
VectorDiffusionIntegrator::ApplyPAKernels::Fallback(int dim, int sdim,
int d1d, int q1d)
{
if (dim == 2)
{
return internal::SmemPAVectorDiffusionApply2D;
}
else if (dim == 3)
{
return internal::SmemPAVectorDiffusionApply3D;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
+381 -198
View File
@@ -9,218 +9,120 @@
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../bilininteg.hpp"
#include "../../general/forall.hpp"
#include "../bilininteg.hpp"
#include "../gridfunc.hpp"
#include "../ceed/integrators/mass/mass.hpp"
#include "./bilininteg_vecmass_pa.hpp" // IWYU pragma: keep
namespace mfem
{
void VectorMassIntegrator::AssemblePA(const FiniteElementSpace &fes)
{
// Assuming the same element type
Mesh *mesh = fes.GetMesh();
const FiniteElement &el = *fes.GetTypicalFE();
ElementTransformation &Trans = *mesh->GetTypicalElementTransformation();
const auto *ir = IntRule ? IntRule : &MassIntegrator::GetRule(el, el, Trans);
ElementTransformation *T = mesh->GetTypicalElementTransformation();
const IntegrationRule *ir
= IntRule ? IntRule : &MassIntegrator::GetRule(el, el, *T);
if (DeviceCanUseCeed())
{
delete ceedOp;
const bool mixed =
mesh->GetNumGeometries(mesh->Dimension()) > 1 || fes.IsVariableOrder();
if (mixed) { ceedOp = new ceed::MixedPAMassIntegrator(*this, fes, Q); }
else { ceedOp = new ceed::PAMassIntegrator(fes, *ir, Q); }
const bool mixed = mesh->GetNumGeometries(mesh->Dimension()) > 1 ||
fes.IsVariableOrder();
if (mixed)
{
ceedOp = new ceed::MixedPAMassIntegrator(*this, fes, Q);
}
else
{
ceedOp = new ceed::PAMassIntegrator(fes, *ir, Q);
}
return;
}
// If vdim is not set, set it to the space dimension
vdim = (vdim == -1) ? Trans.GetSpaceDim() : vdim;
MFEM_VERIFY(vdim == fes.GetVDim(), "vdim != fes.GetVDim()");
MFEM_VERIFY(vdim == mesh->Dimension(), "vdim != dim");
const MemoryType mt = pa_mt == MemoryType::DEFAULT
? Device::GetDeviceMemoryType()
: pa_mt;
ne = mesh->GetNE();
dim = mesh->Dimension();
const int nq = ir->GetNPoints();
const int sdim = mesh->SpaceDimension();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::JACOBIANS, mt);
ne = fes.GetMesh()->GetNE();
nq = ir->GetNPoints();
geom = mesh->GetGeometricFactors(*ir, GeometricFactors::COORDINATES |
GeometricFactors::JACOBIANS);
maps = &el.GetDofToQuad(*ir, DofToQuad::TENSOR);
dofs1D = maps->ndof;
quad1D = maps->nqpt;
const int q1d = quad1D;
if (!(dim == 2 || dim == 3)) { MFEM_ABORT("Dimension not supported."); }
QuadratureSpace qs(*mesh, *ir);
CoefficientVector coeff(qs);
pa_data.SetSize(ne*nq, Device::GetDeviceMemoryType());
real_t coeff = 1.0;
if (Q)
{
coeff.Project(*Q);
ConstantCoefficient *cQ = dynamic_cast<ConstantCoefficient*>(Q);
MFEM_VERIFY(cQ != NULL, "Only ConstantCoefficient is supported.");
coeff = cQ->constant;
}
else if (VQ)
if (!(dim == 2 || dim == 3))
{
coeff.Project(*VQ);
MFEM_VERIFY(VQ->GetVDim() == vdim, "VQ vdim vs. vdim error");
MFEM_ABORT("Dimension not supported.");
}
else if (MQ)
{
coeff.ProjectTranspose(*MQ);
MFEM_VERIFY(MQ->GetVDim() == vdim, "MQ dimension vs. vdim error");
MFEM_VERIFY(coeff.Size() == (vdim*vdim) * ne * nq, "MQ size error");
}
else { coeff.SetConstant(1.0); }
coeff_vdim = coeff.GetVDim();
const bool const_coeff = coeff_vdim == 1;
const bool vector_coeff = coeff_vdim == vdim;
const bool matrix_coeff = coeff_vdim == vdim * vdim;
MFEM_VERIFY(const_coeff + vector_coeff + matrix_coeff == 1, "");
pa_data.SetSize(coeff_vdim * nq * ne, mt);
const auto w_r = ir->GetWeights().Read();
if (dim == 2)
{
const auto W = Reshape(w_r, q1d, q1d);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, ne);
const auto J = Reshape(geom->J.Read(), q1d, q1d, sdim, dim, ne);
auto D = Reshape(pa_data.Write(), q1d, q1d, coeff_vdim, ne);
mfem::forall_2D(ne, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
const real_t constant = coeff;
const int NE = ne;
const int NQ = nq;
auto w = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,2,2,NE);
auto v = Reshape(pa_data.Write(), NQ, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
const real_t J11 = J(qx, qy, 0, 0, e), J12 = J(qx, qy, 1, 0, e);
const real_t J21 = J(qx, qy, 0, 1, e), J22 = J(qx, qy, 1, 1, e);
const real_t detJ = (J11 * J22) - (J21 * J12);
const real_t w_det = W(qx, qy) * detJ;
D(qx, qy, 0, e) = C(0, qx, qy, e) * w_det;
if (const_coeff) { continue; }
D(qx, qy, 1, e) = C(1, qx, qy, e) * w_det;
if (vector_coeff) { continue; }
assert(matrix_coeff);
D(qx, qy, 2, e) = C(2, qx, qy, e) * w_det;
D(qx, qy, 3, e) = C(3, qx, qy, e) * w_det;
}
const real_t J11 = J(q,0,0,e);
const real_t J12 = J(q,1,0,e);
const real_t J21 = J(q,0,1,e);
const real_t J22 = J(q,1,1,e);
const real_t detJ = (J11*J22)-(J21*J12);
v(q,e) = w[q] * constant * detJ;
}
});
}
else if (dim == 3)
if (dim == 3)
{
const auto W = Reshape(w_r, q1d, q1d, q1d);
const auto C = Reshape(coeff.Read(), coeff_vdim, q1d, q1d, q1d, ne);
const auto J = Reshape(geom->J.Read(), q1d, q1d, q1d, sdim, dim, ne);
auto D = Reshape(pa_data.Write(), q1d, q1d, q1d, coeff_vdim, ne);
mfem::forall_3D(ne, q1d, q1d, q1d, [=] MFEM_HOST_DEVICE(int e)
const real_t constant = coeff;
const int NE = ne;
const int NQ = nq;
auto W = ir->GetWeights().Read();
auto J = Reshape(geom->J.Read(), NQ,3,3,NE);
auto v = Reshape(pa_data.Write(), NQ,NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
MFEM_FOREACH_THREAD(qz, z, q1d)
for (int q = 0; q < NQ; ++q)
{
MFEM_FOREACH_THREAD(qy, y, q1d)
{
MFEM_FOREACH_THREAD(qx, x, q1d)
{
const real_t J11 = J(qx, qy, qz, 0, 0, e),
J12 = J(qx, qy, qz, 0, 1, e),
J13 = J(qx, qy, qz, 0, 2, e);
const real_t J21 = J(qx, qy, qz, 1, 0, e),
J22 = J(qx, qy, qz, 1, 1, e),
J23 = J(qx, qy, qz, 1, 2, e);
const real_t J31 = J(qx, qy, qz, 2, 0, e),
J32 = J(qx, qy, qz, 2, 1, e),
J33 = J(qx, qy, qz, 2, 2, e);
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
J21 * (J12 * J33 - J32 * J13) +
J31 * (J12 * J23 - J22 * J13);
const real_t w_det = W(qx, qy, qz) * detJ;
D(qx, qy, qz, 0, e) = C(0, qx, qy, qz, e) * w_det;
if (const_coeff) { continue; }
D(qx, qy, qz, 1, e) = C(1, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 2, e) = C(2, qx, qy, qz, e) * w_det;
if (vector_coeff) { continue; }
D(qx, qy, qz, 3, e) = C(3, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 4, e) = C(4, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 5, e) = C(5, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 6, e) = C(6, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 7, e) = C(7, qx, qy, qz, e) * w_det;
D(qx, qy, qz, 8, e) = C(8, qx, qy, qz, e) * w_det;
}
}
const real_t J11 = J(q,0,0,e), J12 = J(q,0,1,e), J13 = J(q,0,2,e);
const real_t J21 = J(q,1,0,e), J22 = J(q,1,1,e), J23 = J(q,1,2,e);
const real_t J31 = J(q,2,0,e), J32 = J(q,2,1,e), J33 = J(q,2,2,e);
const real_t detJ = J11 * (J22 * J33 - J32 * J23) -
J21 * (J12 * J33 - J32 * J13) +
J31 * (J12 * J23 - J22 * J13);
v(q,e) = W[q] * constant * detJ;
}
});
}
else
{
MFEM_ABORT("Unknown VectorMassIntegrator::AssemblePA kernel for"
<< " dim:" << dim << ", vdim:" << vdim << ", sdim:" << sdim);
}
}
void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
// Use CEED backend if available
if (DeviceCanUseCeed()) { return ceedOp->AddMult(x, y); }
// Add the VectorMassAddMultPA specializations
static const auto vector_mass_kernel_specializations =
( // 2D
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 2,2>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 3,3>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 3,4>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 4,4>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 4,6>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 5,5>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 6,6>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 7,7>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 8,8>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<2, 9,9>::Add(),
// 3D
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 2,2>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 2,3>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 3,4>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 3,5>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,5>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,6>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 4,8>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 5,6>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 5,8>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 6,7>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 7,8>::Add(),
VectorMassIntegrator::VectorMassAddMultPA::Specialization<3, 8,9>::Add(),
true);
MFEM_CONTRACT_VAR(vector_mass_kernel_specializations);
VectorMassAddMultPA::Run(dim, dofs1D, quad1D,
ne, coeff_vdim, maps->B, pa_data, x, y,
dofs1D, quad1D);
}
template <const int T_D1D = 0, const int T_Q1D = 0>
template<const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassAssembleDiagonal2D(const int NE,
const Array<real_t> &b,
const Vector &pa_data, Vector &diag,
const int d1d = 0, const int q1d = 0)
const Array<real_t> &B_,
const Array<real_t> &Bt_,
const Vector &op_,
Vector &diag_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int VDIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 2;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(b.Read(), Q1D, D1D);
const auto D = Reshape(pa_data.Read(), Q1D, Q1D, NE);
auto Y = Reshape(diag.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
auto B = Reshape(B_.Read(), Q1D, D1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
@@ -235,7 +137,7 @@ static void PAVectorMassAssembleDiagonal2D(const int NE,
temp[qx][dy] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
temp[qx][dy] += B(qy, dy) * B(qy, dy) * D(qx, qy, e);
temp[qx][dy] += B(qy, dy) * B(qy, dy) * op(qx, qy, e);
}
}
}
@@ -248,31 +150,33 @@ static void PAVectorMassAssembleDiagonal2D(const int NE,
{
temp1 += B(qx, dx) * B(qx, dx) * temp[qx][dy];
}
Y(dx, dy, 0, e) = temp1;
Y(dx, dy, 1, e) = temp1;
y(dx, dy, 0, e) = temp1;
y(dx, dy, 1, e) = temp1;
}
}
});
}
template <const int T_D1D = 0, const int T_Q1D = 0>
template<const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassAssembleDiagonal3D(const int NE,
const Array<real_t> &B_,
const Vector &pa_data, Vector &diag,
const int d1d = 0, const int q1d = 0)
const Array<real_t> &Bt_,
const Vector &op_,
Vector &diag_,
const int d1d = 0,
const int q1d = 0)
{
constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 3;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(B_.Read(), Q1D, D1D);
MFEM_VERIFY(pa_data.Size() == Q1D * Q1D * Q1D * NE, "pa_data size error");
const auto D = Reshape(pa_data.Read(), Q1D, Q1D, Q1D, NE);
auto Y = Reshape(diag.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE(int e)
auto B = Reshape(B_.Read(), Q1D, D1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
auto y = Reshape(diag_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
@@ -288,8 +192,7 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
temp[qx][qy][dz] = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
temp[qx][qy][dz] +=
B(qz, dz) * B(qz, dz) * D(qx, qy, qz, e);
temp[qx][qy][dz] += B(qz, dz) * B(qz, dz) * op(qx, qy, qz, e);
}
}
}
@@ -304,8 +207,7 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
temp2[qx][dy][dz] = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
temp2[qx][dy][dz] +=
B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
temp2[qx][dy][dz] += B(qy, dy) * B(qy, dy) * temp[qx][qy][dz];
}
}
}
@@ -319,42 +221,323 @@ static void PAVectorMassAssembleDiagonal3D(const int NE,
real_t temp3 = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
temp3 += B(qx, dx) * B(qx, dx) * temp2[qx][dy][dz];
temp3 += B(qx, dx) * B(qx, dx)
* temp2[qx][dy][dz];
}
Y(dx, dy, dz, 0, e) = temp3;
Y(dx, dy, dz, 1, e) = temp3;
Y(dx, dy, dz, 2, e) = temp3;
y(dx, dy, dz, 0, e) = temp3;
y(dx, dy, dz, 1, e) = temp3;
y(dx, dy, dz, 2, e) = temp3;
}
}
}
});
}
static void PAVectorMassAssembleDiagonal(const int dim, const int D1D,
const int Q1D, const int NE,
static void PAVectorMassAssembleDiagonal(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<real_t> &B,
const Vector &pa_data,
Vector &diag)
const Array<real_t> &Bt,
const Vector &op,
Vector &y)
{
if (dim == 2)
{
return PAVectorMassAssembleDiagonal2D(NE, B, pa_data, diag, D1D, Q1D);
return PAVectorMassAssembleDiagonal2D(NE, B, Bt, op, y, D1D, Q1D);
}
else if (dim == 3)
{
return PAVectorMassAssembleDiagonal3D(NE, B, pa_data, diag, D1D, Q1D);
return PAVectorMassAssembleDiagonal3D(NE, B, Bt, op, y, D1D, Q1D);
}
MFEM_ABORT("Dimension not implemented.");
}
void VectorMassIntegrator::AssembleDiagonalPA(Vector &diag)
{
if (DeviceCanUseCeed()) { ceedOp->GetDiagonal(diag); }
if (DeviceCanUseCeed())
{
ceedOp->GetDiagonal(diag);
}
else
{
MFEM_VERIFY(coeff_vdim == 1, "coeff_vdim != 1");
MFEM_VERIFY(!VQ && !MQ, "VQ and MQ not supported");
PAVectorMassAssembleDiagonal(dim, dofs1D, quad1D, ne, maps->B, pa_data, diag);
PAVectorMassAssembleDiagonal(dim, dofs1D, quad1D, ne,
maps->B, maps->Bt,
pa_data, diag);
}
}
template<const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassApply2D(const int NE,
const Array<real_t> &B_,
const Array<real_t> &Bt_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 2;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(B_.Read(), Q1D, D1D);
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, VDIM, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d; // nvcc workaround
const int Q1D = T_Q1D ? T_Q1D : q1d;
// the following variables are evaluated at compile time
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
real_t sol_xy[max_Q1D][max_Q1D];
for (int c = 0; c < VDIM; ++c)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
real_t sol_x[max_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
sol_x[qy] = 0.0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const real_t s = x(dx,dy,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx)* s;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t d2q = B(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] += d2q * sol_x[qx];
}
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] *= op(qx,qy,e);
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
real_t sol_x[max_D1D];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] = 0.0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t s = sol_xy[qy][qx];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] += Bt(dx,qx) * s;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const real_t q2d = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,c,e) += q2d * sol_x[dx];
}
}
}
}
});
}
template<const int T_D1D = 0, const int T_Q1D = 0>
static void PAVectorMassApply3D(const int NE,
const Array<real_t> &B_,
const Array<real_t> &Bt_,
const Vector &op_,
const Vector &x_,
Vector &y_,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int VDIM = 3;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
auto B = Reshape(B_.Read(), Q1D, D1D);
auto Bt = Reshape(Bt_.Read(), D1D, Q1D);
auto op = Reshape(op_.Read(), Q1D, Q1D, Q1D, NE);
auto x = Reshape(x_.Read(), D1D, D1D, D1D, VDIM, NE);
auto y = Reshape(y_.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall(NE, [=] MFEM_HOST_DEVICE (int e)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
constexpr int max_D1D = T_D1D ? T_D1D : DofQuadLimits::MAX_D1D;
constexpr int max_Q1D = T_Q1D ? T_Q1D : DofQuadLimits::MAX_Q1D;
real_t sol_xyz[max_Q1D][max_Q1D][max_Q1D];
for (int c = 0; c < VDIM; ++ c)
{
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] = 0.0;
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
real_t sol_xy[max_Q1D][max_Q1D];
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] = 0.0;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
real_t sol_x[max_Q1D];
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] = 0;
}
for (int dx = 0; dx < D1D; ++dx)
{
const real_t s = x(dx,dy,dz,c,e);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_x[qx] += B(qx,dx) * s;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t wy = B(qy,dy);
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xy[qy][qx] += wy * sol_x[qx];
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const real_t wz = B(qz,dz);
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] += wz * sol_xy[qy][qx];
}
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
for (int qy = 0; qy < Q1D; ++qy)
{
for (int qx = 0; qx < Q1D; ++qx)
{
sol_xyz[qz][qy][qx] *= op(qx,qy,qz,e);
}
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
real_t sol_xy[max_D1D][max_D1D];
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
sol_xy[dy][dx] = 0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
real_t sol_x[max_D1D];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] = 0;
}
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t s = sol_xyz[qz][qy][qx];
for (int dx = 0; dx < D1D; ++dx)
{
sol_x[dx] += Bt(dx,qx) * s;
}
}
for (int dy = 0; dy < D1D; ++dy)
{
const real_t wy = Bt(dy,qy);
for (int dx = 0; dx < D1D; ++dx)
{
sol_xy[dy][dx] += wy * sol_x[dx];
}
}
}
for (int dz = 0; dz < D1D; ++dz)
{
const real_t wz = Bt(dz,qz);
for (int dy = 0; dy < D1D; ++dy)
{
for (int dx = 0; dx < D1D; ++dx)
{
y(dx,dy,dz,c,e) += wz * sol_xy[dy][dx];
}
}
}
}
}
});
}
static void PAVectorMassApply(const int dim,
const int D1D,
const int Q1D,
const int NE,
const Array<real_t> &B,
const Array<real_t> &Bt,
const Vector &op,
const Vector &x,
Vector &y)
{
if (dim == 2)
{
return PAVectorMassApply2D(NE, B, Bt, op, x, y, D1D, Q1D);
}
if (dim == 3)
{
return PAVectorMassApply3D(NE, B, Bt, op, x, y, D1D, Q1D);
}
MFEM_ABORT("Unknown kernel.");
}
void VectorMassIntegrator::AddMultPA(const Vector &x, Vector &y) const
{
if (DeviceCanUseCeed())
{
ceedOp->AddMult(x, y);
}
else
{
PAVectorMassApply(dim, dofs1D, quad1D, ne, maps->B, maps->Bt, pa_data, x, y);
}
}
-212
View File
@@ -1,212 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../config/config.hpp"
#include "../../general/array.hpp"
#include "../../general/forall.hpp"
#include "../../linalg/dtensor.hpp"
#include "../../linalg/vector.hpp"
#include "../bilininteg.hpp"
#include "../kernels.hpp"
using mfem::kernels::internal::SetMaxOf;
namespace mfem
{
/// \cond DO_NOT_DOCUMENT
namespace internal
{
template <int T_D1D = 0, int T_Q1D = 0>
void SmemPAVectorMassApply2D(const int NE,
const int coeff_vdim,
const Array<real_t> &b,
const Vector &d,
const Vector &x,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int DIM = 2, VDIM = 2;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const bool const_coeff = coeff_vdim == 1;
const bool vector_coeff = coeff_vdim == DIM;
const bool matrix_coeff = coeff_vdim == DIM*DIM;
const auto B = b.Read();
const auto D = Reshape(d.Read(), Q1D, Q1D, coeff_vdim, NE);
const auto X = Reshape(x.Read(), D1D, D1D, VDIM, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, VDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
MFEM_SHARED real_t sB[MD1][MQ1], smem[MQ1][MQ1];
kernels::internal::v_regs2d_t<VDIM, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadDofs2d(e, D1D, X, r0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r0, r1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t Qx = r1[0][qy][qx];
const real_t Qy = r1[1][qy][qx];
const real_t D0 = D(qx, qy, 0, e);
if (const_coeff)
{
r0[0][qy][qx] = D0 * Qx;
r0[1][qy][qx] = D0 * Qy;
}
if (vector_coeff)
{
const real_t D1 = D(qx, qy, 1, e);
r0[0][qy][qx] = D0 * Qx;
r0[1][qy][qx] = D1 * Qy;
}
if (matrix_coeff)
{
const real_t D1 = D(qx, qy, 1, e);
const real_t D2 = D(qx, qy, 2, e);
const real_t D3 = D(qx, qy, 3, e);
r0[0][qy][qx] = D0 * Qx + D1 * Qy;
r0[1][qy][qx] = D2 * Qx + D3 * Qy;
}
}
}
kernels::internal::EvalTranspose2d(D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
});
}
template <int T_D1D = 0, int T_Q1D = 0>
void SmemPAVectorMassApply3D(const int NE,
const int coeff_vdim,
const Array<real_t> &b,
const Vector &d,
const Vector &x,
Vector &y,
const int d1d = 0,
const int q1d = 0)
{
static constexpr int VDIM = 3;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
const bool const_coeff = coeff_vdim == 1;
const bool vector_coeff = coeff_vdim == VDIM;
const bool matrix_coeff = coeff_vdim == VDIM*VDIM;
const auto B = b.Read();
const auto D = Reshape(d.Read(), Q1D, Q1D, Q1D, coeff_vdim, NE);
const auto X = Reshape(x.Read(), D1D, D1D, D1D, VDIM, NE);
auto Y = Reshape(y.ReadWrite(), D1D, D1D, D1D, VDIM, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
constexpr int MD1 = T_D1D > 0 ? SetMaxOf(T_D1D) : DofQuadLimits::MAX_T1D;
constexpr int MQ1 = T_Q1D > 0 ? SetMaxOf(T_Q1D) : DofQuadLimits::MAX_T1D;
MFEM_SHARED real_t sB[MD1][MQ1], smem[MQ1][MQ1];
kernels::internal::v_regs3d_t<VDIM, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, B, sB);
kernels::internal::LoadDofs3d(e, D1D, X, r0);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r0, r1);
for (int qz = 0; qz < Q1D; qz++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t Qx = r1[0][qz][qy][qx];
const real_t Qy = r1[1][qz][qy][qx];
const real_t Qz = r1[2][qz][qy][qx];
const real_t D0 = D(qx, qy, qz, 0, e);
if (const_coeff)
{
r0[0][qz][qy][qx] = D0 * Qx;
r0[1][qz][qy][qx] = D0 * Qy;
r0[2][qz][qy][qx] = D0 * Qz;
}
if (vector_coeff)
{
const real_t D1 = D(qx, qy, qz, 1, e);
const real_t D2 = D(qx, qy, qz, 2, e);
r0[0][qz][qy][qx] = D0 * Qx;
r0[1][qz][qy][qx] = D1 * Qy;
r0[2][qz][qy][qx] = D2 * Qz;
}
if (matrix_coeff)
{
const real_t D1 = D(qx, qy, qz, 1, e);
const real_t D2 = D(qx, qy, qz, 2, e);
const real_t D3 = D(qx, qy, qz, 3, e);
const real_t D4 = D(qx, qy, qz, 4, e);
const real_t D5 = D(qx, qy, qz, 5, e);
const real_t D6 = D(qx, qy, qz, 6, e);
const real_t D7 = D(qx, qy, qz, 7, e);
const real_t D8 = D(qx, qy, qz, 8, e);
r0[0][qz][qy][qx] = D0 * Qx + D1 * Qy + D2 * Qz;
r0[1][qz][qy][qx] = D3 * Qx + D4 * Qy + D5 * Qz;
r0[2][qz][qy][qx] = D6 * Qx + D7 * Qy + D8 * Qz;
}
}
}
}
kernels::internal::EvalTranspose3d(D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs3d(e, D1D, r1, Y);
});
}
} // namespace internal
template<int DIM, int T_D1D, int T_Q1D>
VectorMassIntegrator::VectorMassAddMultPAType
VectorMassIntegrator::VectorMassAddMultPA::Kernel()
{
if (DIM == 2)
{
return internal::SmemPAVectorMassApply2D<T_D1D,T_Q1D>;
}
else if (DIM == 3)
{
return internal::SmemPAVectorMassApply3D<T_D1D, T_Q1D>;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
inline VectorMassIntegrator::VectorMassAddMultPAType
VectorMassIntegrator::VectorMassAddMultPA::Fallback(int dim, int d1d, int q1d)
{
if (dim == 2)
{
return internal::SmemPAVectorMassApply2D;
}
else if (dim == 3)
{
return internal::SmemPAVectorMassApply3D;
}
else { MFEM_ABORT("Unsupported kernel"); }
}
/// \endcond DO_NOT_DOCUMENT
} // namespace mfem
+3 -710
View File
@@ -14,7 +14,6 @@
#include "../config/config.hpp"
#include "../linalg/dtensor.hpp"
#include "../linalg/tensor.hpp"
namespace mfem
{
@@ -27,713 +26,7 @@ namespace kernels
namespace internal
{
// Types for tensors mapped to registers
// - N is the number of threads in each of the x and y dimensions
// - N should not be greater than 32, to have a maximum of 1024 threads
// On GPU, the last two dimensions are set to 0 to match a 2D tile of threads
#if ((defined(MFEM_USE_CUDA) && defined(__CUDA_ARCH__)) || \
(defined(MFEM_USE_HIP) && defined(__HIP_DEVICE_COMPILE__)))
template <int N = 0>
using s_regs2d_t = mfem::future::tensor<real_t, 0, 0>;
template <int VDIM, int N>
using v_regs2d_t = mfem::future::tensor<real_t, VDIM, 0, 0>;
template <int VDIM, int DIM, int N = 0>
using vd_regs2d_t = mfem::future::tensor<real_t, VDIM, DIM, 0, 0>;
template <int N>
using s_regs3d_t = mfem::future::tensor<real_t, N, 0, 0>;
template <int VDIM, int N>
using v_regs3d_t = mfem::future::tensor<real_t, VDIM, N, 0, 0>;
template <int VDIM, int DIM, int N>
using vd_regs3d_t = mfem::future::tensor<real_t, VDIM, DIM, N, 0, 0>;
// on GPU, SetMaxOf is a no-op, for minimal register usage
constexpr int SetMaxOf(int n) { return n; }
#else
template <int N>
using s_regs2d_t = mfem::future::tensor<real_t, N, N>;
template <int VDIM, int N>
using v_regs2d_t = mfem::future::tensor<real_t, VDIM, N, N>;
template <int VDIM, int DIM, int N>
using vd_regs2d_t = mfem::future::tensor<real_t, VDIM, DIM, N, N>;
template <int N>
using s_regs3d_t = mfem::future::tensor<real_t, N, N, N>;
template <int VDIM, int N>
using v_regs3d_t = mfem::future::tensor<real_t, VDIM, N, N, N>;
template <int VDIM, int DIM, int N>
using vd_regs3d_t = mfem::future::tensor<real_t, VDIM, DIM, N, N, N>;
// on CPU, get next multiple of 4, allowing better alignments
template <int N>
constexpr int NextMultipleOf(int n)
{
static_assert(N > 0 && (N & (N - 1)) == 0, "N must be a power of 2");
return (n + (N - 1)) & ~(N - 1);
}
constexpr int SetMaxOf(int n) { return NextMultipleOf<4>(n); }
#endif // CUDA/HIP && DEVICE_COMPILE
/// Load 2D matrix into shared memory
template <int MQ1>
inline MFEM_HOST_DEVICE void LoadMatrix(const int d1d, const int q1d,
const real_t *M, real_t (*N)[MQ1])
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, q1d)
{
N[dy][qx] = M[dy * q1d + qx];
}
}
MFEM_SYNC_THREAD;
}
/// Load 2D input VDIM*DIM vector into given register tensor, specific component
template <int VDIM, int DIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d, const int c,
const DeviceTensor<4, const real_t> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
{
for (int d = 0; d < DIM; d++)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[c][d][dy][dx] = X(dx, dy, c, e);
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 2D input VDIM*DIM vector into given register tensor
template <int VDIM, int DIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
const DeviceTensor<4, const real_t> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c) { LoadDofs2d(e, d1d, c, X, Y); }
}
/// Load 2D input VDIM vector into given register tensor
template <int VDIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
const DeviceTensor<4, const real_t> &X,
v_regs2d_t<VDIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[c][dy][dx] = X(dx, dy, c, e);
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 2D input scalar into given register tensor
template <int MQ1 = 0>
inline MFEM_HOST_DEVICE void LoadDofs2d(const int e, const int d1d,
const DeviceTensor<3, const real_t> &X,
s_regs2d_t<MQ1> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[dy][dx] = X(dx, dy, e);
}
}
MFEM_SYNC_THREAD;
}
/// Write 2D vector into given device tensor, with read (i) write (j) indices
template <int VDIM, int DIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
const int i, const int j,
vd_regs2d_t<VDIM, DIM, MQ1> &X,
const DeviceTensor<4, real_t> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
real_t y = 0.0;
for (int d = 0; d < DIM; d++) { y += X(i, d, dy, dx); }
Y(dx, dy, j, e) += y;
}
}
MFEM_SYNC_THREAD;
}
/// Write 2D VDIM*DIM vector into given device tensor
template <int VDIM, int DIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
vd_regs2d_t<VDIM, DIM, MQ1> &X,
const DeviceTensor<4, real_t> &Y)
{
for (int c = 0; c < VDIM; ++c) { WriteDofs2d(e, d1d, c, c, X, Y); }
}
/// Write 2D VDIM vector into given device tensor
template <int VDIM, int MQ1 = 0>
inline MFEM_HOST_DEVICE void WriteDofs2d(const int e, const int d1d,
v_regs2d_t<VDIM, MQ1> &X,
const DeviceTensor<4, real_t> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y(dx, dy, c, e) += X(c, dy, dx);
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 3D input VDIM*DIM vector into given register tensor, specific component
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d, const int c,
const DeviceTensor<5, const real_t> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
{
for (int d = 0; d < DIM; d++)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[c][d][dz][dy][dx] = X(dx, dy, dz, c, e);
}
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 3D input VDIM*DIM vector into given register tensor
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
const DeviceTensor<5, const real_t> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c) { LoadDofs3d(e, d1d, c, X, Y); }
}
/// Load 3D input VDIM vector into given register tensor
template <int VDIM, int MQ1>
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
const DeviceTensor<5, const real_t> &X,
v_regs3d_t<VDIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[c][dz][dy][dx] = X(dx,dy,dz,c,e);
}
}
}
}
MFEM_SYNC_THREAD;
}
/// Load 3D input scalar into given register tensor
template <int MQ1>
inline MFEM_HOST_DEVICE void LoadDofs3d(const int e, const int d1d,
const DeviceTensor<4, const real_t> &X,
s_regs3d_t<MQ1> &Y)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y[dz][dy][dx] = X(dx,dy,dz,e);
}
}
}
MFEM_SYNC_THREAD;
}
/// Write 3D scalar into given device tensor, with read (i) write (j) indices
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
const int i, const int j,
vd_regs3d_t<VDIM, DIM, MQ1> &X,
const DeviceTensor<5, real_t> &Y)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
real_t value = 0.0;
for (int d = 0; d < DIM; d++) { value += X(i, d, dz, dy, dx); }
Y(dx, dy, dz, j, e) += value;
}
}
}
MFEM_SYNC_THREAD;
}
/// Write 3D VDIM*DIM vector into given device tensor
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
vd_regs3d_t<VDIM, DIM, MQ1> &X,
const DeviceTensor<5, real_t> &Y)
{
for (int c = 0; c < VDIM; ++c) { WriteDofs3d(e, d1d, c, c, X, Y); }
}
/// Write 3D VDIM vector into given device tensor
template <int VDIM, int MQ1>
inline MFEM_HOST_DEVICE void WriteDofs3d(const int e, const int d1d,
v_regs3d_t<VDIM, MQ1> &X,
const DeviceTensor<5, real_t> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
for (int dz = 0; dz < d1d; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, d1d)
{
Y(dx, dy, dz, c, e) += X(c, dz, dy, dx);
}
}
}
}
MFEM_SYNC_THREAD;
}
/// 2D scalar contraction, X direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractX2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? q1d : d1d))
{
smem[y][x] = X[y][x];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? d1d : q1d))
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[x][k] : B[k][x]) * smem[y][k];
}
Y[y][x] = u;
}
}
MFEM_SYNC_THREAD;
}
/// 2D scalar contraction, Y direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractY2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? q1d : d1d))
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { smem[y][x] = X[y][x]; }
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? d1d : q1d))
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[y][k] : B[k][y]) * smem[k][x];
}
Y[y][x] = u;
}
}
MFEM_SYNC_THREAD;
}
/// 2D scalar copy
template <int MQ1 = 0>
inline MFEM_HOST_DEVICE void Copy2d(const int q1d,
s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { Y[y][x] = X[y][x]; }
}
MFEM_SYNC_THREAD;
}
/// 2D scalar contraction: X & Y directions, with additional copy
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void Contract2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*Bx)[MQ1],
const real_t (*By)[MQ1],
s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
if (!Transpose)
{
ContractX2d<false>(d1d, q1d, smem, Bx, X, Y);
ContractY2d<false>(d1d, q1d, smem, By, Y, X);
Copy2d(q1d, X, Y);
}
else
{
Copy2d(q1d, X, Y);
ContractY2d<true>(d1d, q1d, smem, By, Y, X);
ContractX2d<true>(d1d, q1d, smem, Bx, X, Y);
}
}
/// 2D scalar evaluation
template <int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Eval2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
s_regs2d_t<MQ1> &X,
s_regs2d_t<MQ1> &Y)
{
Contract2d<Transpose, MQ1>(d1d, q1d, smem, B, B, X, Y);
}
/// 2D vector evaluation
template <int VDIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Eval2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
v_regs2d_t<VDIM, MQ1> &X,
v_regs2d_t<VDIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; c++)
{
Eval2d<MQ1, Transpose>(d1d, q1d, smem, B, X[c], Y[c]);
}
}
/// 2D vector transposed evaluation
template <int VDIM, int MQ1>
inline MFEM_HOST_DEVICE void EvalTranspose2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
v_regs2d_t<VDIM, MQ1> &X,
v_regs2d_t<VDIM, MQ1> &Y)
{
Eval2d<VDIM, MQ1, true>(d1d, q1d, smem, B, X, Y);
}
/// 2D vector gradient, with component
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Grad2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs2d_t<VDIM, DIM, MQ1> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y,
const int c)
{
for (int d = 0; d < DIM; d++)
{
const real_t (*Bx)[MQ1] = (d == 0) ? G : B;
const real_t (*By)[MQ1] = (d == 1) ? G : B;
Contract2d<Transpose>(d1d, q1d, smem, Bx, By, X[c][d], Y[c][d]);
}
}
/// 2D vector gradient
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Grad2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs2d_t<VDIM, DIM, MQ1> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; ++c)
{
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
}
}
/// 2D vector transposed gradient
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void GradTranspose2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs2d_t<VDIM, DIM, MQ1> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y)
{
constexpr bool Transpose = true;
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y);
}
/// 2D scalar contraction, with component
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void GradTranspose2d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs2d_t<VDIM, DIM, MQ1> &X,
vd_regs2d_t<VDIM, DIM, MQ1> &Y,
const int c)
{
constexpr bool Transpose = true;
Grad2d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
}
/// 3D scalar contraction, X direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractX3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
for (int z = 0; z < d1d; ++z)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? q1d : d1d))
{
smem[y][x] = X[z][y][x];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(y, y, d1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, (Transpose ? d1d : q1d))
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[x][k] : B[k][x]) * smem[y][k];
}
Y[z][y][x] = u;
}
}
MFEM_SYNC_THREAD;
}
}
/// 3D scalar contraction, Y direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractY3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
for (int z = 0; z < d1d; ++z)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? q1d : d1d))
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d) { smem[y][x] = X[z][y][x]; }
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(y, y, (Transpose ? d1d : q1d))
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[y][k] : B[k][y]) * smem[k][x];
}
Y[z][y][x] = u;
}
}
MFEM_SYNC_THREAD;
}
}
/// 3D scalar contraction, Z direction
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void ContractZ3d(const int d1d, const int q1d,
const real_t (*B)[MQ1],
const s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
for (int z = 0; z < (Transpose ? d1d : q1d); ++z)
{
MFEM_FOREACH_THREAD_DIRECT(y, y, q1d)
{
MFEM_FOREACH_THREAD_DIRECT(x, x, q1d)
{
real_t u = 0.0;
for (int k = 0; k < (Transpose ? q1d : d1d); ++k)
{
u += (Transpose ? B[z][k] : B[k][z]) * X[k][y][x];
}
Y[z][y][x] = u;
}
}
}
}
/// 3D scalar contraction: X, Y & Z directions
template <bool Transpose, int MQ1>
inline MFEM_HOST_DEVICE void Contract3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*Bx)[MQ1],
const real_t (*By)[MQ1],
const real_t (*Bz)[MQ1],
s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
if (!Transpose)
{
ContractX3d<false>(d1d, q1d, smem, Bx, X, Y);
ContractY3d<false>(d1d, q1d, smem, By, Y, X);
ContractZ3d<false>(d1d, q1d, Bz, X, Y);
}
else
{
ContractZ3d<true>(d1d, q1d, Bz, X, Y);
ContractY3d<true>(d1d, q1d, smem, By, Y, X);
ContractX3d<true>(d1d, q1d, smem, Bx, X, Y);
}
}
/// 3D scalar evaluation
template <int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Eval3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
s_regs3d_t<MQ1> &X,
s_regs3d_t<MQ1> &Y)
{
Contract3d<Transpose>(d1d, q1d, smem, B, B, B, X, Y);
}
/// 3D vector evaluation
template <int VDIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Eval3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
v_regs3d_t<VDIM, MQ1> &X,
v_regs3d_t<VDIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; c++)
{
Eval3d<MQ1, Transpose>(d1d, q1d, smem, B, X[c], Y[c]);
}
}
/// 3D vector transposed evaluation
template <int VDIM, int MQ1>
inline MFEM_HOST_DEVICE void EvalTranspose3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
v_regs3d_t<VDIM, MQ1> &X,
v_regs3d_t<VDIM, MQ1> &Y)
{
Eval3d<VDIM, MQ1, true>(d1d, q1d, smem, B, X, Y);
}
/// 3D vector gradient, with component
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Grad3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs3d_t<VDIM, DIM, MQ1> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y,
const int c)
{
for (int d = 0; d < DIM; d++)
{
const real_t (*Bx)[MQ1] = (d == 0) ? G : B;
const real_t (*By)[MQ1] = (d == 1) ? G : B;
const real_t (*Bz)[MQ1] = (d == 2) ? G : B;
Contract3d<Transpose>(d1d, q1d, smem, Bx, By, Bz, X[c][d], Y[c][d]);
}
}
/// 3D vector gradient
template <int VDIM, int DIM, int MQ1, bool Transpose = false>
inline MFEM_HOST_DEVICE void Grad3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs3d_t<VDIM, DIM, MQ1> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
{
for (int c = 0; c < VDIM; c++)
{
Grad3d<VDIM, DIM, MQ1, Transpose>(d1d, q1d, smem, B, G, X, Y, c);
}
}
/// 3D vector transposed gradient
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void GradTranspose3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs3d_t<VDIM, DIM, MQ1> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y)
{
Grad3d<VDIM, DIM, MQ1, true>(d1d, q1d, smem, B, G, X, Y);
}
/// 3D vector transposed gradient, with component
template <int VDIM, int DIM, int MQ1>
inline MFEM_HOST_DEVICE void GradTranspose3d(const int d1d, const int q1d,
real_t (&smem)[MQ1][MQ1],
const real_t (*B)[MQ1],
const real_t (*G)[MQ1],
vd_regs3d_t<VDIM, DIM, MQ1> &X,
vd_regs3d_t<VDIM, DIM, MQ1> &Y,
const int c)
{
Grad3d<VDIM, DIM, MQ1, true>(d1d, q1d, smem, B, G, X, Y, c);
}
/// Load B1d matrix into shared memory
/// Load B1d matrice into shared memory
template<int MD1, int MQ1>
MFEM_HOST_DEVICE inline void LoadB(const int D1D, const int Q1D,
const ConstDeviceMatrix &b,
@@ -755,7 +48,7 @@ MFEM_HOST_DEVICE inline void LoadB(const int D1D, const int Q1D,
MFEM_SYNC_THREAD;
}
/// Load Bt1d matrix into shared memory
/// Load Bt1d matrices into shared memory
template<int MD1, int MQ1>
MFEM_HOST_DEVICE inline void LoadBt(const int D1D, const int Q1D,
const ConstDeviceMatrix &b,
@@ -2258,7 +1551,7 @@ MFEM_HOST_DEVICE inline void GradXt(const int D1D, const int Q1D,
}
}
} // namespace internal
} // namespace kernels::internal
} // namespace kernels
+4 -4
View File
@@ -5220,10 +5220,10 @@ void ConformingProlongationOperator::Mult(const Vector &x, Vector &y) const
for (int i = 0; i < m; i++)
{
const int end = external_ldofs[i];
if (end > j) { std::copy(xdata+j-i, xdata+end-i, ydata+j); }
std::copy(xdata+j-i, xdata+end-i, ydata+j);
j = end+1;
}
if (Width() > (j-m)) { std::copy(xdata+j-m, xdata+Width(), ydata+j); }
std::copy(xdata+j-m, xdata+Width(), ydata+j);
const int out_layout = 0; // 0 - output is ldofs array
if (!local)
@@ -5251,10 +5251,10 @@ void ConformingProlongationOperator::MultTranspose(
for (int i = 0; i < m; i++)
{
const int end = external_ldofs[i];
if (end > j) { std::copy(xdata+j, xdata+end, ydata+j-i); }
std::copy(xdata+j, xdata+end, ydata+j-i);
j = end+1;
}
if (Height() > j) { std::copy(xdata+j, xdata+Height(), ydata+j-m); }
std::copy(xdata+j, xdata+Height(), ydata+j-m);
const int out_layout = 2; // 2 - output is an array on all ltdofs
if (!local)
-10
View File
@@ -41,12 +41,6 @@ public:
qspace(&qspace_), own_qspace(false), vdim(vdim_)
{ UseDevice(true); }
/// Same as above but specify the device memory type
QuadratureFunction(QuadratureSpaceBase &qspace_, MemoryType mt, int vdim_ = 1)
: Vector(vdim_*qspace_.GetSize(), mt),
qspace(&qspace_), own_qspace(false), vdim(vdim_)
{ UseDevice(true); }
/// Create a QuadratureFunction based on the given QuadratureSpaceBase.
/** The QuadratureFunction does not assume ownership of the
QuadratureSpaceBase.
@@ -54,10 +48,6 @@ public:
QuadratureFunction(QuadratureSpaceBase *qspace_, int vdim_ = 1)
: QuadratureFunction(*qspace_, vdim_) { }
/// Same as above but specify the device memory type
QuadratureFunction(QuadratureSpaceBase *qspace_, MemoryType mt, int vdim_ = 1)
: QuadratureFunction(*qspace_, mt, vdim_) { }
/** @brief Create a QuadratureFunction based on the given QuadratureSpaceBase,
using the external (host) data, @a qf_data. */
/** The QuadratureFunction does not assume ownership of the
-9
View File
@@ -5368,15 +5368,6 @@ void TMOP_Integrator::ParEnableNormalization(const ParGridFunction &x)
}
#endif
void TMOP_Integrator::GetNormalizationFactors(real_t &m_normal,
real_t &l_normal,
real_t &s_normal)
{
m_normal = this->metric_normal;
l_normal = this->lim_normal;
s_normal = this->surf_fit_normal;
}
void TMOP_Integrator::ComputeNormalizationEnergies(const GridFunction &x,
real_t &metric_energy,
real_t &lim_energy)
-38
View File
@@ -452,8 +452,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 9; }
};
/// 2D non-barrier Shape+Size+Orientation (VOS) metric (polyconvex).
@@ -504,8 +502,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 22; }
};
/// 2D barrier shape metric (polyconvex).
@@ -526,8 +522,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 50; }
};
/// 2D non-barrier size (V) metric (not polyconvex).
@@ -599,8 +593,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 58; }
};
/// 2D non-barrier Shape+Size (VS) metric.
@@ -683,8 +675,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 85; }
};
/// 2D compound barrier Shape+Size (VS) metric (balanced).
@@ -742,8 +732,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 98; }
};
/// 2D untangling metric.
@@ -763,8 +751,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 211; }
};
/// Shifted barrier form of metric 56 (area, ideal barrier metric), 2D
@@ -785,8 +771,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 252; }
};
/// 3D barrier Shape (S) metric, well-posed (polyconvex & invex).
@@ -806,8 +790,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 301; }
};
/// 3D barrier Shape (S) metric, well-posed (polyconvex & invex).
@@ -890,8 +872,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 311; }
};
/// 3D Shape (S) metric, untangling version of 303.
@@ -950,8 +930,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 316; }
};
/// 3D Size (V) metric.
@@ -1096,7 +1074,6 @@ public:
AddQualityMetric(sz_metric, gamma);
}
int Id() const override { return 333; }
virtual ~TMOP_Metric_333() { delete sh_metric; delete sz_metric; }
};
@@ -1159,8 +1136,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 342; }
};
/// 3D barrier Shape+Size (VS) metric, well-posed (polyconvex).
@@ -1202,8 +1177,6 @@ public:
void AssembleH(const DenseMatrix &Jpt, const DenseMatrix &DS,
const real_t weight, DenseMatrix &A) const override;
int Id() const override { return 352; }
};
/// 3D non-barrier Shape (S) metric.
@@ -1990,12 +1963,6 @@ class TMOP_Integrator : public NonlinearFormIntegrator
protected:
friend class TMOPNewtonSolver;
friend class TMOPComboIntegrator;
friend class TMOPEnergyPA2D;
friend class TMOPEnergyPA3D;
friend class TMOPAssembleGradPA2D;
friend class TMOPAssembleGradPA3D;
friend class TMOPAddMultPA2D;
friend class TMOPAddMultPA3D;
// Initial positions of the mesh nodes. Not owned. The pointer is set at the
// start of the solve by TMOPNewtonSolver::Mult(), and unset at the end.
@@ -2516,11 +2483,6 @@ public:
void ParEnableNormalization(const ParGridFunction &x);
#endif
/** @brief Get the normalization factors of the metric */
void GetNormalizationFactors(real_t &metric_normal,
real_t &lim_normal,
real_t &surf_fit_normal);
/** @brief Enables FD-based approximation and computes dx. */
void EnableFiniteDifferences(const GridFunction &x);
#ifdef MFEM_USE_MPI
-180
View File
@@ -1,180 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
/* // Original i-j assembly (old invariants code).
for (int e = 0; e < NE; e++)
{
for (int q = 0; q < nqp; q++)
{
el.CalcDShape(ip, DSh);
Mult(DSh, Jrt, DS);
for (int i = 0; i < dof; i++)
{
for (int j = 0; j < dof; j++)
{
for (int r = 0; r < dim; r++)
{
for (int c = 0; c < dim; c++)
{
for (int rr = 0; rr < dim; rr++)
{
for (int cc = 0; cc < dim; cc++)
{
const real_t H = h(r, c, rr, cc);
A(e, i + r*dof, j + rr*dof) +=
weight_q * DS(i, c) * DS(j, cc) * H;
}
}
}
}
}
}
}
}*/
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleDiagPA_2D(const int NE,
const ConstDeviceMatrix &B,
const ConstDeviceMatrix &G,
const DeviceTensor<5, const real_t> &J,
const DeviceTensor<7, const real_t> &H,
DeviceTensor<4> &D,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
// Takes into account Jtr by replacing H with Href at all quad points.
MFEM_SHARED real_t Href_data[2 * 2 * 2 * MQ1 * MQ1];
DeviceTensor<5, real_t> Href(Href_data, 2, 2, 2, MQ1, MQ1);
for (int v = 0; v < 2; v++)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
real_t Jrt_data[4];
ConstDeviceMatrix Jrt(Jrt_data, 2, 2);
kernels::CalcInverse<2>(Jtr, Jrt_data);
for (int m = 0; m < 2; m++)
{
for (int n = 0; n < 2; n++)
{
// Hr_{v,m,n,q} = \sum_{s,t=1}^d
// Jrt_{m,s,q} H_{v,s,v,t,q} Jrt_{n,t,q}
Href(v, m, n, qx, qy) = 0.0;
for (int s = 0; s < 2; s++)
{
for (int t = 0; t < 2; t++)
{
Href(v, m, n, qx, qy) +=
Jrt(m, s) * H(v, s, v, t, qx, qy, e) * Jrt(n, t);
}
}
}
}
}
}
}
MFEM_SHARED real_t qd[2 * 2 * MQ1 * MD1];
DeviceTensor<4, real_t> QD(qd, 2, 2, MQ1, MD1);
for (int v = 0; v < 2; v++)
{
// Contract in y.
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
for (int m = 0; m < 2; m++)
{
for (int n = 0; n < 2; n++) { QD(m, n, qx, dy) = 0.0; }
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = B(qy, dy);
const real_t Gy = G(qy, dy);
for (int m = 0; m < 2; m++)
{
for (int n = 0; n < 2; n++)
{
const real_t L = (m == 1 ? Gy : By);
const real_t R = (n == 1 ? Gy : By);
QD(m, n, qx, dy) += L * Href(v, m, n, qx, qy) * R;
}
}
}
}
}
MFEM_SYNC_THREAD;
// Contract in x.
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t d = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = B(qx, dx);
const real_t Gx = G(qx, dx);
for (int m = 0; m < 2; m++)
{
for (int n = 0; n < 2; n++)
{
const real_t L = (m == 0 ? Gx : Bx);
const real_t R = (n == 0 ? Gx : Bx);
d += L * QD(m, n, qx, dy) * R;
}
}
}
D(dx, dy, v, e) += d;
}
}
MFEM_SYNC_THREAD;
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiag2D, TMOP_AssembleDiagPA_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiag2D);
void TMOP_Integrator::AssembleDiagonalPA_2D(Vector &diagonal) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(PA.maps->B.Read(), q, d);
const auto G = Reshape(PA.maps->G.Read(), q, d);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto H = Reshape(PA.H.Read(), 2, 2, 2, 2, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, 2, NE);
TMOPAssembleDiag2D::Run(d, q, NE, B, G, J, H, D, d, q);
}
} // namespace mfem
-83
View File
@@ -1,83 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../../general/forall.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleDiagPA_C0_2D(const int NE,
const ConstDeviceMatrix &B,
const DeviceTensor<5, const real_t> &H0,
DeviceTensor<4> &D,
const int d1d, const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t qd[MQ1 * MD1];
DeviceTensor<2, real_t> QD(qd, MQ1, MD1);
for (int v = 0; v < 2; v++)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
QD(qx, dy) = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t bb = B(qy, dy) * B(qy, dy);
QD(qx, dy) += bb * H0(v, v, qx, qy, e);
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t d = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t bb = B(qx, dx) * B(qx, dx);
d += bb * QD(qx, dy);
}
D(dx, dy, v, e) += d;
}
}
MFEM_SYNC_THREAD;
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiagCoef2D, TMOP_AssembleDiagPA_C0_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagCoef2D);
void TMOP_Integrator::AssembleDiagonalPA_C0_2D(Vector &diagonal) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(PA.maps->B.Read(), q, d);
const auto H0 = Reshape(PA.H0.Read(), 2, 2, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, 2, NE);
TMOPAssembleDiagCoef2D::Run(d, q, NE, B, H0, D, d, q);
}
} // namespace mfem
-229
View File
@@ -1,229 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleDiagPA_3D(const int NE,
const ConstDeviceMatrix &B,
const ConstDeviceMatrix &G,
const DeviceTensor<6, const real_t> &J,
const DeviceTensor<8, const real_t> &H,
DeviceTensor<5> &D,
const int d1d = 0,
const int q1d = 0)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[3][3][MQ1][MQ1];
kernels::internal::vd_regs3d_t<3, 3, MQ1> rH, r0, r1;
for (int v = 0; v < 3; ++v)
{
// Takes into account Jtr by replacing H with Href at all quad points.
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
real_t Jrt_data[9];
ConstDeviceMatrix Jrt(Jrt_data, 3, 3);
kernels::CalcInverse<3>(Jtr, Jrt_data);
real_t h[3][3];
for (int s = 0; s < 3; s++)
{
for (int t = 0; t < 3; t++)
{
h[s][t] = H(v, s, v, t, qx, qy, qz, e);
}
}
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
// Hr_{v,m,n,q} = \sum_{s,t=1}^d
// Jrt_{m,s,q} H_{v,s,v,t,q} Jrt_{n,t,q}
rH(m, n, qz, qy, qx) = 0.0;
for (int s = 0; s < 3; s++)
{
for (int t = 0; t < 3; t++)
{
rH(m, n, qz, qy, qx) += Jrt(m, s) * h[s][t] * Jrt(n, t);
}
}
}
}
}
}
MFEM_SYNC_THREAD;
}
// Contract in z.
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
r0(m, n, dz, qy, qx) = 0.0;
}
}
for (int qz = 0; qz < Q1D; ++qz)
{
const real_t Bz = B(qz, dz), Gz = G(qz, dz);
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
const real_t L = (m == 2 ? Gz : Bz);
const real_t R = (n == 2 ? Gz : Bz);
r0(m, n, dz, qy, qx) += L * rH(m, n, qz, qy, qx) * R;
}
}
}
}
}
MFEM_SYNC_THREAD;
}
// Contract in y.
for (int dz = 0; dz < D1D; ++dz)
{
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
smem[m][n][qy][qx] = r0(m, n, dz, qy, qx);
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
r1(m, n, dz, dy, qx) = 0.0;
}
}
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = B(qy, dy);
const real_t Gy = G(qy, dy);
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
const real_t L = (m == 1 ? Gy : By);
const real_t R = (n == 1 ? Gy : By);
r1(m, n, dz, dy, qx) += L * smem[m][n][qy][qx] * R;
}
}
}
}
}
MFEM_SYNC_THREAD;
}
// Contract in x.
for (int dz = 0; dz < D1D; ++dz)
{
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
smem[m][n][dy][qx] = r1(m, n, dz, dy, qx);
}
}
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t d = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = B(qx, dx);
const real_t Gx = G(qx, dx);
for (int m = 0; m < 3; m++)
{
for (int n = 0; n < 3; n++)
{
const real_t L = (m == 0 ? Gx : Bx);
const real_t R = (n == 0 ? Gx : Bx);
d += L * smem[m][n][dy][qx] * R;
}
}
}
D(dx, dy, dz, v, e) += d;
}
}
MFEM_SYNC_THREAD;
}
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiag3D, TMOP_AssembleDiagPA_3D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiag3D);
void TMOP_Integrator::AssembleDiagonalPA_3D(Vector &diagonal) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(PA.maps->B.Read(), q, d);
const auto G = Reshape(PA.maps->G.Read(), q, d);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto H = Reshape(PA.H.Read(), 3, 3, 3, 3, q, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, d, 3, NE);
TMOPAssembleDiag3D::Run(d, q, NE, B, G, J, H, D, d, q);
}
} // namespace mfem
-131
View File
@@ -1,131 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleDiagPA_C0_3D(const int NE,
const ConstDeviceMatrix &B,
const DeviceTensor<6, const real_t> &H0,
DeviceTensor<5> &D,
const int d1d, const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::s_regs3d_t<MQ1> r0, r1;
for (int v = 0; v < 3; ++v)
{
// first tensor contraction, along z direction
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
real_t u = 0.0;
for (int qz = 0; qz < Q1D; ++qz)
{
const real_t Bz = B(qz, dz);
u += Bz * H0(v, v, qx, qy, qz, e) * Bz;
}
r0[dz][qy][qx] = u;
}
}
MFEM_SYNC_THREAD;
}
// second tensor contraction, along y direction
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
smem[qy][qx] = r0[dz][qy][qx];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
real_t u = 0.0;
for (int qy = 0; qy < Q1D; ++qy)
{
const real_t By = B(qy, dy);
u += By * smem[qy][qx] * By;
}
r1[dz][dy][qx] = u;
}
}
MFEM_SYNC_THREAD;
}
// third tensor contraction, along x direction
for (int dz = 0; dz < D1D; ++dz)
{
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
smem[dy][qx] = r1[dz][dy][qx];
}
}
MFEM_SYNC_THREAD;
MFEM_FOREACH_THREAD_DIRECT(dy, y, D1D)
{
MFEM_FOREACH_THREAD_DIRECT(dx, x, D1D)
{
real_t u = 0.0;
for (int qx = 0; qx < Q1D; ++qx)
{
const real_t Bx = B(qx, dx);
u += Bx * smem[dy][qx] * Bx;
}
D(dx, dy, dz, v, e) += u;
}
}
MFEM_SYNC_THREAD;
}
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleDiagCoef3D, TMOP_AssembleDiagPA_C0_3D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleDiagCoef3D);
void TMOP_Integrator::AssembleDiagonalPA_C0_3D(Vector &diagonal) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto B = Reshape(PA.maps->B.Read(), q, d);
const auto H0 = Reshape(PA.H0.Read(), 3, 3, q, q, q, NE);
auto D = Reshape(diagonal.ReadWrite(), d, d, d, 3, NE);
TMOPAssembleDiagCoef3D::Run(d, q, NE, B, H0, D, d, q);
}
} // namespace mfem
-35
View File
@@ -1,35 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "grad2.hpp"
namespace mfem
{
void TMOP_Integrator::AssembleGradPA_2D(const Vector &x) const
{
const int mid = metric->Id();
// Calls TMOPAssembleGradPA2D::Mult for the given mid.
TMOPAssembleGradPA2D ker(this, x);
if (mid == 1) { return tmop::Kernel<1>(ker); }
if (mid == 2) { return tmop::Kernel<2>(ker); }
if (mid == 7) { return tmop::Kernel<7>(ker); }
if (mid == 56) { return tmop::Kernel<56>(ker); }
if (mid == 77) { return tmop::Kernel<77>(ker); }
if (mid == 80) { return tmop::Kernel<80>(ker); }
if (mid == 94) { return tmop::Kernel<94>(ker); }
MFEM_ABORT("Unsupported TMOP metric " << mid);
}
} // namespace mfem
-106
View File
@@ -1,106 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
class TMOPAssembleGradPA2D
{
const mfem::TMOP_Integrator *ti; // not owned
const Vector &x;
public:
TMOPAssembleGradPA2D(const TMOP_Integrator *ti, const Vector &x): ti(ti),
x(x) {}
int Ndof() const { return ti->PA.maps->ndof; }
int Nqpt() const { return ti->PA.maps->nqpt; }
template <int MD1, int MQ1, typename METRIC, int T_D1D = 0, int T_Q1D = 0>
static void Mult(TMOPAssembleGradPA2D &ker)
{
const mfem::TMOP_Integrator *ti = ker.ti;
const real_t metric_normal = ti->metric_normal;
const int NE = ti->PA.ne, d1d = ker.Ndof(), q1d = ti->PA.maps->nqpt;
const int D1D = T_D1D ? T_D1D : d1d, Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
Array<real_t> mp;
if (auto m = dynamic_cast<TMOP_Combo_QualityMetric *>(ti->metric))
{
m->GetWeights(mp);
}
const real_t *w = mp.Read();
const auto *b = ti->PA.maps->B.Read(), *g = ti->PA.maps->G.Read();
const auto X = Reshape(ker.x.Read(), D1D, D1D, 2, NE);
const auto W = Reshape(ti->PA.ir->GetWeights().Read(), Q1D, Q1D);
const auto J = Reshape(ti->PA.Jtr.Read(), 2, 2, Q1D, Q1D, NE);
auto H = Reshape(ti->PA.H.Write(), 2, 2, 2, 2, Q1D, Q1D, NE);
const Vector &mc = ti->PA.MC;
const bool const_m0 = mc.Size() == 1;
const auto MC = const_m0
? Reshape(mc.Read(), 1, 1, 1)
: Reshape(mc.Read(), Q1D, Q1D, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs2d_t<2, 2, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs2d(e, D1D, X, r0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
const real_t detJtr = kernels::Det<2>(Jtr);
const real_t m_coef = const_m0 ? MC(0, 0, 0) : MC(qx, qy, e);
const real_t weight = metric_normal * m_coef * W(qx, qy) * detJtr;
// Jrt = Jtr^{-1}
real_t Jrt[4];
kernels::CalcInverse<2>(Jtr, Jrt);
// Jpr = X^t.DSh
const real_t Jpr[4] =
{
r1[0][0][qy][qx], r1[1][0][qy][qx],
r1[0][1][qy][qx], r1[1][1][qy][qx]
};
// Jpt = Jpr.Jrt
real_t Jpt[4];
kernels::Mult(2, 2, 2, Jpr, Jrt, Jpt);
METRIC{}.AssembleH(qx, qy, e, weight, Jpt, w, H);
}
}
});
}
};
} // namespace mfem
-145
View File
@@ -1,145 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleGradPA_C0_2D(const real_t lim_normal,
const ConstDeviceCube &LD,
const bool const_c0,
const DeviceTensor<3, const real_t> &C0,
const int NE,
const DeviceTensor<5, const real_t> &J,
const ConstDeviceMatrix &W,
const real_t *b,
const real_t *bld,
const DeviceTensor<4, const real_t> &X0,
const DeviceTensor<4, const real_t> &X1,
DeviceTensor<5> &H0,
const bool exp_lim,
const int d1d, const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t sB[MD1][MQ1];
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::LoadMatrix(D1D, Q1D, bld, sB);
kernels::internal::s_regs2d_t<MQ1> rm0, rm1; // scalar LD
kernels::internal::LoadDofs2d(e, D1D, LD, rm0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, rm0, rm1);
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::v_regs2d_t<2,MQ1> r00, r01; // vector X0
kernels::internal::LoadDofs2d(e, D1D, X0, r00);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r00, r01);
kernels::internal::v_regs2d_t<2,MQ1> r10, r11; // vector X1
kernels::internal::LoadDofs2d(e, D1D, X1, r10);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r10, r11);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
const real_t detJtr = kernels::Det<2>(Jtr);
const real_t weight = W(qx, qy) * detJtr;
const real_t coeff0 = const_c0 ? C0(0, 0, 0) : C0(qx, qy, e);
const real_t weight_m = weight * lim_normal * coeff0;
const real_t D = rm1(qy, qx);
const real_t p0[2] = { r01(0, qy, qx), r01(1, qy, qx) };
const real_t p1[2] = { r11(0, qy, qx), r11(1, qy, qx) };
const real_t dist = D; // GetValues, default comp set to 0
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
real_t grad_grad[4];
if (!exp_lim)
{
// d2.Diag(1.0 / (dist * dist), x.Size());
const real_t c = 1.0 / (dist * dist);
kernels::Diag<2>(c, grad_grad);
}
else
{
real_t tmp[2];
kernels::Subtract<2>(1.0, p1, p0, tmp);
real_t dsq = kernels::DistanceSquared<2>(p1, p0);
real_t dist_squared = dist * dist;
real_t dist_squared_squared = dist_squared * dist_squared;
real_t f = exp(10.0 * ((dsq / dist_squared) - 1.0));
grad_grad[0] =
((400.0 * tmp[0] * tmp[0] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
grad_grad[1] =
(400.0 * tmp[0] * tmp[1] * f) / dist_squared_squared;
grad_grad[2] = grad_grad[1];
grad_grad[3] =
((400.0 * tmp[1] * tmp[1] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
}
ConstDeviceMatrix gg(grad_grad, 2, 2);
for (int i = 0; i < 2; i++)
{
for (int j = 0; j < 2; j++)
{
H0(i, j, qx, qy, e) = weight_m * gg(i, j);
}
}
}
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleGradCoef2D, TMOP_AssembleGradPA_C0_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleGradCoef2D);
void TMOP_Integrator::AssembleGradPA_C0_2D(const Vector &x) const
{
const int NE = PA.ne, d = PA.maps_lim->ndof, q = PA.maps_lim->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const real_t ln = lim_normal;
const bool const_c0 = PA.C0.Size() == 1;
const auto C0 = PA.C0.Size() == 1
? Reshape(PA.C0.Read(), 1, 1, 1)
: Reshape(PA.C0.Read(), q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
const auto *b = PA.maps->B.Read(), *bld = PA.maps_lim->B.Read();
const auto LD = Reshape(PA.LD.Read(), d, d, NE);
const auto XL = Reshape(PA.XL.Read(), d, d, 2, NE);
const auto X = Reshape(x.Read(), d, d, 2, NE);
auto H0 = Reshape(PA.H0.Write(), 2, 2, q, q, NE);
const auto el = dynamic_cast<TMOP_ExponentialLimiter *>(lim_func);
const bool exp_lim = el ? true : false;
TMOPAssembleGradCoef2D::Run(d, q, ln, LD, const_c0, C0, NE,
J, W, b, bld, XL, X, H0, exp_lim, d, q);
}
} // namespace mfem
-35
View File
@@ -1,35 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "grad3.hpp"
namespace mfem
{
void TMOP_Integrator::AssembleGradPA_3D(const Vector &x) const
{
const int mid = metric->Id();
// Calls TMOPAssembleGradPA3D::Mult for the given mid.
TMOPAssembleGradPA3D ker(this, x);
if (mid == 302) { return tmop::Kernel<302>(ker); }
if (mid == 303) { return tmop::Kernel<303>(ker); }
if (mid == 315) { return tmop::Kernel<315>(ker); }
if (mid == 318) { return tmop::Kernel<318>(ker); }
if (mid == 321) { return tmop::Kernel<321>(ker); }
if (mid == 332) { return tmop::Kernel<332>(ker); }
if (mid == 338) { return tmop::Kernel<338>(ker); }
MFEM_ABORT("Unsupported TMOP metric " << mid);
}
} // namespace mfem
-111
View File
@@ -1,111 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
class TMOPAssembleGradPA3D
{
const TMOP_Integrator *ti; // not owned
const Vector &x;
public:
TMOPAssembleGradPA3D(const TMOP_Integrator *ti, const Vector &x): ti(ti),
x(x) {}
int Ndof() const { return ti->PA.maps->ndof; }
int Nqpt() const { return ti->PA.maps->nqpt; }
template <int MD1, int MQ1, typename METRIC, int T_D1D = 0, int T_Q1D = 0>
static void Mult(TMOPAssembleGradPA3D &ker)
{
const TMOP_Integrator *ti = ker.ti;
const real_t metric_normal = ti->metric_normal;
const int NE = ti->PA.ne, d1d = ker.Ndof(), q1d = ti->PA.maps->nqpt;
const int D1D = T_D1D ? T_D1D : d1d, Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
Array<real_t> mp;
if (auto m = dynamic_cast<TMOP_Combo_QualityMetric *>(ti->metric))
{
m->GetWeights(mp);
}
const real_t *w = mp.Read();
const auto *b = ti->PA.maps->B.Read(), *g = ti->PA.maps->G.Read();
const auto X = Reshape(ker.x.Read(), D1D, D1D, D1D, 3, NE);
const auto W = Reshape(ti->PA.ir->GetWeights().Read(), Q1D, Q1D, Q1D);
const auto J = Reshape(ti->PA.Jtr.Read(), 3, 3, Q1D, Q1D, Q1D, NE);
auto H = Reshape(ti->PA.H.Write(), 3, 3, 3, 3, Q1D, Q1D, Q1D, NE);
const Vector &mc = ti->PA.MC;
const bool const_m0 = mc.Size() == 1;
const auto MC = const_m0
? Reshape(mc.Read(), 1, 1, 1, 1)
: Reshape(mc.Read(), Q1D, Q1D, Q1D, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs3d_t<3, 3, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs3d(e, D1D, X, r0);
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1);
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
const real_t detJtr = kernels::Det<3>(Jtr);
const real_t m_coef = const_m0 ?
MC(0, 0, 0, 0) :
MC(qx, qy, qz, e);
const real_t weight = metric_normal * m_coef * W(qx, qy, qz) * detJtr;
// Jrt = Jtr^{-1}
real_t Jrt[9];
kernels::CalcInverse<3>(Jtr, Jrt);
// Jpr = X^T.DSh
real_t Jpr[9] =
{
r1(0, 0, qz, qy, qx), r1(1, 0, qz, qy, qx), r1(2, 0, qz, qy, qx),
r1(0, 1, qz, qy, qx), r1(1, 1, qz, qy, qx), r1(2, 1, qz, qy, qx),
r1(0, 2, qz, qy, qx), r1(1, 2, qz, qy, qx), r1(2, 2, qz, qy, qx)
};
// Jpt = X^T . DS = (X^T.DSh) . Jrt = Jpr . Jrt
real_t Jpt[9];
kernels::Mult(3, 3, 3, Jpr, Jrt, Jpt);
METRIC{}.AssembleH(qx, qy, qz, e, weight, Jrt, Jpr, Jpt, w, H);
}
}
}
});
}
};
} // namespace mfem
-167
View File
@@ -1,167 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AssembleGradPA_C0_3D(const real_t lim_normal,
const DeviceTensor<4, const real_t> &LD,
const bool const_c0,
const DeviceTensor<4, const real_t> &C0,
const int NE,
const DeviceTensor<6, const real_t> &J,
const ConstDeviceCube &W,
const real_t *b,
const real_t *bld,
const DeviceTensor<5, const real_t> &X0,
const DeviceTensor<5, const real_t> &X1,
DeviceTensor<6> &H0,
const bool exp_lim,
const int d1d,
const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t sB[MD1][MQ1];
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::LoadMatrix(D1D, Q1D, bld, sB);
kernels::internal::s_regs3d_t<MQ1> rm0, rm1; // scalar LD
kernels::internal::LoadDofs3d(e, D1D, LD, rm0);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, rm0, rm1);
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::v_regs3d_t<3,MQ1> r00, r01; // vector X0
kernels::internal::LoadDofs3d(e, D1D, X0, r00);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r00, r01);
kernels::internal::v_regs3d_t<3,MQ1> r10, r11; // vector X1
kernels::internal::LoadDofs3d(e, D1D, X1, r10);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r10, r11);
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
const real_t detJtr = kernels::Det<3>(Jtr);
const real_t weight = W(qx, qy, qz) * detJtr;
const real_t coeff0 = const_c0
? C0(0, 0, 0, 0)
: C0(qx, qy, qz, e);
const real_t weight_m = weight * lim_normal * coeff0;
const real_t D = rm1(qz, qy, qx);
const real_t p0[3] = { r01(0, qz, qy, qx),
r01(1, qz, qy, qx),
r01(2, qz, qy, qx)
};
const real_t p1[3] = { r11(0, qz, qy, qx),
r11(1, qz, qy, qx),
r11(2, qz, qy, qx)
};
const real_t dist = D; // GetValues, default comp set to 0
// lim_func->Eval_d2(p1, p0, d_vals(q), grad_grad);
real_t grad_grad[9];
if (!exp_lim)
{
// d2.Diag(1.0 / (dist * dist), x.Size());
const real_t c = 1.0 / (dist * dist);
kernels::Diag<3>(c, grad_grad);
}
else
{
real_t tmp[3];
kernels::Subtract<3>(1.0, p1, p0, tmp);
real_t dsq = kernels::DistanceSquared<3>(p1, p0);
real_t dist_squared = dist * dist;
real_t dist_squared_squared = dist_squared * dist_squared;
real_t f = exp(10.0 * ((dsq / dist_squared) - 1.0));
grad_grad[0] =
((400.0 * tmp[0] * tmp[0] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
grad_grad[1] =
(400.0 * tmp[0] * tmp[1] * f) / dist_squared_squared;
grad_grad[2] =
(400.0 * tmp[0] * tmp[2] * f) / dist_squared_squared;
grad_grad[3] = grad_grad[1];
grad_grad[4] =
((400.0 * tmp[1] * tmp[1] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
grad_grad[5] =
(400.0 * tmp[1] * tmp[2] * f) / dist_squared_squared;
grad_grad[6] = grad_grad[2];
grad_grad[7] = grad_grad[5];
grad_grad[8] =
((400.0 * tmp[2] * tmp[2] * f) / dist_squared_squared) +
(20.0 * f / dist_squared);
}
ConstDeviceMatrix gg(grad_grad, 3, 3);
for (int i = 0; i < 3; i++)
{
for (int j = 0; j < 3; j++)
{
H0(i, j, qx, qy, qz, e) = weight_m * gg(i, j);
}
}
}
}
}
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPAssembleGradCoef3D, TMOP_AssembleGradPA_C0_3D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPAssembleGradCoef3D);
void TMOP_Integrator::AssembleGradPA_C0_3D(const Vector &x) const
{
const real_t ln = lim_normal;
const bool const_c0 = PA.C0.Size() == 1;
const int NE = PA.ne, d = PA.maps_lim->ndof, q = PA.maps_lim->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto C0 = const_c0
? Reshape(PA.C0.Read(), 1, 1, 1, 1)
: Reshape(PA.C0.Read(), q, q, q, NE);
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q, q);
const auto *b = PA.maps->B.Read(), *bld = PA.maps_lim->B.Read();
const auto LD = Reshape(PA.LD.Read(), d, d, d, NE);
const auto XL = Reshape(PA.XL.Read(), d, d, d, 3, NE);
const auto X = Reshape(x.Read(), d, d, d, 3, NE);
auto H0 = Reshape(PA.H0.Write(), 3, 3, q, q, q, NE);
auto el = dynamic_cast<TMOP_ExponentialLimiter *>(lim_func);
const bool exp_lim = (el) ? true : false;
TMOPAssembleGradCoef3D::Run(d, q, ln, LD, const_c0, C0, NE,
J, W, b, bld, XL, X, H0, exp_lim, d, q);
}
} // namespace mfem
-76
View File
@@ -1,76 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_001 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return ie.Get_I1();
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
{
MFEM_CONTRACT_VAR(w);
real_t dI1[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI1(dI1));
kernels::Set(2, 2, 1.0, ie.Get_dI1(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
{
MFEM_CONTRACT_VAR(w);
// weight * ddI1
real_t ddI1[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).ddI1(ddI1));
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1(ie.Get_ddI1(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t h = ddi1(r, c);
H(r, c, i, j, qx, qy, e) = weight * h;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_001;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 1);
} // namespace mfem
-74
View File
@@ -1,74 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_002 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return 0.5 * ie.Get_I1b() - 1.0;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
{
MFEM_CONTRACT_VAR(w);
real_t dI1b[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI1b(dI1b).dI2b(dI2b));
kernels::Set(2, 2, 1. / 2., ie.Get_dI1b(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx, const int qy, const int e, const real_t weight,
const real_t (&Jpt)[4], const real_t *w,
const DeviceTensor<7> &H) override
{
MFEM_CONTRACT_VAR(w);
// 0.5 * weight * dI1b
real_t ddI1[4], ddI1b[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).ddI1(ddI1).ddI1b(ddI1b).dI2b(dI2b));
const real_t half_weight = 0.5 * weight;
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t h = ddi1b(r, c);
H(r, c, i, j, qx, qy, e) = half_weight * h;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_002;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 2);
} // namespace mfem
-88
View File
@@ -1,88 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_007 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
return ie.Get_I1() * (1.0 + 1.0 / ie.Get_I2()) - 4.0;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
{
MFEM_CONTRACT_VAR(w);
real_t dI1[4], dI2[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).dI1(dI1).dI2(dI2).dI2b(dI2b));
const real_t I2 = ie.Get_I2();
kernels::Add(2, 2, 1.0 + 1.0 / I2, ie.Get_dI1(), -ie.Get_I1() / (I2 * I2),
ie.Get_dI2(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
{
MFEM_CONTRACT_VAR(w);
real_t ddI1[4], ddI2[4], dI1[4], dI2[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).ddI1(ddI1).ddI2(ddI2).dI1(dI1).dI2(dI2).dI2b(dI2b));
const real_t c1 = 1. / ie.Get_I2();
const real_t c2 = weight * c1 * c1;
const real_t c3 = ie.Get_I1() * c2;
ConstDeviceMatrix di1(ie.Get_dI1(), DIM, DIM);
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1(ie.Get_ddI1(i, j), DIM, DIM);
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r, c, i, j, qx, qy, e) =
weight * (1.0 + c1) * ddi1(r, c) - c3 * ddi2(r, c) -
c2 * (di1(i, j) * di2(r, c) + di2(i, j) * di1(r, c)) +
2.0 * c1 * c3 * di2(r, c) * di2(i, j);
}
}
}
}
}
};
using metric = TMOP_PA_Metric_007;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 7);
} // namespace mfem
-82
View File
@@ -1,82 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_056 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t I2b = ie.Get_I2b();
return 0.5 * (I2b + 1.0 / I2b) - 1.0;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
{
MFEM_CONTRACT_VAR(w);
// 0.5*(1 - 1/I2b^2)*dI2b
real_t dI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2b(dI2b));
const real_t I2b = ie.Get_I2b();
kernels::Set(2, 2, 0.5 * (1.0 - 1.0 / (I2b * I2b)), ie.Get_dI2b(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
{
MFEM_CONTRACT_VAR(w);
// (0.5 - 0.5/I2b^2)*ddI2b + (1/I2b^3)*(dI2b x dI2b)
real_t dI2b[4], ddI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2b(dI2b).ddI2b(ddI2b));
const real_t I2b = ie.Get_I2b();
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r, c, i, j, qx, qy, e) =
weight * (0.5 - 0.5 / (I2b * I2b)) * ddi2b(r, c) +
weight / (I2b * I2b * I2b) * di2b(r, c) * di2b(i, j);
}
}
}
}
}
};
using metric = TMOP_PA_Metric_056;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 56);
} // namespace mfem
-81
View File
@@ -1,81 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_077 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *) override
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t I2b = ie.Get_I2b();
return 0.5 * (I2b * I2b + 1. / (I2b * I2b) - 2.);
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
{
MFEM_CONTRACT_VAR(w);
real_t dI2[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI2(dI2).dI2b(dI2b));
const real_t I2 = ie.Get_I2();
kernels::Set(2, 2, 0.5 * (1.0 - 1.0 / (I2 * I2)), ie.Get_dI2(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
{
MFEM_CONTRACT_VAR(w);
real_t dI2[4], dI2b[4], ddI2[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).dI2(dI2).dI2b(dI2b).ddI2(ddI2));
const real_t I2 = ie.Get_I2(), I2inv_sq = 1.0 / (I2 * I2);
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r, c, i, j, qx, qy, e) =
weight * 0.5 * (1.0 - I2inv_sq) * ddi2(r, c) +
weight * (I2inv_sq / I2) * di2(r, c) * di2(i, j);
}
}
}
}
}
};
using metric = TMOP_PA_Metric_077;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 77);
} // namespace mfem
-89
View File
@@ -1,89 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_080 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *w) override
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t eval_w_02 = 0.5 * ie.Get_I1b() - 1.0;
const real_t I2b = ie.Get_I2b();
const real_t eval_w_77 = 0.5 * (I2b * I2b + 1. / (I2b * I2b) - 2.);
return w[0] * eval_w_02 + w[1] * eval_w_77;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
{
MFEM_CONTRACT_VAR(w);
// w0 P_2 + w1 P_77
real_t dI1b[4], dI2[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).dI1b(dI1b).dI2(dI2).dI2b(dI2b));
kernels::Set(2, 2, w[0] * 0.5, ie.Get_dI1b(), P);
const real_t I2 = ie.Get_I2();
kernels::Add(2, 2, w[1] * 0.5 * (1.0 - 1.0 / (I2 * I2)), ie.Get_dI2(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
{
// w0 H_2 + w1 H_77
real_t ddI1[4], ddI1b[4], dI2[4], dI2b[4], ddI2[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).dI2(dI2).ddI1(ddI1).ddI1b(ddI1b).dI2b(dI2b).ddI2(ddI2));
const real_t I2 = ie.Get_I2(), I2inv_sq = 1.0 / (I2 * I2);
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r, c, i, j, qx, qy, e) =
w[0] * 0.5 * weight * ddi1b(r, c) +
w[1] * (weight * 0.5 * (1.0 - I2inv_sq) * ddi2(r, c) +
weight * (I2inv_sq / I2) * di2(r, c) * di2(i, j));
}
}
}
}
}
};
using metric = TMOP_PA_Metric_080;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 80);
} // namespace mfem
-88
View File
@@ -1,88 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult2.hpp"
#include "../tools/energy2.hpp"
#include "../assemble/grad2.hpp"
namespace mfem
{
struct TMOP_PA_Metric_094 : TMOP_PA_Metric_2D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[4], const real_t *w) override
{
kernels::InvariantsEvaluator2D ie(Args().J(Jpt));
const real_t eval_w_02 = 0.5 * ie.Get_I1b() - 1.0;
const real_t I2b = ie.Get_I2b();
const real_t eval_w_56 = 0.5 * (I2b + 1.0 / I2b) - 1.0;
return w[0] * eval_w_02 + w[1] * eval_w_56;
};
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[4], const real_t *w, real_t (&P)[4]) override
{
// w0 P_2 + w1 P_56
real_t dI1b[4], dI2b[4];
kernels::InvariantsEvaluator2D ie(Args().J(Jpt).dI1b(dI1b).dI2b(dI2b));
kernels::Set(2, 2, w[0] * 0.5, ie.Get_dI1b(), P);
const real_t I2b = ie.Get_I2b();
kernels::Add(2, 2, w[1] * 0.5 * (1.0 - 1.0 / (I2b * I2b)), ie.Get_dI2b(),
P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int e,
const real_t weight,
const real_t (&Jpt)[4],
const real_t *w,
const DeviceTensor<7> &H) override
{
// w0 H_2 + w1 H_56
real_t ddI1[4], ddI1b[4], dI2b[4], ddI2b[4];
kernels::InvariantsEvaluator2D ie(
Args().J(Jpt).ddI1(ddI1).ddI1b(ddI1b).dI2b(dI2b).ddI2b(ddI2b));
const real_t I2b = ie.Get_I2b();
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
H(r, c, i, j, qx, qy, e) =
w[0] * 0.5 * weight * ddi1b(r, c) +
w[1] *
(weight * (0.5 - 0.5 / (I2b * I2b)) * ddi2b(r, c) +
weight / (I2b * I2b * I2b) * di2b(r, c) * di2b(i, j));
}
}
}
}
}
};
using metric = TMOP_PA_Metric_094;
using assemble = TMOPAssembleGradPA2D;
using energy = TMOPEnergyPA2D;
using mult = TMOPAddMultPA2D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 94);
} // namespace mfem
-111
View File
@@ -1,111 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult3.hpp"
#include "../tools/energy3.hpp"
#include "../assemble/grad3.hpp"
namespace mfem
{
struct TMOP_PA_Metric_302 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
// I1b * I2b / 9 - 1
return ie.Get_I1b() * ie.Get_I2b() / 9. - 1.;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
{
MFEM_CONTRACT_VAR(w);
// (I1b/9)*dI2b + (I2b/9)*dI1b
real_t B[9];
real_t dI1b[9], dI2[9], dI2b[9], dI3b[9];
kernels::InvariantsEvaluator3D ie(
Args().J(Jpt).B(B).dI1b(dI1b).dI2(dI2).dI2b(dI2b).dI3b(dI3b));
const real_t alpha = ie.Get_I1b() / 9.;
const real_t beta = ie.Get_I2b() / 9.;
kernels::Add(3, 3, alpha, ie.Get_dI2b(), beta, ie.Get_dI1b(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
{
MFEM_CONTRACT_VAR(Jrt);
MFEM_CONTRACT_VAR(Jpr);
MFEM_CONTRACT_VAR(w);
real_t B[9];
real_t dI1b[9], ddI1b[9];
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
real_t dI3b[9]; // = Jrt;
// (dI2b*dI1b + dI1b*dI2b)/9 + (I1b/9)*ddI2b + (I2b/9)*ddI1b
kernels::InvariantsEvaluator3D ie(Args()
.J(Jpt)
.B(B)
.dI1b(dI1b)
.ddI1b(ddI1b)
.dI2(dI2)
.dI2b(dI2b)
.ddI2(ddI2)
.ddI2b(ddI2b)
.dI3b(dI3b));
const real_t c1 = weight / 9.;
const real_t I1b = ie.Get_I1b();
const real_t I2b = ie.Get_I2b();
ConstDeviceMatrix di1b(ie.Get_dI1b(), DIM, DIM);
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t dp =
(di2b(r, c) * di1b(i, j) + di1b(r, c) * di2b(i, j)) +
ddi2b(r, c) * I1b + ddi1b(r, c) * I2b;
H(r, c, i, j, qx, qy, qz, e) = c1 * dp;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_302;
using assemble = TMOPAssembleGradPA3D;
using energy = TMOPEnergyPA3D;
using mult = TMOPAddMultPA3D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 302);
} // namespace mfem
-103
View File
@@ -1,103 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult3.hpp"
#include "../tools/energy3.hpp"
#include "../assemble/grad3.hpp"
namespace mfem
{
struct TMOP_PA_Metric_303 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
// mu_303 = I1b/3 - 1
return ie.Get_I1b() / 3. - 1.;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
{
MFEM_CONTRACT_VAR(w);
// dI1b/3
real_t B[9];
real_t dI1b[9], dI3b[9];
kernels::InvariantsEvaluator3D ie(
Args().J(Jpt).B(B).dI1b(dI1b).dI3b(dI3b));
kernels::Set(3, 3, 1. / 3., ie.Get_dI1b(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
{
MFEM_CONTRACT_VAR(w);
real_t B[9];
real_t dI1b[9], ddI1[9], ddI1b[9];
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
real_t *dI3b = Jrt, *ddI3b = Jpr;
// ddI1b/3
kernels::InvariantsEvaluator3D ie(Args()
.J(Jpt)
.B(B)
.dI1b(dI1b)
.ddI1(ddI1)
.ddI1b(ddI1b)
.dI2(dI2)
.dI2b(dI2b)
.ddI2(ddI2)
.ddI2b(ddI2b)
.dI3b(dI3b)
.ddI3b(ddI3b));
const real_t c1 = weight / 3.;
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t dp = ddi1b(r, c);
H(r, c, i, j, qx, qy, qz, e) = c1 * dp;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_303;
using assemble = TMOPAssembleGradPA3D;
using energy = TMOPEnergyPA3D;
using mult = TMOPAddMultPA3D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 303);
} // namespace mfem
-91
View File
@@ -1,91 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult3.hpp"
#include "../tools/energy3.hpp"
#include "../assemble/grad3.hpp"
namespace mfem
{
struct TMOP_PA_Metric_315 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
// (I3b - 1)^2
const real_t a = ie.Get_I3b() - 1.0;
return a * a;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
{
MFEM_CONTRACT_VAR(w);
// 2*(I3b - 1)*dI3b
real_t dI3b[9];
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).dI3b(dI3b));
real_t sign_detJ;
const real_t I3b = ie.Get_I3b(sign_detJ);
kernels::Set(3, 3, 2.0 * (I3b - 1.0), ie.Get_dI3b(sign_detJ), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
{
MFEM_CONTRACT_VAR(w);
real_t *dI3b = Jrt, *ddI3b = Jpr;
// 2*(dI3b x dI3b) + 2*(I3b - 1)*ddI3b
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).dI3b(dI3b).ddI3b(ddI3b));
real_t sign_detJ;
const real_t I3b = ie.Get_I3b(sign_detJ);
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t dp = 2.0 * weight * (I3b - 1.0) * ddi3b(r, c) +
2.0 * weight * di3b(r, c) * di3b(i, j);
H(r, c, i, j, qx, qy, qz, e) = dp;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_315;
using assemble = TMOPAssembleGradPA3D;
using energy = TMOPEnergyPA3D;
using mult = TMOPAddMultPA3D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 315);
} // namespace mfem
-97
View File
@@ -1,97 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult3.hpp"
#include "../tools/energy3.hpp"
#include "../assemble/grad3.hpp"
namespace mfem
{
struct TMOP_PA_Metric_318 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
// 0.5 * (I3 + 1/I3) - 1.
const real_t I3 = ie.Get_I3();
return 0.5 * (I3 + 1.0 / I3) - 1.0;
}
// P_318 = (I3b - 1/I3b^3)*dI3b.
// Uses the I3b form, as dI3 and ddI3 were not implemented at the time
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
{
MFEM_CONTRACT_VAR(w);
real_t dI3b[9];
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).dI3b(dI3b));
real_t sign_detJ;
const real_t I3b = ie.Get_I3b(sign_detJ);
kernels::Set(3, 3, I3b - 1.0 / (I3b * I3b * I3b), ie.Get_dI3b(sign_detJ),
P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
{
MFEM_CONTRACT_VAR(w);
real_t *dI3b = Jrt, *ddI3b = Jpr;
// dP_318 = (I3b - 1/I3b^3)*ddI3b + (1 + 3/I3b^4)*(dI3b x dI3b)
// Uses the I3b form, as dI3 and ddI3 were not implemented at the time
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).dI3b(dI3b).ddI3b(ddI3b));
real_t sign_detJ;
const real_t I3b = ie.Get_I3b(sign_detJ);
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t dp =
weight * (I3b - 1.0 / (I3b * I3b * I3b)) * ddi3b(r, c) +
weight * (1.0 + 3.0 / (I3b * I3b * I3b * I3b)) *
di3b(r, c) * di3b(i, j);
H(r, c, i, j, qx, qy, qz, e) = dp;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_318;
using assemble = TMOPAssembleGradPA3D;
using energy = TMOPEnergyPA3D;
using mult = TMOPAddMultPA3D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 318);
} // namespace mfem
-123
View File
@@ -1,123 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult3.hpp"
#include "../tools/energy3.hpp"
#include "../assemble/grad3.hpp"
namespace mfem
{
struct TMOP_PA_Metric_321 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
// I1 + I2/I3 - 6
return ie.Get_I1() + ie.Get_I2() / ie.Get_I3() - 6.0;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
{
MFEM_CONTRACT_VAR(w);
// dI1 + (1/I3)*dI2 - (2*I2/I3b^3)*dI3b
real_t B[9];
real_t dI1[9], dI2[9], dI3b[9];
kernels::InvariantsEvaluator3D ie(
Args().J(Jpt).B(B).dI1(dI1).dI2(dI2).dI3b(dI3b));
real_t sign_detJ;
const real_t I3 = ie.Get_I3();
const real_t alpha = 1.0 / I3;
const real_t beta = -2. * ie.Get_I2() / (I3 * ie.Get_I3b(sign_detJ));
kernels::Add(3, 3, alpha, ie.Get_dI2(), beta, ie.Get_dI3b(sign_detJ), P);
kernels::Add(3, 3, ie.Get_dI1(), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
{
MFEM_CONTRACT_VAR(w);
real_t B[9];
real_t dI1b[9], ddI1[9], ddI1b[9];
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
real_t *dI3b = Jrt, *ddI3b = Jpr;
// ddI1 + (-2/I3b^3)*(dI2 x dI3b + dI3b x dI2)
// + (1/I3)*ddI2
// + (6*I2/I3b^4)*(dI3b x dI3b)
// + (-2*I2/I3b^3)*ddI3b
kernels::InvariantsEvaluator3D ie(Args()
.J(Jpt)
.B(B)
.dI1b(dI1b)
.ddI1(ddI1)
.ddI1b(ddI1b)
.dI2(dI2)
.dI2b(dI2b)
.ddI2(ddI2)
.ddI2b(ddI2b)
.dI3b(dI3b)
.ddI3b(ddI3b));
real_t sign_detJ;
const real_t I2 = ie.Get_I2();
const real_t I3b = ie.Get_I3b(sign_detJ);
ConstDeviceMatrix di2(ie.Get_dI2(), DIM, DIM);
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
const real_t c0 = 1.0 / I3b;
const real_t c1 = weight * c0 * c0;
const real_t c2 = -2 * c0 * c1;
const real_t c3 = c2 * I2;
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1(ie.Get_ddI1(i, j), DIM, DIM);
ConstDeviceMatrix ddi2(ie.Get_ddI2(i, j), DIM, DIM);
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t dp =
weight * ddi1(r, c) + c1 * ddi2(r, c) + c3 * ddi3b(r, c) +
c2 * ((di2(r, c) * di3b(i, j) + di3b(r, c) * di2(i, j))) -
3 * c0 * c3 * di3b(r, c) * di3b(i, j);
H(r, c, i, j, qx, qy, qz, e) = dp;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_321;
using assemble = TMOPAssembleGradPA3D;
using energy = TMOPEnergyPA3D;
using mult = TMOPAddMultPA3D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 321);
} // namespace mfem
-120
View File
@@ -1,120 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult3.hpp"
#include "../tools/energy3.hpp"
#include "../assemble/grad3.hpp"
namespace mfem
{
struct TMOP_PA_Metric_332 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
const real_t eval_w_302 = ie.Get_I1b() * ie.Get_I2b() / 9. - 1.;
const real_t a = ie.Get_I3b() - 1.0;
const real_t eval_w_315 = a * a;
return w[0] * eval_w_302 + w[1] * eval_w_315;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
{
// w0 P_302 + w1 P_315
real_t B[9];
real_t dI1b[9], dI2[9], dI2b[9], dI3b[9];
kernels::InvariantsEvaluator3D ie(
Args().J(Jpt).B(B).dI1b(dI1b).dI2(dI2).dI2b(dI2b).dI3b(dI3b));
const real_t alpha = w[0] * ie.Get_I1b() / 9.;
const real_t beta = w[0] * ie.Get_I2b() / 9.;
kernels::Add(3, 3, alpha, ie.Get_dI2b(), beta, ie.Get_dI1b(), P);
real_t sign_detJ;
const real_t I3b = ie.Get_I3b(sign_detJ);
kernels::Add(3, 3, w[1] * 2.0 * (I3b - 1.0), ie.Get_dI3b(sign_detJ), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
{
real_t B[9];
real_t dI1b[9], /*ddI1[9],*/ ddI1b[9];
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
real_t *dI3b = Jrt, *ddI3b = Jpr;
// w0 H_302 + w1 H_315
kernels::InvariantsEvaluator3D ie(Args()
.J(Jpt)
.B(B)
.dI1b(dI1b)
.ddI1b(ddI1b)
.dI2(dI2)
.dI2b(dI2b)
.ddI2(ddI2)
.ddI2b(ddI2b)
.dI3b(dI3b)
.ddI3b(ddI3b));
real_t sign_detJ;
const real_t c1 = weight / 9.0;
const real_t I1b = ie.Get_I1b();
const real_t I2b = ie.Get_I2b();
const real_t I3b = ie.Get_I3b(sign_detJ);
ConstDeviceMatrix di1b(ie.Get_dI1b(), DIM, DIM);
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t dp_302 =
(di2b(r, c) * di1b(i, j) + di1b(r, c) * di2b(i, j)) +
ddi2b(r, c) * I1b + ddi1b(r, c) * I2b;
const real_t dp_315 =
2.0 * weight * (I3b - 1.0) * ddi3b(r, c) +
2.0 * weight * di3b(r, c) * di3b(i, j);
H(r, c, i, j, qx, qy, qz, e) =
w[0] * c1 * dp_302 + w[1] * dp_315;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_332;
using assemble = TMOPAssembleGradPA3D;
using energy = TMOPEnergyPA3D;
using mult = TMOPAddMultPA3D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 332);
} // namespace mfem
-122
View File
@@ -1,122 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../mult/mult3.hpp"
#include "../tools/energy3.hpp"
#include "../assemble/grad3.hpp"
namespace mfem
{
struct TMOP_PA_Metric_338 : TMOP_PA_Metric_3D
{
MFEM_HOST_DEVICE real_t EvalW(const real_t (&Jpt)[DIM * DIM],
const real_t *w) override
{
real_t B[9];
MFEM_CONTRACT_VAR(w);
kernels::InvariantsEvaluator3D ie(Args().J(Jpt).B(B));
const real_t eval_w_302 = ie.Get_I1b() * ie.Get_I2b() / 9. - 1.;
const real_t I3 = ie.Get_I3();
const real_t eval_w_318 = 0.5 * (I3 + 1.0 / I3) - 1.0;
return w[0] * eval_w_302 + w[1] * eval_w_318;
}
MFEM_HOST_DEVICE
void EvalP(const real_t (&Jpt)[9], const real_t *w, real_t (&P)[9]) override
{
// w0 P_302 + w1 P_318
real_t B[9];
real_t dI1b[9], dI2[9], dI2b[9], dI3b[9];
kernels::InvariantsEvaluator3D ie(
Args().J(Jpt).B(B).dI1b(dI1b).dI2(dI2).dI2b(dI2b).dI3b(dI3b));
const real_t alpha = w[0] * ie.Get_I1b() / 9.;
const real_t beta = w[0] * ie.Get_I2b() / 9.;
kernels::Add(3, 3, alpha, ie.Get_dI2b(), beta, ie.Get_dI1b(), P);
real_t sign_detJ;
const real_t I3b = ie.Get_I3b(sign_detJ);
kernels::Add(3, 3, w[1] * (I3b - 1.0 / (I3b * I3b * I3b)),
ie.Get_dI3b(sign_detJ), P);
}
MFEM_HOST_DEVICE
void AssembleH(const int qx,
const int qy,
const int qz,
const int e,
const real_t weight,
real_t *Jrt,
real_t *Jpr,
const real_t (&Jpt)[9],
const real_t *w,
const DeviceTensor<8> &H) const override
{
real_t B[9];
real_t dI1b[9], ddI1b[9];
real_t dI2[9], dI2b[9], ddI2[9], ddI2b[9];
real_t *dI3b = Jrt, *ddI3b = Jpr;
// w0 H_302 + w1 H_318
kernels::InvariantsEvaluator3D ie(Args()
.J(Jpt)
.B(B)
.dI1b(dI1b)
.ddI1b(ddI1b)
.dI2(dI2)
.dI2b(dI2b)
.ddI2(ddI2)
.ddI2b(ddI2b)
.dI3b(dI3b)
.ddI3b(ddI3b));
real_t sign_detJ;
const real_t c1 = weight / 9.;
const real_t I1b = ie.Get_I1b();
const real_t I2b = ie.Get_I2b();
const real_t I3b = ie.Get_I3b(sign_detJ);
ConstDeviceMatrix di1b(ie.Get_dI1b(), DIM, DIM);
ConstDeviceMatrix di2b(ie.Get_dI2b(), DIM, DIM);
ConstDeviceMatrix di3b(ie.Get_dI3b(sign_detJ), DIM, DIM);
for (int i = 0; i < DIM; i++)
{
for (int j = 0; j < DIM; j++)
{
ConstDeviceMatrix ddi1b(ie.Get_ddI1b(i, j), DIM, DIM);
ConstDeviceMatrix ddi2b(ie.Get_ddI2b(i, j), DIM, DIM);
ConstDeviceMatrix ddi3b(ie.Get_ddI3b(i, j), DIM, DIM);
for (int r = 0; r < DIM; r++)
{
for (int c = 0; c < DIM; c++)
{
const real_t dp_302 =
(di2b(r, c) * di1b(i, j) + di1b(r, c) * di2b(i, j)) +
ddi2b(r, c) * I1b + ddi1b(r, c) * I2b;
const real_t dp_318 =
weight * (I3b - 1.0 / (I3b * I3b * I3b)) * ddi3b(r, c) +
weight * (1.0 + 3.0 / (I3b * I3b * I3b * I3b)) *
di3b(r, c) * di3b(i, j);
H(r, c, i, j, qx, qy, qz, e) =
w[0] * c1 * dp_302 + w[1] * dp_318;
}
}
}
}
}
};
using metric = TMOP_PA_Metric_338;
using assemble = TMOPAssembleGradPA3D;
using energy = TMOPEnergyPA3D;
using mult = TMOPAddMultPA3D;
MFEM_TMOP_REGISTER_METRIC(metric, assemble, energy, mult, 338);
} // namespace mfem
-118
View File
@@ -1,118 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AddMultGradPA_2D(const int NE,
const real_t *b,
const real_t *g,
const DeviceTensor<5, const real_t> &J,
const DeviceTensor<7, const real_t> &H,
const DeviceTensor<4, const real_t> &X,
DeviceTensor<4> &Y,
const int d1d,
const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs2d_t<2, 2, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs2d(e, D1D, X, r0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
// Jrt = Jtr^{-1}
real_t Jrt[4];
kernels::CalcInverse<2>(Jtr, Jrt);
// Jpr = X^T.DSh
const real_t Jpr[4] =
{
r1[0][0][qy][qx], r1[1][0][qy][qx],
r1[0][1][qy][qx], r1[1][1][qy][qx]
};
// Jpt = Jpr . Jrt
real_t Jpt[4];
kernels::Mult(2, 2, 2, Jpr, Jrt, Jpt);
// B = Jpt : H
real_t B[4];
DeviceMatrix M(B, 2, 2);
ConstDeviceMatrix J(Jpt, 2, 2);
for (int i = 0; i < 2; i++)
{
for (int j = 0; j < 2; j++)
{
M(i, j) = 0.0;
for (int r = 0; r < 2; r++)
{
for (int c = 0; c < 2; c++)
{
M(i, j) += H(r, c, i, j, qx, qy, e) * J(r, c);
}
}
}
}
// C = Jrt . B
real_t C[4];
kernels::MultABt(2, 2, 2, Jrt, B, C);
r0[0][0][qy][qx] = C[0], r0[0][1][qy][qx] = C[1];
r0[1][0][qy][qx] = C[2], r0[1][1][qy][qx] = C[3];
}
}
MFEM_SYNC_THREAD;
kernels::internal::GradTranspose2d(D1D, Q1D, smem, sB, sG, r0, r1);
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPMultGradKernels, TMOP_AddMultGradPA_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultGradKernels);
void TMOP_Integrator::AddMultGradPA_2D(const Vector &R, Vector &C) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto *b = PA.maps->B.Read(), *g = PA.maps->G.Read();
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto H = Reshape(PA.H.Read(), 2, 2, 2, 2, q, q, NE);
const auto X = Reshape(R.Read(), d, d, 2, NE);
auto Y = Reshape(C.ReadWrite(), d, d, 2, NE);
TMOPMultGradKernels::Run(d, q, NE, b, g, J, H, X, Y, d, q);
}
} // namespace mfem
-88
View File
@@ -1,88 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AddMultGradPA_C0_2D(const int NE,
const real_t *b,
const DeviceTensor<5, const real_t> &H0,
const DeviceTensor<4, const real_t> &X,
DeviceTensor<4> &Y,
const int d1d,
const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t sB[MD1][MQ1];
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::v_regs2d_t<2,MQ1> r0, r1;
kernels::internal::LoadDofs2d(e, D1D, X, r0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r0, r1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
// Xh = X^T . Sh
const real_t Xh[2] = { r1(0, qy, qx), r1(1, qy, qx) };
real_t H_data[4];
DeviceMatrix H(H_data, 2, 2);
for (int i = 0; i < 2; i++)
{
for (int j = 0; j < 2; j++) { H(i, j) = H0(i, j, qx, qy, e); }
}
// p2 = H . Xh
real_t p2[2];
kernels::Mult(2, 2, H_data, Xh, p2);
r0(0,qy,qx) = p2[0];
r0(1,qy,qx) = p2[1];
}
}
MFEM_SYNC_THREAD;
kernels::internal::EvalTranspose2d(D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPMultGradCoefKernels, TMOP_AddMultGradPA_C0_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultGradCoefKernels);
void TMOP_Integrator::AddMultGradPA_C0_2D(const Vector &R, Vector &C) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto H0 = Reshape(PA.H0.Read(), 2, 2, q, q, NE);
const auto *b = PA.maps->B.Read();
const auto X = Reshape(R.Read(), d, d, 2, NE);
auto Y = Reshape(C.ReadWrite(), d, d, 2, NE);
TMOPMultGradCoefKernels::Run(d, q, NE, b, H0, X, Y, d, q);
}
} // namespace mfem
-121
View File
@@ -1,121 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AddMultGradPA_3D(const int NE,
const real_t *b,
const real_t *g,
const DeviceTensor<6, const real_t> &J,
const DeviceTensor<8, const real_t> &H,
const DeviceTensor<5, const real_t> &X,
DeviceTensor<5> &Y, const int d1d, const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs3d_t<3, 3, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs3d(e, D1D, X, r0);
kernels::internal::Grad3d(D1D, Q1D, smem, sB, sG, r0, r1);
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, qz, e);
// Jrt = Jtr^{-1}
real_t Jrt[9];
kernels::CalcInverse<3>(Jtr, Jrt);
// Jpr = X^T.DSh
const real_t Jpr[9] =
{
r1(0, 0, qz, qy, qx), r1(1, 0, qz, qy, qx), r1(2, 0, qz, qy, qx),
r1(0, 1, qz, qy, qx), r1(1, 1, qz, qy, qx), r1(2, 1, qz, qy, qx),
r1(0, 2, qz, qy, qx), r1(1, 2, qz, qy, qx), r1(2, 2, qz, qy, qx)
};
// Jpt = X^T.DS = (X^T.DSh).Jrt = Jpr.Jrt
real_t Jpt[9];
kernels::Mult(3, 3, 3, Jpr, Jrt, Jpt);
// B = Jpt : H
real_t B[9];
DeviceMatrix M(B, 3, 3);
ConstDeviceMatrix J(Jpt, 3, 3);
for (int i = 0; i < 3; i++)
{
for (int j = 0; j < 3; j++)
{
M(i, j) = 0.0;
for (int r = 0; r < 3; r++)
{
for (int c = 0; c < 3; c++)
{
M(i, j) += H(r, c, i, j, qx, qy, qz, e) * J(r, c);
}
}
}
}
// Y += DS . M^t += DSh . (Jrt . M^t)
real_t A[9];
kernels::MultABt(3, 3, 3, Jrt, B, A);
r0(0,0, qz,qy,qx) = A[0], r0(0,1, qz,qy,qx) = A[1], r0(0,2, qz,qy,qx) = A[2];
r0(1,0, qz,qy,qx) = A[3], r0(1,1, qz,qy,qx) = A[4], r0(1,2, qz,qy,qx) = A[5];
r0(2,0, qz,qy,qx) = A[6], r0(2,1, qz,qy,qx) = A[7], r0(2,2, qz,qy,qx) = A[8];
}
}
}
MFEM_SYNC_THREAD;
kernels::internal::GradTranspose3d(D1D, Q1D, smem, sB, sG, r0, r1);
kernels::internal::WriteDofs3d(e, D1D, r1, Y);
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPMultGradKernels3D, TMOP_AddMultGradPA_3D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultGradKernels3D);
void TMOP_Integrator::AddMultGradPA_3D(const Vector &R, Vector &C) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto *b = PA.maps->B.Read(), *g = PA.maps->G.Read();
const auto J = Reshape(PA.Jtr.Read(), 3, 3, q, q, q, NE);
const auto X = Reshape(R.Read(), d, d, d, 3, NE);
const auto H = Reshape(PA.H.Read(), 3, 3, 3, 3, q, q, q, NE);
auto Y = Reshape(C.ReadWrite(), d, d, d, 3, NE);
TMOPMultGradKernels3D::Run(d, q, NE, b, g, J, H, X, Y, d, q);
}
} // namespace mfem
-101
View File
@@ -1,101 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AddMultGradPA_C0_3D(const int NE,
const real_t *b,
const DeviceTensor<6, const real_t> &H0,
const DeviceTensor<5, const real_t> &X,
DeviceTensor<5> &Y,
const int d1d,
const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t sB[MD1][MQ1];
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::v_regs3d_t<3,MQ1> r0, r1; // vector X
kernels::internal::LoadDofs3d(e, D1D, X, r0);
kernels::internal::Eval3d(D1D, Q1D, smem, sB, r0, r1);
for (int qz = 0; qz < Q1D; ++qz)
{
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
// Xh = X^T . Sh
const real_t Xh[3] =
{
r1(0, qz, qy, qx),
r1(1, qz, qy, qx),
r1(2, qz, qy, qx)
};
real_t H_data[9];
DeviceMatrix H(H_data, 3, 3);
for (int i = 0; i < 3; i++)
{
for (int j = 0; j < 3; j++)
{
H(i, j) = H0(i, j, qx, qy, qz, e);
}
}
// p2 = H . Xh
real_t p2[3];
kernels::Mult(3, 3, H_data, Xh, p2);
r0(0,qz,qy,qx) = p2[0];
r0(1,qz,qy,qx) = p2[1];
r0(2,qz,qy,qx) = p2[2];
}
}
}
MFEM_SYNC_THREAD;
kernels::internal::EvalTranspose3d(D1D, Q1D, smem, sB, r0, r1);
kernels::internal::WriteDofs3d(e, D1D, r1, Y);
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPMultGradCoefKernels3D, TMOP_AddMultGradPA_C0_3D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultGradCoefKernels3D);
void TMOP_Integrator::AddMultGradPA_C0_3D(const Vector &R, Vector &C) const
{
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
const auto H0 = Reshape(PA.H0.Read(), 3, 3, q, q, q, NE);
const auto *b = PA.maps->B.Read();
const auto X = Reshape(R.Read(), d, d, d, 3, NE);
auto Y = Reshape(C.ReadWrite(), d, d, d, 3, NE);
TMOPMultGradCoefKernels3D::Run(d, q, NE, b, H0, X, Y, d, q);
}
} // namespace mfem
-36
View File
@@ -1,36 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "mult2.hpp"
namespace mfem
{
void TMOP_Integrator::AddMultPA_2D(const Vector &x, Vector &y) const
{
const int mid = metric->Id();
TMOPAddMultPA2D ker(this, x, y);
// Calls TMOPAddMultPA2D::Mult for the given mid.
if (mid == 1) { return tmop::Kernel<1>(ker); }
if (mid == 2) { return tmop::Kernel<2>(ker); }
if (mid == 7) { return tmop::Kernel<7>(ker); }
if (mid == 56) { return tmop::Kernel<56>(ker); }
if (mid == 77) { return tmop::Kernel<77>(ker); }
if (mid == 80) { return tmop::Kernel<80>(ker); }
if (mid == 94) { return tmop::Kernel<94>(ker); }
MFEM_ABORT("Unsupported TMOP metric " << mid);
}
} // namespace mfem
-126
View File
@@ -1,126 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#pragma once
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
class TMOPAddMultPA2D
{
const mfem::TMOP_Integrator *ti; // not owned
const Vector &x;
Vector &y;
public:
TMOPAddMultPA2D(const TMOP_Integrator *ti, const Vector &x, Vector &y):
ti(ti),
x(x),
y(y)
{
}
int Ndof() const { return ti->PA.maps->ndof; }
int Nqpt() const { return ti->PA.maps->nqpt; }
template <int MD1, int MQ1, typename METRIC, int T_D1D = 0, int T_Q1D = 0>
static void Mult(TMOPAddMultPA2D &ker)
{
const mfem::TMOP_Integrator *ti = ker.ti;
const real_t metric_normal = ti->metric_normal;
const int NE = ti->PA.ne, d1d = ti->PA.maps->ndof, q1d = ti->PA.maps->nqpt;
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
MFEM_VERIFY(D1D <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(Q1D <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
Array<real_t> mp;
if (auto m = dynamic_cast<TMOP_Combo_QualityMetric *>(ti->metric))
{
m->GetWeights(mp);
}
const real_t *w = mp.Read();
const auto *b = ti->PA.maps->B.Read(), *g = ti->PA.maps->G.Read();
const auto X = Reshape(ker.x.Read(), D1D, D1D, 2, NE);
const auto J = Reshape(ti->PA.Jtr.Read(), 2, 2, Q1D, Q1D, NE);
const auto W = Reshape(ti->PA.ir->GetWeights().Read(), Q1D, Q1D);
auto Y = Reshape(ker.y.ReadWrite(), D1D, D1D, 2, NE);
const Vector &mc = ti->PA.MC;
const bool const_m0 = mc.Size() == 1;
const auto MC = const_m0
? Reshape(mc.Read(), 1, 1, 1)
: Reshape(mc.Read(), Q1D, Q1D, NE);
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t smem[MQ1][MQ1];
MFEM_SHARED real_t sB[MD1][MQ1], sG[MD1][MQ1];
kernels::internal::vd_regs2d_t<2, 2, MQ1> r0, r1;
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::LoadMatrix(D1D, Q1D, g, sG);
kernels::internal::LoadDofs2d(e, D1D, X, r0);
kernels::internal::Grad2d(D1D, Q1D, smem, sB, sG, r0, r1);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
const real_t detJtr = kernels::Det<2>(Jtr);
const real_t m_coef = const_m0 ? MC(0, 0, 0) : MC(qx, qy, e);
const real_t weight = metric_normal * m_coef * W(qx, qy) * detJtr;
// Jrt = Jtr^{-1}
real_t Jrt[4];
kernels::CalcInverse<2>(Jtr, Jrt);
// Jpr = X{^T}.DSh
const real_t Jpr[4] =
{
r1[0][0][qy][qx], r1[1][0][qy][qx],
r1[0][1][qy][qx], r1[1][1][qy][qx]
};
// Jpt = X{^T}.DS = (X{^T}.DSh).Jrt = Jpr.Jrt
real_t Jpt[4];
kernels::Mult(2, 2, 2, Jpr, Jrt, Jpt);
real_t P[4];
METRIC{}.EvalP(Jpt, w, P);
for (int i = 0; i < 4; i++) { P[i] *= weight; }
// PMatO += DS . P^t += DSh . (Jrt . P^t)
real_t A[4];
kernels::MultABt(2, 2, 2, Jrt, P, A);
r0[0][0][qy][qx] = A[0], r0[0][1][qy][qx] = A[1];
r0[1][0][qy][qx] = A[2], r0[1][1][qy][qx] = A[3];
}
}
MFEM_SYNC_THREAD;
kernels::internal::GradTranspose2d(D1D, Q1D, smem, sB, sG, r0, r1);
kernels::internal::WriteDofs2d(e, D1D, r1, Y);
});
}
};
} // namespace mfem
-143
View File
@@ -1,143 +0,0 @@
// Copyright (c) 2010-2025, Lawrence Livermore National Security, LLC. Produced
// at the Lawrence Livermore National Laboratory. All Rights reserved. See files
// LICENSE and NOTICE for details. LLNL-CODE-806117.
//
// This file is part of the MFEM library. For more information and source code
// availability visit https://mfem.org.
//
// MFEM is free software; you can redistribute it and/or modify it under the
// terms of the BSD-3 license. We welcome feedback and contributions, see file
// CONTRIBUTING.md for details.
#include "../pa.hpp"
#include "../../tmop.hpp"
#include "../../kernels.hpp"
#include "../../../general/forall.hpp"
#include "../../../linalg/kernels.hpp"
namespace mfem
{
template <int MD1, int MQ1, int T_D1D = 0, int T_Q1D = 0>
void TMOP_AddMultPA_C0_2D(const real_t lim_normal,
const ConstDeviceCube &LD,
const bool const_c0,
const DeviceTensor<3, const real_t> &C0,
const int NE,
const DeviceTensor<5, const real_t> &J,
const ConstDeviceMatrix &W,
const real_t *b,
const real_t *bld,
const DeviceTensor<4, const real_t> &X0,
const DeviceTensor<4, const real_t> &X1,
DeviceTensor<4> &Y,
const bool exp_lim,
const int d1d,
const int q1d)
{
const int D1D = T_D1D ? T_D1D : d1d;
const int Q1D = T_Q1D ? T_Q1D : q1d;
mfem::forall_2D(NE, Q1D, Q1D, [=] MFEM_HOST_DEVICE(int e)
{
MFEM_SHARED real_t sB[MD1][MQ1];
MFEM_SHARED real_t smem[MQ1][MQ1];
kernels::internal::LoadMatrix(D1D, Q1D, bld, sB);
kernels::internal::s_regs2d_t<MQ1> rm0, rm1; // scalar LD
kernels::internal::LoadDofs2d(e, D1D, LD, rm0);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, rm0, rm1);
kernels::internal::LoadMatrix(D1D, Q1D, b, sB);
kernels::internal::v_regs2d_t<2,MQ1> r00, r01; // vector X0
kernels::internal::LoadDofs2d(e, D1D, X0, r00);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r00, r01);
kernels::internal::v_regs2d_t<2,MQ1> r10, r11; // vector X1
kernels::internal::LoadDofs2d(e, D1D, X1, r10);
kernels::internal::Eval2d(D1D, Q1D, smem, sB, r10, r11);
MFEM_FOREACH_THREAD_DIRECT(qy, y, Q1D)
{
MFEM_FOREACH_THREAD_DIRECT(qx, x, Q1D)
{
const real_t *Jtr = &J(0, 0, qx, qy, e);
const real_t detJtr = kernels::Det<2>(Jtr);
const real_t weight = W(qx, qy) * detJtr;
const real_t ld = rm1(qy, qx);
const real_t p0[2] = { r01(0, qy, qx), r01(1, qy, qx) };
const real_t p1[2] = { r11(0, qy, qx), r11(1, qy, qx) };
const real_t coeff0 = const_c0 ? C0(0, 0, 0) : C0(qx, qy, e);
const real_t dist = ld; // GetValues, default comp set to 0
real_t d1[2];
// Eval_d1 (Quadratic Limiter)
// subtract(1.0 / (dist * dist), x, x0, d1);
// z = a * (x - y)
// grad = a * (x - x0)
// Eval_d1 (Exponential Limiter)
// real_t dist_squared = dist*dist;
// subtract(20.0*exp(10.0*((x.DistanceSquaredTo(x0) / dist_squared)
// - 1.0)) / dist_squared, x, x0, d1); z = a * (x - y) grad = a * (x
// - x0)
real_t a = 0.0;
const real_t w = weight * lim_normal * coeff0;
const real_t dist_squared = dist * dist;
if (!exp_lim) { a = 1.0 / dist_squared; }
else
{
real_t dsq = kernels::DistanceSquared<2>(p1, p0) / dist_squared;
a = 20.0 * exp(10.0 * (dsq - 1.0)) / dist_squared;
}
kernels::Subtract<2>(w * a, p1, p0, d1);
r00(0,qy,qx) = d1[0];
r00(1,qy,qx) = d1[1];
}
}
MFEM_SYNC_THREAD;
kernels::internal::EvalTranspose2d(D1D, Q1D, smem, sB, r00, r01);
kernels::internal::WriteDofs2d(e, D1D, r01, Y);
});
}
MFEM_TMOP_MDQ_REGISTER(TMOPMultCoefKernels, TMOP_AddMultPA_C0_2D);
MFEM_TMOP_MDQ_SPECIALIZE(TMOPMultCoefKernels);
void TMOP_Integrator::AddMultPA_C0_2D(const Vector &x, Vector &y) const
{
const real_t ln = lim_normal;
const int NE = PA.ne, d = PA.maps->ndof, q = PA.maps->nqpt;
MFEM_VERIFY(d <= DeviceDofQuadLimits::Get().MAX_D1D, "");
MFEM_VERIFY(q <= DeviceDofQuadLimits::Get().MAX_Q1D, "");
MFEM_VERIFY(PA.maps_lim->ndof == d, "");
MFEM_VERIFY(PA.maps_lim->nqpt == q, "");
const bool const_c0 = PA.C0.Size() == 1;
const auto C0 = const_c0
? Reshape(PA.C0.Read(), 1, 1, 1)
: Reshape(PA.C0.Read(), q, q, NE);
const auto LD = Reshape(PA.LD.Read(), d, d, NE);
const auto J = Reshape(PA.Jtr.Read(), 2, 2, q, q, NE);
const auto *b = PA.maps->B.Read(), *bld = PA.maps_lim->B.Read();
const auto W = Reshape(PA.ir->GetWeights().Read(), q, q);
const auto XL = Reshape(PA.XL.Read(), d, d, 2, NE);
const auto X = Reshape(x.Read(), d, d, 2, NE);
auto Y = Reshape(y.ReadWrite(), d, d, 2, NE);
auto el = dynamic_cast<TMOP_ExponentialLimiter *>(lim_func);
const bool exp_lim = (el) ? true : false;
TMOPMultCoefKernels::Run(d, q, ln, LD, const_c0, C0, NE, J, W, b, bld, XL, X,
Y, exp_lim, d, q);
}
} // namespace mfem

Some files were not shown because too many files have changed in this diff Show More