diff --git a/cmake/MFCTargets.cmake b/cmake/MFCTargets.cmake index e2ed8e736..39e4a4e0a 100644 --- a/cmake/MFCTargets.cmake +++ b/cmake/MFCTargets.cmake @@ -205,13 +205,16 @@ exit 0 # (a suffix of loop iterations silently skipped), which -O3 happens to # mask today. Codegen is unchanged on amdflang; it is a win on upstream # flang. + # assume-no-thread-state: no kernel changes OpenMP ICVs on the device, so + # the per-kernel thread-state bookkeeping is dropped (~1.2x on AFAR 24.3). target_compile_options(${a_target} PRIVATE -fopenmp --offload-arch=gfx90a -O3 -fopenmp-assume-threads-oversubscription -fopenmp-assume-teams-oversubscription - -fopenmp-assume-no-nested-parallelism) + -fopenmp-assume-no-nested-parallelism + -fopenmp-assume-no-thread-state) # attributor-max-pi-accesses: amdflang generates device code for the WHOLE # image at link time, and once the image carries enough target regions the # device link's Attributor exceeds its AAPointerInfo access cap on a diff --git a/src/simulation/m_global_parameters.fpp b/src/simulation/m_global_parameters.fpp index e72afb668..583298952 100644 --- a/src/simulation/m_global_parameters.fpp +++ b/src/simulation/m_global_parameters.fpp @@ -9,7 +9,7 @@ module m_global_parameters #ifdef MFC_MPI - use mpi !< Message passing interface (MPI) module + use mpi #endif use m_derived_types