diff --git a/cmake/MFCTargets.cmake b/cmake/MFCTargets.cmake index e2ed8e736..3951d4272 100644 --- a/cmake/MFCTargets.cmake +++ b/cmake/MFCTargets.cmake @@ -222,7 +222,8 @@ exit 0 # Raising the cap restores full pointer precision for the whole image and # makes kernel quality independent of unrelated edits, at the price of a # longer device link. See docs/documentation/gpuParallelization.md - # ("AMD flang known issues") for the failure signature. + # ("AMD flang known issues") for the failure signature. Still needed on AFAR 24.3 + # (cap unchanged; ROCm/llvm-project#4070). target_link_options(${a_target} PRIVATE -fopenmp --offload-arch=gfx90a -flto-partitions=${MFC_BUILD_JOBS} "SHELL:-Xoffload-linker -mllvm -Xoffload-linker -attributor-max-pi-accesses=16384") diff --git a/docs/documentation/gpuParallelization.md b/docs/documentation/gpuParallelization.md index 4a03b2f23..a62317035 100644 --- a/docs/documentation/gpuParallelization.md +++ b/docs/documentation/gpuParallelization.md @@ -885,7 +885,9 @@ MFC's build raises the cap (`-attributor-max-pi-accesses=16384`, passed to the o linker in `cmake/MFCTargets.cmake`), which restores full pointer precision for the whole image and makes kernel quality independent of unrelated edits. The cost is a longer device link. If a build's device link is unexpectedly slow, this flag is why — do not -remove it; kernel performance becomes nondeterministic across commits without it. +remove it; kernel performance becomes nondeterministic across commits without it. The cap +is unchanged in AFAR 24.3 ([ROCm/llvm-project#4070](https://github.com/ROCm/llvm-project/issues/4070); +fix proposed in [#4094](https://github.com/ROCm/llvm-project/pull/4094)). The failure signature without the flag: after adding a kernel, unrelated kernels' resource usage shifts image-wide (uniform LDS increase, scratch/spill jumps visible in @@ -902,7 +904,8 @@ while the host still registers it. The first launch aborts with omptarget error: Failed to load kernel ... followed by a segmentation fault. Never place a GPU kernel inside a `block` construct; -hoist it into its own (module) subroutine with the locals passed as arguments. +hoist it into its own (module) subroutine with the locals passed as arguments. A minimal +reproducer runs correctly on AFAR 24.3, but this is not yet verified inside MFC. ## Silent-Failure Traps {#silent-failure-traps} @@ -940,14 +943,13 @@ answer is wrong, or one backend diverges from all the others. passes a `parameter` array from `m_thermochem`, such as `molecular_weights`, into a declare-target routine. Read such arrays directly in the kernel, or pass a plain local computed from them. -- **The `USING_AMD` fypp guards are load-bearing, not a stale workaround.** They swap a - device-global array bound for a literal in `src/common/include/shared_parallel_macros.fpp` - and its 86 use sites. Setting `USING_AMD = False` and rebuilding amdflang `--gpu mp` - without case optimization compiles completely clean, then produces NaNs in CBC, the - `wave_speeds=2` Riemann path, immersed boundaries, surface tension, QBMM and viscous - cases, and MHD HLLD, while both Lagrange bubble cases complete with out-of-tolerance - answers. A compile-only check returns green, so any attempt to remove these must run the - tests rather than just build. +- **The `USING_AMD` fypp guards are load-bearing for performance.** They swap a device-global + array bound for a literal in `src/common/include/shared_parallel_macros.fpp` and its use + sites. On AFAR 24.3, `USING_AMD = False` (no case optimization) gives correct results but runs + 4-5x slower: private arrays sized by runtime globals move from registers to scratch (the + WENO kernel goes from 0 to 400 B of scratch per thread and runs 11x slower; HLLC 4.9x). On + AFAR 23.2.x it also produced NaNs in CBC, `wave_speeds=2`, IBM, surface tension, QBMM, + viscous, and MHD HLLD cases. Removing them needs a benchmark, not just the tests. - `@:ACC_SETUP_VFs` and `@:ACC_SETUP_SFs` compile only under Cray. Around MPI, use `GPU_UPDATE(host=...)` before a send and `GPU_UPDATE(device=...)` after a receive. diff --git a/src/simulation/m_riemann_solver_hllc.fpp b/src/simulation/m_riemann_solver_hllc.fpp index a94badf27..7bb7eec1a 100644 --- a/src/simulation/m_riemann_solver_hllc.fpp +++ b/src/simulation/m_riemann_solver_hllc.fpp @@ -886,7 +886,7 @@ contains ! after the .fpp line of its GPU_PARALLEL_LOOP, so one shared call would give ! both emissions the same name; amdflang then launches the wrong one and a ! hypoelastic run faults inside the pure-fluid kernel. Two call sites are what - ! give two line numbers. Do not merge them back into one. + ! give two line numbers. Do not merge them back into one (still faults on AFAR 24.3). #:if HYPO $:GPU_PARALLEL_LOOP(collapse=3, private=_hllc_priv, copyin='[is1, is2, is3]', & & firstprivate='[Re_size_loc1, Re_size_loc2]') diff --git a/src/simulation/m_weno.fpp b/src/simulation/m_weno.fpp index ef18917ab..89cc45f77 100644 --- a/src/simulation/m_weno.fpp +++ b/src/simulation/m_weno.fpp @@ -390,18 +390,7 @@ contains & i + 1) = ((w(1) - w(5))*(w(2) - w(5))*(w(3) - w(5)))/((w(1) - w(8))*(w(2) - w(8)) & & *(w(3) - w(8))) - ! Element-wise on purpose - do NOT rewrite as a reversed-stride section (e.g. s_cb(i+1:i-2:-1)): - ! negative-stride sections of descriptor arrays lower to address arithmetic whose no-wrap - ! (nuw) claims are false, which amdflang (AFAR drop-23.2.x, flang PR #184573; fixed upstream - ! in #198014) turns into silently wrong WENO7 coefficients at -O2/-O3. - w(1) = s_cb(i + 4) - s_cb(i) - w(2) = s_cb(i + 3) - s_cb(i) - w(3) = s_cb(i + 2) - s_cb(i) - w(4) = s_cb(i + 1) - s_cb(i) - w(5) = s_cb(i) - s_cb(i) - w(6) = s_cb(i - 1) - s_cb(i) - w(7) = s_cb(i - 2) - s_cb(i) - w(8) = s_cb(i - 3) - s_cb(i) + w = s_cb(i + 4:i - 3:-1) - s_cb(i) d_cbL_${XYZ}$ (0, & & i + 1) = ((w(1) - w(5))*(w(2) - w(5))*(w(3) - w(5)))/((w(1) - w(8))*(w(2) - w(8)) & & *(w(3) - w(8))) @@ -471,11 +460,7 @@ contains & 2) = (y(4)*(y(3) + y(4))*(y(2) + y(3) + y(4)))/((y(1) + y(2))*(y(1) + y(2) & & + y(3))*(y(1) + y(2) + y(3) + y(4))) - ! Element-wise: see the no-reversed-sections note above. - y(1) = s_cb(i + 1) - s_cb(i) - y(2) = s_cb(i) - s_cb(i - 1) - y(3) = s_cb(i - 1) - s_cb(i - 2) - y(4) = s_cb(i - 2) - s_cb(i - 3) + y = s_cb(i + 1:i - 2:-1) - s_cb(i:i - 3:-1) poly_coef_cbL_${XYZ}$ (i + 1, 3, & & 2) = (y(1)*y(2)*(y(2) + y(3)))/((y(3) + y(4))*(y(2) + y(3) + y(4))*(y(1) & & + y(2) + y(3) + y(4))) @@ -488,11 +473,7 @@ contains & + 4*y(2)*y(3) + 2*y(4)*y(2) + y(3)**2 + y(4)*y(3)))/((y(1) + y(2))*(y(1) & & + y(2) + y(3))*(y(1) + y(2) + y(3) + y(4))) - ! Element-wise: see the no-reversed-sections note above. - y(1) = s_cb(i + 2) - s_cb(i + 1) - y(2) = s_cb(i + 1) - s_cb(i) - y(3) = s_cb(i) - s_cb(i - 1) - y(4) = s_cb(i - 1) - s_cb(i - 2) + y = s_cb(i + 2:i - 1:-1) - s_cb(i + 1:i - 2:-1) poly_coef_cbL_${XYZ}$ (i + 1, 2, & & 2) = -(y(2)*y(3)*(y(1) + y(2)))/((y(3) + y(4))*(y(2) + y(3) + y(4))*(y(1) & & + y(2) + y(3) + y(4))) @@ -504,11 +485,7 @@ contains & 0) = (y(2)*y(3)*(y(3) + y(4)))/((y(1) + y(2))*(y(1) + y(2) + y(3))*(y(1) & & + y(2) + y(3) + y(4))) - ! Element-wise: see the no-reversed-sections note above. - y(1) = s_cb(i + 3) - s_cb(i + 2) - y(2) = s_cb(i + 2) - s_cb(i + 1) - y(3) = s_cb(i + 1) - s_cb(i) - y(4) = s_cb(i) - s_cb(i - 1) + y = s_cb(i + 3:i:-1) - s_cb(i + 2:i - 1:-1) poly_coef_cbL_${XYZ}$ (i + 1, 1, & & 2) = (y(3)*(y(2) + y(3))*(y(1) + y(2) + y(3)))/((y(3) + y(4))*(y(2) + y(3) & & + y(4))*(y(1) + y(2) + y(3) + y(4))) @@ -520,11 +497,7 @@ contains & 0) = -(y(3)*y(4)*(y(2) + y(3)))/((y(1) + y(2))*(y(1) + y(2) + y(3))*(y(1) & & + y(2) + y(3) + y(4))) - ! Element-wise: see the no-reversed-sections note above. - y(1) = s_cb(i + 4) - s_cb(i + 3) - y(2) = s_cb(i + 3) - s_cb(i + 2) - y(3) = s_cb(i + 2) - s_cb(i + 1) - y(4) = s_cb(i + 1) - s_cb(i) + y = s_cb(i + 4:i + 1:-1) - s_cb(i + 3:i:-1) poly_coef_cbL_${XYZ}$ (i + 1, 0, & & 2) = (y(4)*(y(2)**2 + 4*y(2)*y(3) + 4*y(2)*y(4) + y(1)*y(2) + 3*y(3)**2 & & + 6*y(3)*y(4) + 2*y(1)*y(3) + 3*y(4)**2 + 2*y(1)*y(4)))/((y(3) + y(4))*(y(2) &