From cf8040f9174d1812ed4039580d4074e217dc8aeb Mon Sep 17 00:00:00 2001 From: Spencer Bryngelson Date: Fri, 11 Sep 2026 10:38:32 -0400 Subject: [PATCH] Make the IB force-reduction receive buffers device resident `s_communicate_ib_forces` passed `recv_forces_snap`, `recv_torques_snap`, `recv_ids` and `recv_ft` into GPU kernels through `copy`/`copyin`, but allocated them with a plain `allocate`, so they never entered the device present table. On Cray OpenMP offload this aborts at the first time step of any moving-IB case run on more than one rank: ACC: find_in_present_table failed for 'recv_forces_snap(:,:)' from m_ibm.fpp:1338 ACC: libcrayacc/acc_runtime.c:703 CRAY_ACC_ERROR - Variable not found in present table The sibling send buffers `send_ids`/`send_ft` are already `@:ALLOCATE`d and pushed with `GPU_UPDATE`, so the receive side was simply inconsistent with them. Allocate the four receive arrays the same way and update them to the device after the host writes (the zeroing before each accumulation pass, and each `MPI_UNPACK`), which also lets the kernels drop the `copy`/`copyin` of those arrays. Reproduced on Frontier (`./mfc.sh build --gpu mp`, cpe/25.03, rocm/6.3.1) with a 2D moving flat plate on 4 ranks; the same case runs to completion with this change, and an OpenACC build was unaffected either way. Fixes #1840 Claude-Session: https://claude.ai/code/session_01HMJ7cycfo7kTFSFq5yhHLG --- src/simulation/m_ibm.fpp | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/src/simulation/m_ibm.fpp b/src/simulation/m_ibm.fpp index 99a212260..ac09cee94 100644 --- a/src/simulation/m_ibm.fpp +++ b/src/simulation/m_ibm.fpp @@ -97,8 +97,8 @@ contains #ifdef MFC_MPI if (num_procs > 1) then @:ALLOCATE(send_ids(size(patch_ib)), send_ft(6, size(patch_ib))) - allocate (recv_forces_snap(size(patch_ib), 3), recv_torques_snap(size(patch_ib), 3), recv_ids(size(patch_ib)), & - & recv_ft(6, size(patch_ib))) + @:ALLOCATE(recv_forces_snap(size(patch_ib), 3), recv_torques_snap(size(patch_ib), 3), recv_ids(size(patch_ib)), & + & recv_ft(6, size(patch_ib))) end if #endif @@ -1310,6 +1310,7 @@ contains recv_forces_snap = 0._wp recv_torques_snap = 0._wp + $:GPU_UPDATE(device='[recv_forces_snap, recv_torques_snap]') tag = 300 do k = 1, min(2*ib_neighborhood_radius, num_procs_${X}$ - 1) @@ -1335,8 +1336,8 @@ contains call MPI_UNPACK(ib_force_recv_buf, buf_size, unpack_pos, recv_ids, recv_count, MPI_INTEGER, & & MPI_COMM_WORLD, ierr) call MPI_UNPACK(ib_force_recv_buf, buf_size, unpack_pos, recv_ft, 6*recv_count, mpi_p, MPI_COMM_WORLD, ierr) - $:GPU_PARALLEL_LOOP(private='[i, j]', copyin='[recv_ft, recv_ids]', copy='[forces, torques, & - & recv_forces_snap, recv_torques_snap]') + $:GPU_UPDATE(device='[recv_ids, recv_ft]') + $:GPU_PARALLEL_LOOP(private='[i, j]', copy='[forces, torques]') do i = 1, recv_count call s_get_neighborhood_idx(recv_ids(i), j) if (j > 0) then @@ -1381,7 +1382,8 @@ contains call MPI_UNPACK(ib_force_recv_buf, buf_size, unpack_pos, recv_ids, recv_count, MPI_INTEGER, & & MPI_COMM_WORLD, ierr) call MPI_UNPACK(ib_force_recv_buf, buf_size, unpack_pos, recv_ft, 6*recv_count, mpi_p, MPI_COMM_WORLD, ierr) - $:GPU_PARALLEL_LOOP(private='[i, j]', copyin='[recv_ft, recv_ids]', copy='[forces, torques]') + $:GPU_UPDATE(device='[recv_ids, recv_ft]') + $:GPU_PARALLEL_LOOP(private='[i, j]', copy='[forces, torques]') do i = 1, recv_count call s_get_neighborhood_idx(recv_ids(i), j) if (j > 0) then @@ -1599,7 +1601,7 @@ contains #ifdef MFC_MPI if (num_procs > 1) then @:DEALLOCATE(send_ids, send_ft) - deallocate (recv_forces_snap, recv_torques_snap, recv_ids, recv_ft) + @:DEALLOCATE(recv_forces_snap, recv_torques_snap, recv_ids, recv_ft) end if #endif