subroutine ocean_pressure_force_apply(pgf, ms, dt, no_wait)
!! Forward-Euler accumulation of the PGF acceleration onto the face
!! velocities. Additive (not overwriting), so apply ordering vs the
!! Coriolis apply doesn't matter before the next tendency-compute.
!! `no_wait` (optional, default .false. ⇒ blocking): when .true. the
!! apply loops run on OpenACC queue 1 and return WITHOUT syncing, so
!! the batched velocity-apply chain can `!$acc wait(1)` ONCE. Not
!! `pure` (async/wait directives); still functionally pure.
type(ocean_pressure_force_t), intent(in) :: pgf
type(multilayer_state_t), intent(inout) :: ms
real(wp), intent(in) :: dt
logical, intent(in), optional :: no_wait
integer :: i, j, k, nx_face, ny_uface, nx_vface, ny_face, nz
logical :: lwait
lwait = .true.
if (present(no_wait)) lwait = .not. no_wait
nx_face = size(ms%u_face_x_layer, 1)
ny_uface = size(ms%u_face_x_layer, 2)
nx_vface = size(ms%v_face_y_layer, 1)
ny_face = size(ms%v_face_y_layer, 2)
nz = ms%nz_ml
!$acc kernels async(1)
do concurrent(k=1:nz, j=1:ny_uface, i=1:nx_face)
ms%u_face_x_layer(i, j, k) = ms%u_face_x_layer(i, j, k) + &
dt*pgf%dpdx_face%data(i, j, k)
end do
do concurrent(k=1:nz, j=1:ny_face, i=1:nx_vface)
ms%v_face_y_layer(i, j, k) = ms%v_face_y_layer(i, j, k) + &
dt*pgf%dpdy_face%data(i, j, k)
end do
!$acc end kernels
if (lwait) then
!$acc wait(1)
end if
end subroutine ocean_pressure_force_apply