subroutine coriolis_adv_apply_tendencies(this, ms, dt, no_wait)
!! Per-layer forward-Euler velocity update.
!! `no_wait` (optional, default .false.): when .true. the apply DC
!! loops are issued on OpenACC queue 1 and the routine returns WITHOUT
!! syncing, so a batched caller (`run_stage_split` velocity-apply chain)
!! can pipeline the whole additive apply sequence and `!$acc wait(1)`
!! ONCE. Default ⇒ self-contained blocking apply (historical, safe for
!! non-batched callers — e.g. the unsplit `run_stage`). Not `pure`
!! because of the async/wait directives; still functionally pure.
type(coriolis_adv_t), intent(in) :: this
type(multilayer_state_t), intent(inout) :: ms
real(wp), intent(in) :: dt
logical, intent(in), optional :: no_wait
integer :: i, j, k, nx_face, ny_uface, nx_vface, ny_face, nz
logical :: lwait
lwait = .true.
if (present(no_wait)) lwait = .not. no_wait
nx_face = size(ms%u_face_x_layer, 1)
ny_uface = size(ms%u_face_x_layer, 2)
nx_vface = size(ms%v_face_y_layer, 1)
ny_face = size(ms%v_face_y_layer, 2)
nz = ms%nz_ml
!$acc kernels async(1)
do concurrent(k=1:nz, j=1:ny_uface, i=1:nx_face)
ms%u_face_x_layer(i, j, k) = ms%u_face_x_layer(i, j, k) + &
dt*this%pv_flux_x%data(i, j, k)
end do
do concurrent(k=1:nz, j=1:ny_face, i=1:nx_vface)
ms%v_face_y_layer(i, j, k) = ms%v_face_y_layer(i, j, k) + &
dt*this%pv_flux_y%data(i, j, k)
end do
!$acc end kernels
if (lwait) then
!$acc wait(1)
end if
end subroutine coriolis_adv_apply_tendencies