subroutine ocean_bottom_drag_apply_tendencies(this, ms, dt, no_wait)
!! `no_wait` (optional, default .false.): when .true. the apply DC
!! loops run on OpenACC queue 1 and the routine returns WITHOUT
!! syncing, so the batched velocity-apply chain in `run_stage_split`
!! `!$acc wait(1)`s ONCE. Default ⇒ blocking (safe for the unsplit
!! `run_stage`). Not `pure` because of the async/wait directives.
type(ocean_bottom_drag_t), intent(in) :: this
type(multilayer_state_t), intent(inout) :: ms
real(wp), intent(in) :: dt
logical, intent(in), optional :: no_wait
integer :: i, j, k, nx_face, ny_uface, nx_vface, ny_face, nz
logical :: lwait
lwait = .true.
if (present(no_wait)) lwait = .not. no_wait
nx_face = size(ms%u_face_x_layer, 1)
ny_uface = size(ms%u_face_x_layer, 2)
nx_vface = size(ms%v_face_y_layer, 1)
ny_face = size(ms%v_face_y_layer, 2)
nz = ms%nz_ml
!$acc kernels async(1)
do concurrent(k=1:nz, j=1:ny_uface, i=1:nx_face)
ms%u_face_x_layer(i, j, k) = ms%u_face_x_layer(i, j, k) + &
dt*this%du_drag%data(i, j, k)
end do
do concurrent(k=1:nz, j=1:ny_face, i=1:nx_vface)
ms%v_face_y_layer(i, j, k) = ms%v_face_y_layer(i, j, k) + &
dt*this%dv_drag%data(i, j, k)
end do
!$acc end kernels
if (lwait) then
!$acc wait(1)
end if
end subroutine ocean_bottom_drag_apply_tendencies