pure subroutine fold_sample_unmasked_impl(accum, out, dt, time_op)
!! Whole-buffer fold without a region mask. One `do concurrent`
!! per op so the compiler can specialise — case-inside-loop blocks
!! NVHPC device codegen.
! assumed-shape-ok: diag fold — fires once per output frame (cadence-bounded).
real(wp), intent(inout) :: accum(:, :, :)
real(wp), intent(in) :: out(:, :, :) ! assumed-shape-ok: diag fold — cadence-bounded
real(wp), intent(in) :: dt
integer, intent(in) :: time_op
integer :: i, j, k, nx, ny, nz
nx = size(accum, 1)
ny = size(accum, 2)
nz = size(accum, 3)
select case (time_op)
case (DIAG_OP_MEAN, DIAG_OP_INTEGRAL)
do concurrent(k=1:nz, j=1:ny, i=1:nx)
accum(i, j, k) = accum(i, j, k) + out(i, j, k)*dt
end do
case (DIAG_OP_MAX)
do concurrent(k=1:nz, j=1:ny, i=1:nx)
accum(i, j, k) = max(accum(i, j, k), out(i, j, k))
end do
case (DIAG_OP_MIN)
do concurrent(k=1:nz, j=1:ny, i=1:nx)
accum(i, j, k) = min(accum(i, j, k), out(i, j, k))
end do
case default
! INSTANT and unknown ops don't accumulate. The caller in
! `fold_sample` already returns early when `accumulator` is
! unallocated (which is the INSTANT path), so reaching here
! with an unknown op is a defensive no-op.
end select
end subroutine fold_sample_unmasked_impl