halo_allreduce_sum_i8 Subroutine

public subroutine halo_allreduce_sum_i8(local_val, global_val)

Cross-rank int64 sum for the decomposition-invariant chksum bitcount (rdb_ocean_chksum). pic_mpi_lib exposes no integer(int64) allreduce overload (MPI_INTEGER8 reaches only send/recv), so the value rides the EXACT real64 allreduce: a per-field POPCNT sum is bounded by (#elements x 64), which stays FAR below the double-mantissa bound 253 for any realistic grid (253/64 ~ 1.4e14 cells), so both the local->double cast and every partial MPI_SUM are exact — the invariant survives.

Arguments

Type IntentOptional Attributes Name
integer(kind=int64), intent(in) :: local_val
integer(kind=int64), intent(out) :: global_val

Calls

proc~~halo_allreduce_sum_i8~~CallsGraph proc~halo_allreduce_sum_i8 halo_allreduce_sum_i8 allreduce allreduce proc~halo_allreduce_sum_i8->allreduce proc~comm_env_compute_comm comm_env_compute_comm proc~halo_allreduce_sum_i8->proc~comm_env_compute_comm comm_world comm_world proc~comm_env_compute_comm->comm_world

Called by

proc~~halo_allreduce_sum_i8~~CalledByGraph proc~halo_allreduce_sum_i8 halo_allreduce_sum_i8 proc~chksum_row chksum_row proc~chksum_row->proc~halo_allreduce_sum_i8 proc~engine_step_ice engine_step_ice proc~engine_step_ice->proc~halo_allreduce_sum_i8 proc~driver_run_ocean driver_run_ocean proc~driver_run_ocean->proc~engine_step_ice proc~rdb_debug_chksum_2d rdb_debug_chksum_2d proc~rdb_debug_chksum_2d->proc~chksum_row proc~rdb_debug_chksum_3d rdb_debug_chksum_3d proc~rdb_debug_chksum_3d->proc~chksum_row proc~rdb_ocean_step rdb_ocean_step proc~rdb_ocean_step->proc~engine_step_ice interface~rdb_debug_chksum rdb_debug_chksum interface~rdb_debug_chksum->proc~rdb_debug_chksum_2d interface~rdb_debug_chksum->proc~rdb_debug_chksum_3d proc~driver_run driver_run proc~driver_run->proc~driver_run_ocean proc~chksum_bt chksum_bt proc~chksum_bt->interface~rdb_debug_chksum proc~chksum_state chksum_state proc~chksum_state->interface~rdb_debug_chksum proc~ocean_dyn_step_split ocean_dyn_step_split proc~ocean_dyn_step_split->proc~chksum_state proc~run_stage_split run_stage_split proc~run_stage_split->proc~chksum_bt proc~run_stage_split->proc~chksum_state

Variables

Type Visibility Attributes Name Initial
real(kind=real64), private :: acc_global
real(kind=real64), private :: acc_local
type(comm_t), private :: comm

Source Code

   subroutine halo_allreduce_sum_i8(local_val, global_val)
      !! Cross-rank int64 sum for the decomposition-invariant chksum
      !! bitcount (`rdb_ocean_chksum`).  `pic_mpi_lib` exposes no
      !! `integer(int64)` allreduce overload (MPI_INTEGER8 reaches only
      !! send/recv), so the value rides the EXACT `real64` allreduce: a
      !! per-field POPCNT sum is bounded by (#elements x 64), which stays
      !! FAR below the double-mantissa bound 2**53 for any realistic grid
      !! (2**53/64 ~ 1.4e14 cells), so both the local->double cast and
      !! every partial MPI_SUM are exact — the invariant survives.
      integer(int64), intent(in) :: local_val
      integer(int64), intent(out) :: global_val

      type(comm_t) :: comm
      real(real64) :: acc_local, acc_global

      comm = comm_env_compute_comm()
      ! Single rank: the reduction is the identity, so return this rank's
      ! own contribution without entering a collective.  Not just an
      ! optimisation -- pic-mpi's serial backend (PIC_ENABLE_MPI=OFF)
      ! deliberately `error stop`s in `allreduce`, pushing the size()==1
      ! case onto the caller.  This IS that case.
      if (comm%size() == 1) then
         global_val = local_val
         return
      end if

      acc_local = real(local_val, real64)
      call allreduce(comm, acc_local, acc_global, op=MPI_SUM)
      global_val = int(acc_global, int64)

   end subroutine halo_allreduce_sum_i8