report_throughput Subroutine

private subroutine report_throughput(global_cells, n_steps, elapsed, compute_size, compute_rank, total_mcells)

Log per-GPU and total horizontal throughput on the root compute rank.

global_cells is the total horizontal cells across all ranks – the caller does any reduction needed to obtain it (the ocean path already knows the global grid). Per-GPU is the balanced-partition average (total / compute_size). Throughput is horizontal cells x steps, NOT x nz (the codebase convention). Optionally returns the total so a caller can thread it into a later summary.

Arguments

Type IntentOptional Attributes Name
real(kind=wp), intent(in) :: global_cells
integer, intent(in) :: n_steps
real(kind=wp), intent(in) :: elapsed
integer, intent(in) :: compute_size
integer, intent(in) :: compute_rank
real(kind=wp), intent(out), optional :: total_mcells

Calls

proc~~report_throughput~~CallsGraph proc~report_throughput report_throughput info info proc~report_throughput->info to_string to_string proc~report_throughput->to_string

Called by

proc~~report_throughput~~CalledByGraph proc~report_throughput report_throughput proc~driver_run_ocean driver_run_ocean proc~driver_run_ocean->proc~report_throughput proc~driver_run driver_run proc~driver_run->proc~driver_run_ocean

Variables

Type Visibility Attributes Name Initial
real(kind=wp), private :: per_gpu
real(kind=wp), private :: total

Source Code

   subroutine report_throughput(global_cells, n_steps, elapsed, compute_size, &
                                compute_rank, total_mcells)
      !! Log per-GPU and total horizontal throughput on the root compute rank.
      !!
      !! `global_cells` is the total horizontal cells across all ranks -- the
      !! caller does any reduction needed to obtain it (the ocean path
      !! already knows the global grid). Per-GPU is the balanced-partition
      !! average
      !! (`total / compute_size`). Throughput is horizontal cells x steps, NOT
      !! x nz (the codebase convention). Optionally returns the total so a
      !! caller can thread it into a later summary.
      real(wp), intent(in) :: global_cells
      integer, intent(in) :: n_steps
      real(wp), intent(in) :: elapsed
      integer, intent(in) :: compute_size
      integer, intent(in) :: compute_rank
      real(wp), intent(out), optional :: total_mcells

      real(wp) :: per_gpu, total

      total = 0.0_wp
      per_gpu = 0.0_wp
      if (elapsed > 0.0_wp .and. compute_size > 0) then
         total = global_cells*real(n_steps, wp)/elapsed/1.0e6_wp
         per_gpu = total/real(compute_size, wp)
      end if
      if (present(total_mcells)) total_mcells = total
      if (compute_rank == 0) then
         call logger%info("  throughput: "//to_string(per_gpu)//" Mcells/s per GPU, "// &
                          to_string(total)//" Mcells/s total ("// &
                          to_string(compute_size)//" rank(s))")
      end if
   end subroutine report_throughput