Log per-GPU and total horizontal throughput on the root compute rank.
global_cells is the total horizontal cells across all ranks – the
caller does any reduction needed to obtain it (the ocean path
already knows the global grid). Per-GPU is the balanced-partition
average
(total / compute_size). Throughput is horizontal cells x steps, NOT
x nz (the codebase convention). Optionally returns the total so a
caller can thread it into a later summary.
| Type | Intent | Optional | Attributes | Name | ||
|---|---|---|---|---|---|---|
| real(kind=wp), | intent(in) | :: | global_cells | |||
| integer, | intent(in) | :: | n_steps | |||
| real(kind=wp), | intent(in) | :: | elapsed | |||
| integer, | intent(in) | :: | compute_size | |||
| integer, | intent(in) | :: | compute_rank | |||
| real(kind=wp), | intent(out), | optional | :: | total_mcells |
| Type | Visibility | Attributes | Name | Initial | |||
|---|---|---|---|---|---|---|---|
| real(kind=wp), | private | :: | per_gpu | ||||
| real(kind=wp), | private | :: | total |
subroutine report_throughput(global_cells, n_steps, elapsed, compute_size, & compute_rank, total_mcells) !! Log per-GPU and total horizontal throughput on the root compute rank. !! !! `global_cells` is the total horizontal cells across all ranks -- the !! caller does any reduction needed to obtain it (the ocean path !! already knows the global grid). Per-GPU is the balanced-partition !! average !! (`total / compute_size`). Throughput is horizontal cells x steps, NOT !! x nz (the codebase convention). Optionally returns the total so a !! caller can thread it into a later summary. real(wp), intent(in) :: global_cells integer, intent(in) :: n_steps real(wp), intent(in) :: elapsed integer, intent(in) :: compute_size integer, intent(in) :: compute_rank real(wp), intent(out), optional :: total_mcells real(wp) :: per_gpu, total total = 0.0_wp per_gpu = 0.0_wp if (elapsed > 0.0_wp .and. compute_size > 0) then total = global_cells*real(n_steps, wp)/elapsed/1.0e6_wp per_gpu = total/real(compute_size, wp) end if if (present(total_mcells)) total_mcells = total if (compute_rank == 0) then call logger%info(" throughput: "//to_string(per_gpu)//" Mcells/s per GPU, "// & to_string(total)//" Mcells/s total ("// & to_string(compute_size)//" rank(s))") end if end subroutine report_throughput