Per-rank one-shot diagnostic: world rank -> requested device num / actual device num reported by the OpenACC runtime. Lets us confirm that mpirun is binding each rank to its own GPU rather than serialising N ranks on device 0.
| Type | Intent | Optional | Attributes | Name | ||
|---|---|---|---|---|---|---|
| integer, | intent(in) | :: | requested_gpu |
| Type | Visibility | Attributes | Name | Initial | |||
|---|---|---|---|---|---|---|---|
| integer, | private | :: | actual_dev |
subroutine print_gpu_binding(requested_gpu) !! Per-rank one-shot diagnostic: world rank -> requested device num / !! actual device num reported by the OpenACC runtime. Lets us !! confirm that mpirun is binding each rank to its own GPU rather !! than serialising N ranks on device 0. integer, intent(in) :: requested_gpu integer :: actual_dev #ifdef RDB_HAS_OPENACC_RUNTIME ! Works on either backend: stdpar+OpenACC and -mp=gpu / -h omp ! both keep the OpenACC runtime live, and acc_get_device_num ! reports the same device that stdpar/-acc dispatches against. actual_dev = acc_get_device_num(acc_get_device_type()) #else actual_dev = -1 #endif write (output_unit, "(A,I0,A,I0,A,I0,A,I0,A,I0)") & "[gpu-bind] world_rank=", cached_rank, & "/", cached_size, & " compute_rank=", cached_compute_rank, & " requested_dev=", requested_gpu, & " actual_dev=", actual_dev if (cached_rank == 0) then #ifdef RDB_CUDA_AWARE_MPI write (output_unit, "(A)") "[gpu-bind] halo path: GPU-direct (CUDA-aware MPI)" #else write (output_unit, "(A)") "[gpu-bind] halo path: HOST-STAGED (rebuild with -DRDB_CUDA_AWARE_MPI=ON for GPU-direct)" #endif end if flush (output_unit) end subroutine print_gpu_binding