59 integer :: nbatch_to_receive
60 integer :: nbatch_to_send
63 integer :: nblock_to_receive
64 integer :: nblock_to_send
66 logical :: gpu_aware = .false.
68 type(MPI_Request),
allocatable :: send_req(:)
97 class(states_elec_all_to_all_communications_t),
intent(inout) :: this
98 type(states_elec_t),
intent(in) :: st
99 integer,
intent(in) :: task_from, task_to
100 logical,
optional,
intent(in) :: gpu_aware
104 this%task_from = task_from
105 this%task_to = task_to
113 this%n_comms = max(this%nbatch_to_send, this%nbatch_to_receive)
121 type(states_elec_t),
intent(in) :: st
122 integer,
intent(in) :: task_from
123 integer,
intent(out) :: nblock_to_receive
125 integer :: st_start, st_end, kpt_start, kpt_end, ib
129 nbatch_to_receive = 0
130 nblock_to_receive = 0
133 if (task_from > -1)
then
134 st_start = st%st_kpt_task(task_from, 1)
135 st_end = st%st_kpt_task(task_from, 2)
136 kpt_start = st%st_kpt_task(task_from, 3)
137 kpt_end = st%st_kpt_task(task_from, 4)
139 nblock_to_receive = 0
140 do ib = 1, st%group%nblocks
141 if (st%group%block_range(ib, 1) >= st_start .and. st%group%block_range(ib, 2) <= st_end)
then
142 nblock_to_receive = nblock_to_receive + 1
145 nbatch_to_receive = nblock_to_receive * (kpt_end-kpt_start+1)
148 write(
message(1),
'(a,i5,a,i5,a,i5)')
'Debug: Task ', st%st_kpt_mpi_grp%rank,
' will receive ', &
149 nbatch_to_receive,
' batches from task ', task_from
159 integer,
intent(in) :: task_to
160 integer,
intent(out) :: nblock_to_send
167 if (task_to > -1)
then
168 nblock_to_send = (st%group%block_end-st%group%block_start+1)
169 nbatch_to_send = nblock_to_send*(st%d%kpt%end-st%d%kpt%start+1)
172 write(
message(1),
'(a,i5,a,i5,a,i5)')
'Debug: Task ', st%st_kpt_mpi_grp%rank,
' will send ', nbatch_to_send, &
173 ' batches to task ', task_to
184 n_comms = this%n_comms
192 nbatch_to_receive = this%nbatch_to_receive
197 integer pure function states_elec_all_to_all_communications_get_nsend(this) result(nbatch_to_send)
200 nbatch_to_send = this%nbatch_to_send
207 type(states_elec_t),
intent(in) :: st
208 integer,
intent(in) :: icom
209 integer,
intent(in) :: np
210 type(wfs_elec_t),
intent(out) :: psib
212 integer :: block_id, ib, ik
217 block_id = mod(icom-1, this%nblock_to_receive)+1
218 ik = int((icom-block_id)/this%nblock_to_receive) + st%st_kpt_task(this%task_from, 3)
219 ib = block_id - 1 + st%group%iblock(st%st_kpt_task(this%task_from, 1))
221 write(message(1),
'(a,i5,a,i5,a,i5)')
'Debug: Task ', st%st_kpt_mpi_grp%rank,
' allocates memory for block ', &
222 ib,
' and k-point ', ik
223 call messages_info(1, all_nodes=.
true., debug_only=.
true.)
225 call states_elec_parallel_allocate_batch(st, psib, np, ib, ik, packed=.
true.)
228 if (this%gpu_aware)
then
229 call psib%do_pack(batch_device_packed, copy = .false.)
239 type(states_elec_t),
intent(in) :: st
240 integer,
intent(in) :: icom
241 integer,
intent(out) :: ib
242 integer,
intent(out) :: ik
247 ib = mod(icom-1, this%nblock_to_send) + 1
248 ik = int((icom-ib)/this%nblock_to_send) + st%d%kpt%start
249 ib = ib - 1 + st%group%iblock(st%st_start)
251 write(message(1),
'(a,i5,a,i5,a,i5)')
'Debug: Task ', st%st_kpt_mpi_grp%rank,
' will send the block ', &
252 ib,
' with k-point ', ik
253 call messages_info(1, all_nodes=.
true., debug_only=.
true.)
262 type(states_elec_t),
intent(in) :: st
263 integer,
intent(in) :: icom
264 integer,
intent(out) :: ib
265 integer,
intent(out) :: ik
270 ib = mod(icom-1, this%nblock_to_receive)+1
271 ik = int((icom-ib)/this%nblock_to_receive) + st%st_kpt_task(this%task_from, 3)
272 ib = ib - 1 + st%group%iblock(st%st_kpt_task(this%task_from, 1))
275 write(message(1),
'(a,i5,a,i5,a,i5)')
'Task ', st%st_kpt_mpi_grp%rank,
' will receive the block ', &
276 ib,
' with k-point ', ik
277 call messages_info(1, all_nodes=.
true.)
287 type(states_elec_t),
intent(in) :: st
290 call profiling_in(
"ALL_TO_ALL_COMM")
292 if (
allocated(this%send_req))
then
294 call st%st_kpt_mpi_grp%wait(this%nbatch_to_send, this%send_req)
296 safe_deallocate_a(this%send_req)
300 call profiling_out(
"ALL_TO_ALL_COMM")
308#include "states_elec_all_to_all_communications_inc.F90"
311#include "complex.F90"
312#include "states_elec_all_to_all_communications_inc.F90"
This module implements batches of mesh functions.
This module is intended to contain "only mathematical" functions and procedures.
This module defines the meshes, which are used in Octopus.
character(len=256), dimension(max_lines), public message
to be output by fatal, warning
subroutine, public messages_info(no_lines, iunit, debug_only, stress, all_nodes, namespace)
This module provides routines for communicating all batches in a ring-pattern scheme.
subroutine states_elec_all_to_all_communications_get_receive_indices(this, st, icom, ib, ik)
Given the icom step, returns the block and k-point indices to be received.
integer function states_elec_all_to_all_communications_eval_nsend(st, task_to, nblock_to_send)
How many batches we will send from task_send.
subroutine zstates_elec_all_to_all_communications_isend_batch_to(this, st, np, psib, node_to, icom, send_req)
Post a single MPI isend for an explicit batch to node_to.
subroutine zstates_elec_all_to_all_communications_mpi_irecv_batch(this, st, np, node_fr, icom, psib_receiv, recv_req)
Allocate a batch and post the MPI_Irecv for it. The caller must wait on recv_req before using the bat...
subroutine states_elec_all_to_all_communications_wait_all_isend(this, st)
Do a MPI waitall for the isend requests.
subroutine zstates_elec_all_to_all_communications_post_all_mpi_isend(this, st, np, node_to)
Post all isend commands for all batches of a given task.
integer pure function states_elec_all_to_all_communications_get_nreceive(this)
Returns the number of receiv calls.
subroutine states_elec_all_to_all_communications_alloc_receive_batch(this, st, icom, np, psib)
Given the icom step, allocate the receiv buffer (wfs_elec_t)
subroutine dstates_elec_all_to_all_communications_isend_batch(this, st, np, ib_send, ik_send, node_to, icom, send_req)
Post a single MPI isend for the batch (ib_send, ik_send) to node_to.
integer pure function states_elec_all_to_all_communications_get_ncom(this)
Returns the number of communications.
subroutine dstates_elec_all_to_all_communications_post_all_mpi_isend(this, st, np, node_to)
Post all isend commands for all batches of a given task.
subroutine states_elec_all_to_all_communications_get_send_indices(this, st, icom, ib, ik)
Given the icom step, returns the block and k-point indices to be sent.
integer function states_elec_all_to_all_communications_eval_nreceive(st, task_from, nblock_to_receive)
How many batches we will receive from task_from.
subroutine states_elec_all_to_all_communications_start(this, st, task_from, task_to, gpu_aware)
Given a task to send to, and a task to receive from, initializes a states_elec_all_to_all_communicati...
subroutine dstates_elec_all_to_all_communications_mpi_irecv_batch(this, st, np, node_fr, icom, psib_receiv, recv_req)
Allocate a batch and post the MPI_Irecv for it. The caller must wait on recv_req before using the bat...
subroutine zstates_elec_all_to_all_communications_recv_batch_from(this, st, np, psib, node_fr, icom)
Blocking MPI recv into an already-allocated batch from node_fr.
subroutine dstates_elec_all_to_all_communications_isend_batch_to(this, st, np, psib, node_to, icom, send_req)
Post a single MPI isend for an explicit batch to node_to.
integer pure function states_elec_all_to_all_communications_get_nsend(this)
Returns the number send calls.
subroutine dstates_elec_all_to_all_communications_recv_batch_from(this, st, np, psib, node_fr, icom)
Blocking MPI recv into an already-allocated batch from node_fr.
subroutine zstates_elec_all_to_all_communications_isend_batch(this, st, np, ib_send, ik_send, node_to, icom, send_req)
Post a single MPI isend for the batch (ib_send, ik_send) to node_to.
This module provides routines for communicating states when using states parallelization.
The states_elec_t class contains all electronic wave functions.