Octopus
states_elec_all_to_all_communications.F90
Go to the documentation of this file.
1!! Copyright (C) 2023 N. Tancogne-Dejean
2!!
3!! This program is free software; you can redistribute it and/or modify
4!! it under the terms of the GNU General Public License as published by
5!! the Free Software Foundation; either version 2, or (at your option)
6!! any later version.
7!!
8!! This program is distributed in the hope that it will be useful,
9!! but WITHOUT ANY WARRANTY; without even the implied warranty of
10!! MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
11!! GNU General Public License for more details.
12!!
13!! You should have received a copy of the GNU General Public License
14!! along with this program; if not, write to the Free Software
15!! Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA
16!! 02110-1301, USA.
17!!
18
19#include "global.h"
20
21
32 use accel_oct_m
33 use batch_oct_m
34 use debug_oct_m
35 use global_oct_m
36 use math_oct_m
37 use mesh_oct_m
39 use mpi_oct_m
45
46 implicit none
47
48 private
49
50 public :: &
52
54 private
55
56 integer :: task_from
57 integer :: task_to
58
59 integer :: nbatch_to_receive
60 integer :: nbatch_to_send
61 integer :: n_comms
62
63 integer :: nblock_to_receive
64 integer :: nblock_to_send
65
66 logical :: gpu_aware = .false.
67
68 type(MPI_Request), allocatable :: send_req(:)
69 contains
74 procedure :: alloc_receive_batch => states_elec_all_to_all_communications_alloc_receive_batch
75 procedure :: get_send_indices => states_elec_all_to_all_communications_get_send_indices
76 procedure :: get_receive_indices => states_elec_all_to_all_communications_get_receive_indices
77 procedure :: dpost_all_mpi_isend => dstates_elec_all_to_all_communications_post_all_mpi_isend
78 procedure :: zpost_all_mpi_isend => zstates_elec_all_to_all_communications_post_all_mpi_isend
81 procedure :: dmpi_recv_batch => dstates_elec_all_to_all_communications_mpi_recv_batch
82 procedure :: zmpi_recv_batch => zstates_elec_all_to_all_communications_mpi_recv_batch
83 procedure :: wait_all_isend => states_elec_all_to_all_communications_wait_all_isend
85
86contains
87
88 !------------------------------------------------------------
92 subroutine states_elec_all_to_all_communications_start(this, st, task_from, task_to, gpu_aware)
93 class(states_elec_all_to_all_communications_t), intent(inout) :: this
94 type(states_elec_t), intent(in) :: st
95 integer, intent(in) :: task_from, task_to
96 logical, optional, intent(in) :: gpu_aware
97
99
100 this%task_from = task_from
101 this%task_to = task_to
102
103 this%gpu_aware = optional_default(gpu_aware, .false.)
104
105 this%nbatch_to_receive = states_elec_all_to_all_communications_eval_nreceive(st, task_from, this%nblock_to_receive)
106 this%nbatch_to_send = states_elec_all_to_all_communications_eval_nsend(st, task_to, this%nblock_to_send)
107
108 !Number of communications
109 this%n_comms = max(this%nbatch_to_send, this%nbatch_to_receive)
110
113
114 !------------------------------------------------------------
116 integer function states_elec_all_to_all_communications_eval_nreceive(st, task_from, nblock_to_receive) result(nbatch_to_receive)
117 type(states_elec_t), intent(in) :: st
118 integer, intent(in) :: task_from
119 integer, intent(out) :: nblock_to_receive
120
121 integer :: st_start, st_end, kpt_start, kpt_end, ib
122
124
125 nbatch_to_receive = 0
126 nblock_to_receive = 0
127
128 !What we receive for the wfn
129 if (task_from > -1) then
130 st_start = st%st_kpt_task(task_from, 1)
131 st_end = st%st_kpt_task(task_from, 2)
132 kpt_start = st%st_kpt_task(task_from, 3)
133 kpt_end = st%st_kpt_task(task_from, 4)
134
135 nblock_to_receive = 0
136 do ib = 1, st%group%nblocks
137 if (st%group%block_range(ib, 1) >= st_start .and. st%group%block_range(ib, 2) <= st_end) then
138 nblock_to_receive = nblock_to_receive + 1
139 end if
140 end do
141 nbatch_to_receive = nblock_to_receive * (kpt_end-kpt_start+1)
142 end if
143
144 write(message(1), '(a,i5,a,i5,a,i5)') 'Debug: Task ', st%st_kpt_mpi_grp%rank, ' will receive ', &
145 nbatch_to_receive, ' batches from task ', task_from
146 call messages_info(1, all_nodes=.true., debug_only=.true.)
147
150
151 !------------------------------------------------------------
153 integer function states_elec_all_to_all_communications_eval_nsend(st, task_to, nblock_to_send) result(nbatch_to_send)
154 type(states_elec_t), intent(in) :: st
155 integer, intent(in) :: task_to
156 integer, intent(out) :: nblock_to_send
157
160 nbatch_to_send = 0
161 nblock_to_send = 0
162
163 if (task_to > -1) then
164 nblock_to_send = (st%group%block_end-st%group%block_start+1)
165 nbatch_to_send = nblock_to_send*(st%d%kpt%end-st%d%kpt%start+1)
166 end if
168 write(message(1), '(a,i5,a,i5,a,i5)') 'Debug: Task ', st%st_kpt_mpi_grp%rank, ' will send ', nbatch_to_send, &
169 ' batches to task ', task_to
170 call messages_info(1, all_nodes=.true., debug_only=.true.)
175 !------------------------------------------------------------
177 integer pure function states_elec_all_to_all_communications_get_ncom(this) result(n_comms)
179
180 n_comms = this%n_comms
182
183 !------------------------------------------------------------
185 integer pure function states_elec_all_to_all_communications_get_nreceive(this) result(nbatch_to_receive)
186 class(states_elec_all_to_all_communications_t), intent(in) :: this
188 nbatch_to_receive = this%nbatch_to_receive
190
191 !------------------------------------------------------------
193 integer pure function states_elec_all_to_all_communications_get_nsend(this) result(nbatch_to_send)
194 class(states_elec_all_to_all_communications_t), intent(in) :: this
195
196 nbatch_to_send = this%nbatch_to_send
198
199 !------------------------------------------------------------
201 subroutine states_elec_all_to_all_communications_alloc_receive_batch(this, st, icom, np, psib)
202 class(states_elec_all_to_all_communications_t), intent(in) :: this
203 type(states_elec_t), intent(in) :: st
204 integer, intent(in) :: icom
205 integer, intent(in) :: np
206 type(wfs_elec_t), intent(out) :: psib
207
208 integer :: block_id, ib, ik
209
212 ! Given the icom, returns the id of the block of state communicated
213 block_id = mod(icom-1, this%nblock_to_receive)+1
214 ik = int((icom-block_id)/this%nblock_to_receive) + st%st_kpt_task(this%task_from, 3)
215 ib = block_id - 1 + st%group%iblock(st%st_kpt_task(this%task_from, 1))
216
217 write(message(1), '(a,i5,a,i5,a,i5)') 'Debug: Task ', st%st_kpt_mpi_grp%rank, ' allocates memory for block ', &
218 ib, ' and k-point ', ik
219 call messages_info(1, all_nodes=.true., debug_only=.true.)
220
221 call states_elec_parallel_allocate_batch(st, psib, np, ib, ik, packed=.true.)
222
223 ! For CUDA-aware MPI, the batch must be on GPU to receive directly in the buffer.
224 if (this%gpu_aware) then
225 call psib%do_pack(batch_device_packed, copy = .false.)
226 end if
227
230
231 !------------------------------------------------------------
233 subroutine states_elec_all_to_all_communications_get_send_indices(this, st, icom, ib, ik)
234 class(states_elec_all_to_all_communications_t), intent(in) :: this
235 type(states_elec_t), intent(in) :: st
236 integer, intent(in) :: icom
237 integer, intent(out) :: ib
238 integer, intent(out) :: ik
239
241
242 ! Given the icom, returns the id of the block of state communicated
243 ib = mod(icom-1, this%nblock_to_send) + 1
244 ik = int((icom-ib)/this%nblock_to_send) + st%d%kpt%start
245 ib = ib - 1 + st%group%iblock(st%st_start)
246
247 write(message(1), '(a,i5,a,i5,a,i5)') 'Debug: Task ', st%st_kpt_mpi_grp%rank, ' will send the block ', &
248 ib, ' with k-point ', ik
249 call messages_info(1, all_nodes=.true., debug_only=.true.)
250
253
254 !------------------------------------------------------------
256 subroutine states_elec_all_to_all_communications_get_receive_indices(this, st, icom, ib, ik)
257 class(states_elec_all_to_all_communications_t), intent(in) :: this
258 type(states_elec_t), intent(in) :: st
259 integer, intent(in) :: icom
260 integer, intent(out) :: ib
261 integer, intent(out) :: ik
262
264
265 ! Given the icom, returns the id of the block of state communicated
266 ib = mod(icom-1, this%nblock_to_receive)+1
267 ik = int((icom-ib)/this%nblock_to_receive) + st%st_kpt_task(this%task_from, 3)
268 ib = ib - 1 + st%group%iblock(st%st_kpt_task(this%task_from, 1))
269
270 if (debug%info) then
271 write(message(1), '(a,i5,a,i5,a,i5)') 'Task ', st%st_kpt_mpi_grp%rank, ' will receive the block ', &
272 ib, ' with k-point ', ik
273 call messages_info(1, all_nodes=.true.)
274 end if
275
278
279 !------------------------------------------------------------
282 class(states_elec_all_to_all_communications_t), intent(inout) :: this
283 type(states_elec_t), intent(in) :: st
284
286 call profiling_in("ALL_TO_ALL_COMM")
287
288 if (allocated(this%send_req)) then
289
290 call st%st_kpt_mpi_grp%wait(this%nbatch_to_send, this%send_req)
291
292 safe_deallocate_a(this%send_req)
293
294 end if
295
296 call profiling_out("ALL_TO_ALL_COMM")
299
300
301
302#include "undef.F90"
303#include "real.F90"
304#include "states_elec_all_to_all_communications_inc.F90"
305
306#include "undef.F90"
307#include "complex.F90"
308#include "states_elec_all_to_all_communications_inc.F90"
309#include "undef.F90"
310
312
313!! Local Variables:
314!! mode: f90
315!! coding: utf-8
316!! End:
This module implements batches of mesh functions.
Definition: batch.F90:135
This module is intended to contain "only mathematical" functions and procedures.
Definition: math.F90:117
This module defines the meshes, which are used in Octopus.
Definition: mesh.F90:120
character(len=256), dimension(max_lines), public message
to be output by fatal, warning
Definition: messages.F90:162
subroutine, public messages_info(no_lines, iunit, debug_only, stress, all_nodes, namespace)
Definition: messages.F90:594
This module provides routines for communicating all batches in a ring-pattern scheme.
subroutine states_elec_all_to_all_communications_get_receive_indices(this, st, icom, ib, ik)
Given the icom step, returns the block and k-point indices to be received.
integer function states_elec_all_to_all_communications_eval_nsend(st, task_to, nblock_to_send)
How many batches we will send from task_send.
subroutine states_elec_all_to_all_communications_wait_all_isend(this, st)
Do a MPI waitall for the isend requests.
subroutine zstates_elec_all_to_all_communications_post_all_mpi_isend(this, st, np, node_to)
Post all isend commands for all batches of a given task.
subroutine dstates_elec_all_to_all_communications_mpi_recv_batch(this, st, np, node_fr, icom, psib_receiv)
Allocate a batch and perform the MPI_Recv. On exit, the batch contains the received information.
integer pure function states_elec_all_to_all_communications_get_nreceive(this)
Returns the number of receiv calls.
subroutine states_elec_all_to_all_communications_alloc_receive_batch(this, st, icom, np, psib)
Given the icom step, allocate the receiv buffer (wfs_elec_t)
subroutine dstates_elec_all_to_all_communications_isend_batch(this, st, np, ib_send, ik_send, node_to, icom, send_req)
Post a single MPI isend for the batch (ib_send, ik_send) to node_to.
integer pure function states_elec_all_to_all_communications_get_ncom(this)
Returns the number of communications.
subroutine dstates_elec_all_to_all_communications_post_all_mpi_isend(this, st, np, node_to)
Post all isend commands for all batches of a given task.
subroutine states_elec_all_to_all_communications_get_send_indices(this, st, icom, ib, ik)
Given the icom step, returns the block and k-point indices to be sent.
integer function states_elec_all_to_all_communications_eval_nreceive(st, task_from, nblock_to_receive)
How many batches we will receive from task_from.
subroutine states_elec_all_to_all_communications_start(this, st, task_from, task_to, gpu_aware)
Given a task to send to, and a task to receive from, initializes a states_elec_all_to_all_communicati...
integer pure function states_elec_all_to_all_communications_get_nsend(this)
Returns the number send calls.
subroutine zstates_elec_all_to_all_communications_mpi_recv_batch(this, st, np, node_fr, icom, psib_receiv)
Allocate a batch and perform the MPI_Recv. On exit, the batch contains the received information.
subroutine zstates_elec_all_to_all_communications_isend_batch(this, st, np, ib_send, ik_send, node_to, icom, send_req)
Post a single MPI isend for the batch (ib_send, ik_send) to node_to.
This module provides routines for communicating states when using states parallelization.
The states_elec_t class contains all electronic wave functions.
int true(void)