28 USE cp_dbcsr_api,
ONLY: dbcsr_convert_sizes_to_offsets, &
72#include "./base/base_uses.f90"
77 LOGICAL,
PRIVATE,
PARAMETER :: debug_this_module = .false.
81 CHARACTER(len=*),
PARAMETER,
PRIVATE :: moduleN =
'task_list_methods'
117 reorder_rs_grid_ranks, skip_load_balance_distributed, &
118 pw_env_external, sab_orb_external, ext_kpoints)
122 CHARACTER(LEN=*),
INTENT(IN) :: basis_type
123 LOGICAL,
INTENT(IN) :: reorder_rs_grid_ranks, &
124 skip_load_balance_distributed
125 TYPE(
pw_env_type),
OPTIONAL,
POINTER :: pw_env_external
127 OPTIONAL,
POINTER :: sab_orb_external
128 TYPE(
kpoint_type),
OPTIONAL,
POINTER :: ext_kpoints
130 CHARACTER(LEN=*),
PARAMETER :: routinen =
'generate_qs_task_list'
131 INTEGER,
PARAMETER :: max_tasks = 2000
133 INTEGER :: cindex, curr_tasks, handle, i, iatom, iatom_old, igrid_level, igrid_level_old, &
134 ikind, ilevel, img, img_old, ipair, ipgf, iset, itask, jatom, jatom_old, jkind, jpgf, &
135 jset, maxpgf, maxset, natoms, nimages, nkind, nseta, nsetb, slot
136 INTEGER,
ALLOCATABLE,
DIMENSION(:, :, :) :: blocks
137 INTEGER,
DIMENSION(3) :: cellind
138 INTEGER,
DIMENSION(:),
POINTER :: la_max, la_min, lb_max, lb_min, npgfa, &
140 INTEGER,
DIMENSION(:, :, :),
POINTER :: cell_to_index
142 REAL(kind=
dp) :: kind_radius_a, kind_radius_b
143 REAL(kind=
dp),
DIMENSION(3) :: ra, rab
144 REAL(kind=
dp),
DIMENSION(:),
POINTER :: set_radius_a, set_radius_b
145 REAL(kind=
dp),
DIMENSION(:, :),
POINTER :: rpgfa, rpgfb, zeta, zetb
157 TYPE(
qs_kind_type),
DIMENSION(:),
POINTER :: qs_kind_set
162 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
164 CALL timeset(routinen, handle)
167 qs_kind_set=qs_kind_set, &
169 particle_set=particle_set, &
170 dft_control=dft_control)
173 IF (
PRESENT(sab_orb_external)) sab_orb => sab_orb_external
176 IF (
PRESENT(pw_env_external)) pw_env => pw_env_external
177 CALL pw_env_get(pw_env, rs_descs=rs_descs, rs_grids=rs_grids)
180 gridlevel_info => pw_env%gridlevel_info
181 cube_info => pw_env%cube_info
184 nkind =
SIZE(qs_kind_set)
185 natoms =
SIZE(particle_set)
189 qs_kind => qs_kind_set(ikind)
191 basis_set=orb_basis_set, basis_type=basis_type)
193 IF (.NOT.
ASSOCIATED(orb_basis_set)) cycle
196 maxset = max(nseta, maxset)
197 maxpgf = max(maxval(npgfa), maxpgf)
201 nimages = dft_control%nimages
203 NULLIFY (cell_to_index)
205 IF (
PRESENT(ext_kpoints))
THEN
206 IF (
ASSOCIATED(ext_kpoints))
THEN
209 nimages =
SIZE(ext_kpoints%index_to_cell, 2)
212 IF (.NOT. dokp .AND. nimages > 1)
THEN
214 CALL get_ks_env(ks_env=ks_env, kpoints=kpoints)
219 IF (
ASSOCIATED(task_list%atom_pair_send))
DEALLOCATE (task_list%atom_pair_send)
220 IF (
ASSOCIATED(task_list%atom_pair_recv))
DEALLOCATE (task_list%atom_pair_recv)
223 IF (.NOT.
ASSOCIATED(task_list%tasks))
THEN
227 curr_tasks =
SIZE(task_list%tasks)
229 ALLOCATE (basis_set_list(nkind))
231 qs_kind => qs_kind_set(ikind)
232 CALL get_qs_kind(qs_kind=qs_kind, basis_set=basis_set_a, &
233 basis_type=basis_type)
234 IF (
ASSOCIATED(basis_set_a))
THEN
235 basis_set_list(ikind)%gto_basis_set => basis_set_a
237 NULLIFY (basis_set_list(ikind)%gto_basis_set)
249 DO slot = 1, sab_orb(1)%nl_size
250 ikind = sab_orb(1)%nlist_task(slot)%ikind
251 jkind = sab_orb(1)%nlist_task(slot)%jkind
252 iatom = sab_orb(1)%nlist_task(slot)%iatom
253 jatom = sab_orb(1)%nlist_task(slot)%jatom
254 rab(1:3) = sab_orb(1)%nlist_task(slot)%r(1:3)
255 cellind(1:3) = sab_orb(1)%nlist_task(slot)%cell(1:3)
257 basis_set_a => basis_set_list(ikind)%gto_basis_set
258 IF (.NOT.
ASSOCIATED(basis_set_a)) cycle
259 basis_set_b => basis_set_list(jkind)%gto_basis_set
260 IF (.NOT.
ASSOCIATED(basis_set_b)) cycle
261 ra(:) =
pbc(particle_set(iatom)%r, cell)
263 la_max => basis_set_a%lmax
264 la_min => basis_set_a%lmin
265 npgfa => basis_set_a%npgf
266 nseta = basis_set_a%nset
267 rpgfa => basis_set_a%pgf_radius
268 set_radius_a => basis_set_a%set_radius
269 kind_radius_a = basis_set_a%kind_radius
270 zeta => basis_set_a%zet
272 lb_max => basis_set_b%lmax
273 lb_min => basis_set_b%lmin
274 npgfb => basis_set_b%npgf
275 nsetb = basis_set_b%nset
276 rpgfb => basis_set_b%pgf_radius
277 set_radius_b => basis_set_b%set_radius
278 kind_radius_b = basis_set_b%kind_radius
279 zetb => basis_set_b%zet
282 cindex = cell_to_index(cellind(1), cellind(2), cellind(3))
288 rs_descs, dft_control, cube_info, gridlevel_info, cindex, &
289 iatom, jatom, rpgfa, rpgfb, zeta, zetb, kind_radius_b, &
290 set_radius_a, set_radius_b, ra, rab, &
291 la_max, la_min, lb_max, lb_min, npgfa, npgfb, nseta, nsetb)
298 rs_descs=rs_descs, ntasks=task_list%ntasks, natoms=natoms, &
299 tasks=task_list%tasks, atom_pair_send=task_list%atom_pair_send, &
300 atom_pair_recv=task_list%atom_pair_recv, symmetric=.true., &
301 reorder_rs_grid_ranks=reorder_rs_grid_ranks, &
302 skip_load_balance_distributed=skip_load_balance_distributed)
305 ALLOCATE (nsgf(natoms))
306 CALL get_particle_set(particle_set, qs_kind_set, basis=basis_set_list, nsgf=nsgf)
307 IF (
ASSOCIATED(task_list%atom_pair_send))
THEN
309 CALL rs_calc_offsets(pairs=task_list%atom_pair_send, &
311 group_size=rs_descs(1)%rs_desc%group_size, &
312 pair_offsets=task_list%pair_offsets_send, &
313 rank_offsets=task_list%rank_offsets_send, &
314 rank_sizes=task_list%rank_sizes_send, &
315 buffer_size=task_list%buffer_size_send)
317 CALL rs_calc_offsets(pairs=task_list%atom_pair_recv, &
319 group_size=rs_descs(1)%rs_desc%group_size, &
320 pair_offsets=task_list%pair_offsets_recv, &
321 rank_offsets=task_list%rank_offsets_recv, &
322 rank_sizes=task_list%rank_sizes_recv, &
323 buffer_size=task_list%buffer_size_recv)
324 DEALLOCATE (basis_set_list, nsgf)
327 IF (reorder_rs_grid_ranks)
THEN
328 DO i = 1, gridlevel_info%ngrid_levels
329 IF (rs_descs(i)%rs_desc%distributed)
THEN
336 CALL create_grid_task_list(task_list=task_list, &
337 qs_kind_set=qs_kind_set, &
338 particle_set=particle_set, &
340 basis_type=basis_type, &
346 IF (
ASSOCIATED(task_list%taskstart))
THEN
347 DEALLOCATE (task_list%taskstart)
349 IF (
ASSOCIATED(task_list%taskstop))
THEN
350 DEALLOCATE (task_list%taskstop)
352 IF (
ASSOCIATED(task_list%npairs))
THEN
353 DEALLOCATE (task_list%npairs)
358 ALLOCATE (task_list%npairs(
SIZE(rs_descs)))
360 iatom_old = -1; jatom_old = -1; igrid_level_old = -1; img_old = -1
364 DO i = 1, task_list%ntasks
365 igrid_level = task_list%tasks(i)%grid_level
366 img = task_list%tasks(i)%image
367 iatom = task_list%tasks(i)%iatom
368 jatom = task_list%tasks(i)%jatom
369 iset = task_list%tasks(i)%iset
370 jset = task_list%tasks(i)%jset
371 ipgf = task_list%tasks(i)%ipgf
372 jpgf = task_list%tasks(i)%jpgf
373 IF (igrid_level /= igrid_level_old)
THEN
374 IF (igrid_level_old /= -1)
THEN
375 task_list%npairs(igrid_level_old) = ipair
378 igrid_level_old = igrid_level
382 ELSE IF (iatom /= iatom_old .OR. jatom /= jatom_old .OR. img /= img_old)
THEN
390 IF (task_list%ntasks /= 0)
THEN
391 task_list%npairs(igrid_level) = ipair
398 ALLOCATE (task_list%taskstart(maxval(task_list%npairs),
SIZE(rs_descs)))
399 ALLOCATE (task_list%taskstop(maxval(task_list%npairs),
SIZE(rs_descs)))
401 iatom_old = -1; jatom_old = -1; igrid_level_old = -1; img_old = -1
403 task_list%taskstart = 0
404 task_list%taskstop = 0
406 DO i = 1, task_list%ntasks
407 igrid_level = task_list%tasks(i)%grid_level
408 img = task_list%tasks(i)%image
409 iatom = task_list%tasks(i)%iatom
410 jatom = task_list%tasks(i)%jatom
411 iset = task_list%tasks(i)%iset
412 jset = task_list%tasks(i)%jset
413 ipgf = task_list%tasks(i)%ipgf
414 jpgf = task_list%tasks(i)%jpgf
415 IF (igrid_level /= igrid_level_old)
THEN
416 IF (igrid_level_old /= -1)
THEN
417 task_list%taskstop(ipair, igrid_level_old) = i - 1
420 task_list%taskstart(ipair, igrid_level) = i
421 igrid_level_old = igrid_level
425 ELSE IF (iatom /= iatom_old .OR. jatom /= jatom_old .OR. img /= img_old)
THEN
427 task_list%taskstart(ipair, igrid_level) = i
428 task_list%taskstop(ipair - 1, igrid_level) = i - 1
435 IF (task_list%ntasks /= 0)
THEN
436 task_list%taskstop(ipair, igrid_level) = task_list%ntasks
440 IF (debug_this_module)
THEN
441 tasks => task_list%tasks
443 WRITE (6, *)
"Total number of tasks ", task_list%ntasks
444 DO igrid_level = 1, gridlevel_info%ngrid_levels
445 WRITE (6, *)
"Total number of pairs(grid_level) ", &
446 igrid_level, task_list%npairs(igrid_level)
450 DO igrid_level = 1, gridlevel_info%ngrid_levels
452 ALLOCATE (blocks(natoms, natoms, nimages))
454 DO ipair = 1, task_list%npairs(igrid_level)
455 itask = task_list%taskstart(ipair, igrid_level)
456 ilevel = task_list%tasks(itask)%grid_level
457 img = task_list%tasks(itask)%image
458 iatom = task_list%tasks(itask)%iatom
459 jatom = task_list%tasks(itask)%jatom
460 iset = task_list%tasks(itask)%iset
461 jset = task_list%tasks(itask)%jset
462 ipgf = task_list%tasks(itask)%ipgf
463 jpgf = task_list%tasks(itask)%jpgf
464 IF (blocks(iatom, jatom, img) == -1 .AND. blocks(jatom, iatom, img) == -1)
THEN
465 blocks(iatom, jatom, img) = 1
466 blocks(jatom, iatom, img) = 1
468 WRITE (6, *)
"TASK LIST CONFLICT IN PAIR ", ipair
469 WRITE (6, *)
"Reuse of iatom, jatom, image ", iatom, jatom, img
475 DO itask = task_list%taskstart(ipair, igrid_level), task_list%taskstop(ipair, igrid_level)
476 ilevel = task_list%tasks(itask)%grid_level
477 img = task_list%tasks(itask)%image
478 iatom = task_list%tasks(itask)%iatom
479 jatom = task_list%tasks(itask)%jatom
480 iset = task_list%tasks(itask)%iset
481 jset = task_list%tasks(itask)%jset
482 ipgf = task_list%tasks(itask)%ipgf
483 jpgf = task_list%tasks(itask)%jpgf
484 IF (iatom /= iatom_old .OR. jatom /= jatom_old .OR. img /= img_old)
THEN
485 WRITE (6, *)
"TASK LIST CONFLICT IN TASK ", itask
486 WRITE (6, *)
"Inconsistent iatom, jatom, image ", iatom, jatom, img
487 WRITE (6, *)
"Should be iatom, jatom, image ", iatom_old, jatom_old, img_old
498 CALL timestop(handle)
506 SUBROUTINE create_grid_task_list(task_list, qs_kind_set, particle_set, cell, basis_type, rs_grids)
508 TYPE(
qs_kind_type),
DIMENSION(:),
POINTER :: qs_kind_set
511 CHARACTER(LEN=*) :: basis_type
515 INTEGER :: nset, natoms, nkinds, ntasks, &
516 ikind, iatom, itask, nsgf
517 INTEGER,
DIMENSION(:),
ALLOCATABLE :: atom_kinds, level_list, iatom_list, jatom_list, &
518 iset_list, jset_list, ipgf_list, jpgf_list, &
519 border_mask_list, block_num_list
520 REAL(kind=
dp),
DIMENSION(:),
ALLOCATABLE :: radius_list
521 REAL(kind=
dp),
DIMENSION(:, :),
ALLOCATABLE :: rab_list, atom_positions
522 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
523 INTEGER,
DIMENSION(:, :),
POINTER :: first_sgf
524 REAL(kind=
dp),
DIMENSION(:, :),
POINTER :: sphi, zet
525 INTEGER,
DIMENSION(:),
POINTER :: lmax, lmin, npgf, nsgf_set
527 nkinds =
SIZE(qs_kind_set)
528 natoms =
SIZE(particle_set)
529 ntasks = task_list%ntasks
530 tasks => task_list%tasks
532 IF (.NOT.
ASSOCIATED(task_list%grid_basis_sets))
THEN
534 ALLOCATE (task_list%grid_basis_sets(nkinds))
536 CALL get_qs_kind(qs_kind_set(ikind), basis_type=basis_type, basis_set=orb_basis_set)
542 first_sgf=first_sgf, &
549 maxco=
SIZE(sphi, 1), &
550 maxpgf=
SIZE(zet, 1), &
555 first_sgf=first_sgf, &
558 basis_set=task_list%grid_basis_sets(ikind))
563 ALLOCATE (atom_kinds(natoms), atom_positions(3, natoms))
565 atom_kinds(iatom) = particle_set(iatom)%atomic_kind%kind_number
566 atom_positions(:, iatom) =
pbc(particle_set(iatom)%r, cell)
569 ALLOCATE (level_list(ntasks), iatom_list(ntasks), jatom_list(ntasks))
570 ALLOCATE (iset_list(ntasks), jset_list(ntasks), ipgf_list(ntasks), jpgf_list(ntasks))
571 ALLOCATE (border_mask_list(ntasks), block_num_list(ntasks))
572 ALLOCATE (radius_list(ntasks), rab_list(3, ntasks))
575 level_list(itask) = tasks(itask)%grid_level
576 iatom_list(itask) = tasks(itask)%iatom
577 jatom_list(itask) = tasks(itask)%jatom
578 iset_list(itask) = tasks(itask)%iset
579 jset_list(itask) = tasks(itask)%jset
580 ipgf_list(itask) = tasks(itask)%ipgf
581 jpgf_list(itask) = tasks(itask)%jpgf
582 IF (tasks(itask)%dist_type == 2)
THEN
583 border_mask_list(itask) = iand(63, not(tasks(itask)%subpatch_pattern))
585 border_mask_list(itask) = 0
587 block_num_list(itask) = tasks(itask)%pair_index
588 radius_list(itask) = tasks(itask)%radius
589 rab_list(:, itask) = tasks(itask)%rab(:)
595 nblocks=
SIZE(task_list%pair_offsets_recv), &
596 block_offsets=task_list%pair_offsets_recv, &
597 atom_positions=atom_positions, &
598 atom_kinds=atom_kinds, &
599 basis_sets=task_list%grid_basis_sets, &
600 level_list=level_list, &
601 iatom_list=iatom_list, &
602 jatom_list=jatom_list, &
603 iset_list=iset_list, &
604 jset_list=jset_list, &
605 ipgf_list=ipgf_list, &
606 jpgf_list=jpgf_list, &
607 border_mask_list=border_mask_list, &
608 block_num_list=block_num_list, &
609 radius_list=radius_list, &
612 task_list=task_list%grid_task_list)
617 END SUBROUTINE create_grid_task_list
652 cube_info, gridlevel_info, cindex, &
653 iatom, jatom, rpgfa, rpgfb, zeta, zetb, kind_radius_b, set_radius_a, set_radius_b, ra, rab, &
654 la_max, la_min, lb_max, lb_min, npgfa, npgfb, nseta, nsetb)
656 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
657 INTEGER :: ntasks, curr_tasks
663 INTEGER :: cindex, iatom, jatom
664 REAL(kind=
dp),
DIMENSION(:, :),
POINTER :: rpgfa, rpgfb, zeta, zetb
665 REAL(kind=
dp) :: kind_radius_b
666 REAL(kind=
dp),
DIMENSION(:),
POINTER :: set_radius_a, set_radius_b
667 REAL(kind=
dp),
DIMENSION(3) :: ra, rab
668 INTEGER,
DIMENSION(:),
POINTER :: la_max, la_min, lb_max, lb_min, npgfa, &
670 INTEGER :: nseta, nsetb
672 INTEGER :: cube_center(3), igrid_level, ipgf, iset, &
673 jpgf, jset, lb_cube(3), ub_cube(3)
674 REAL(kind=
dp) :: dab, rab2, radius, zetp
676 rab2 = rab(1)*rab(1) + rab(2)*rab(2) + rab(3)*rab(3)
679 loop_iset:
DO iset = 1, nseta
681 IF (set_radius_a(iset) + kind_radius_b < dab) cycle loop_iset
683 loop_jset:
DO jset = 1, nsetb
685 IF (set_radius_a(iset) + set_radius_b(jset) < dab) cycle loop_jset
687 loop_ipgf:
DO ipgf = 1, npgfa(iset)
689 IF (rpgfa(ipgf, iset) + set_radius_b(jset) < dab) cycle loop_ipgf
691 loop_jpgf:
DO jpgf = 1, npgfb(jset)
693 IF (rpgfa(ipgf, iset) + rpgfb(jpgf, jset) < dab) cycle loop_jpgf
695 zetp = zeta(ipgf, iset) + zetb(jpgf, jset)
698 CALL compute_pgf_properties(cube_center, lb_cube, ub_cube, radius, &
699 rs_descs(igrid_level)%rs_desc, cube_info(igrid_level), &
700 la_max(iset), zeta(ipgf, iset), la_min(iset), &
701 lb_max(jset), zetb(jpgf, jset), lb_min(jset), &
702 ra, rab, rab2, dft_control%qs_control%eps_rho_rspace)
704 CALL pgf_to_tasks(tasks, ntasks, curr_tasks, &
705 rab, cindex, iatom, jatom, iset, jset, ipgf, jpgf, &
706 la_max(iset), lb_max(jset), rs_descs(igrid_level)%rs_desc, &
707 igrid_level, gridlevel_info%ngrid_levels, cube_center, &
708 lb_cube, ub_cube, radius)
754 SUBROUTINE compute_pgf_properties(cube_center, lb_cube, ub_cube, radius, &
755 rs_desc, cube_info, la_max, zeta, la_min, lb_max, zetb, lb_min, ra, rab, rab2, eps)
757 INTEGER,
DIMENSION(3),
INTENT(OUT) :: cube_center, lb_cube, ub_cube
758 REAL(kind=
dp),
INTENT(OUT) :: radius
761 INTEGER,
INTENT(IN) :: la_max
762 REAL(kind=
dp),
INTENT(IN) :: zeta
763 INTEGER,
INTENT(IN) :: la_min, lb_max
764 REAL(kind=
dp),
INTENT(IN) :: zetb
765 INTEGER,
INTENT(IN) :: lb_min
766 REAL(kind=
dp),
INTENT(IN) :: ra(3), rab(3), rab2, eps
769 INTEGER,
DIMENSION(:),
POINTER :: sphere_bounds
770 REAL(kind=
dp) :: cutoff, f, prefactor, rb(3), zetp
771 REAL(kind=
dp),
DIMENSION(3) :: rp
776 rp(:) = ra(:) + zetb/zetp*rab(:)
777 rb(:) = ra(:) + rab(:)
780 prefactor = exp(-zeta*f*rab2)
782 zetp=zetp, eps=eps, prefactor=prefactor, cutoff=cutoff)
786 cube_center(:) =
modulo(cube_center(:), rs_desc%npts(:))
787 cube_center(:) = cube_center(:) + rs_desc%lb(:)
789 IF (rs_desc%orthorhombic)
THEN
790 CALL return_cube(cube_info, radius, lb_cube, ub_cube, sphere_bounds)
794 extent(:) = ub_cube(:) - lb_cube(:)
795 lb_cube(:) = -extent(:)/2 - 1
796 ub_cube(:) = extent(:)/2
799 END SUBROUTINE compute_pgf_properties
815 INTEGER FUNCTION cost_model(lb_cube, ub_cube, fraction, lmax, is_ortho)
816 INTEGER,
DIMENSION(3),
INTENT(IN) :: lb_cube, ub_cube
817 REAL(kind=
dp),
INTENT(IN) :: fraction
822 REAL(kind=
dp) :: v1, v2, v3, v4, v5
824 cmax = maxval(((ub_cube - lb_cube) + 1)/2)
839 cost_model = ceiling(((lmax + v1)*(cmax + v2)**3*v3*fraction + v4 + v5*lmax**7)/1000.0_dp)
841 END FUNCTION cost_model
875 SUBROUTINE pgf_to_tasks(tasks, ntasks, curr_tasks, &
876 rab, cindex, iatom, jatom, iset, jset, ipgf, jpgf, &
877 la_max, lb_max, rs_desc, igrid_level, n_levels, &
878 cube_center, lb_cube, ub_cube, radius)
880 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
881 INTEGER,
INTENT(INOUT) :: ntasks, curr_tasks
882 REAL(kind=
dp),
DIMENSION(3),
INTENT(IN) :: rab
883 INTEGER,
INTENT(IN) :: cindex, iatom, jatom, iset, jset, ipgf, &
886 INTEGER,
INTENT(IN) :: igrid_level, n_levels
887 INTEGER,
DIMENSION(3),
INTENT(IN) :: cube_center, lb_cube, ub_cube
888 REAL(kind=
dp),
INTENT(IN) :: radius
890 INTEGER,
PARAMETER :: add_tasks = 1000
891 REAL(kind=
dp),
PARAMETER :: mult_tasks = 2.0_dp
893 INTEGER :: added_tasks, cost, j, lmax
895 REAL(kind=
dp) :: tfraction
899 IF (ntasks > curr_tasks)
THEN
900 curr_tasks = int((curr_tasks + add_tasks)*mult_tasks)
905 IF (rs_desc%distributed)
THEN
909 CALL rs_find_node(rs_desc, igrid_level, n_levels, cube_center, &
910 ntasks=ntasks, tasks=tasks, lb_cube=lb_cube, ub_cube=ub_cube, added_tasks=added_tasks)
913 tasks(ntasks)%destination = encode_rank(rs_desc%my_pos, igrid_level, n_levels)
914 tasks(ntasks)%dist_type = 0
915 tasks(ntasks)%subpatch_pattern = 0
919 lmax = la_max + lb_max
920 is_ortho = (tasks(ntasks)%dist_type == 0 .OR. tasks(ntasks)%dist_type == 1) .AND. rs_desc%orthorhombic
923 tfraction = 1.0_dp/added_tasks
925 cost = cost_model(lb_cube, ub_cube, tfraction, lmax, is_ortho)
927 DO j = 1, added_tasks
928 tasks(ntasks - added_tasks + j)%source = encode_rank(rs_desc%my_pos, igrid_level, n_levels)
929 tasks(ntasks - added_tasks + j)%cost = cost
930 tasks(ntasks - added_tasks + j)%grid_level = igrid_level
931 tasks(ntasks - added_tasks + j)%image = cindex
932 tasks(ntasks - added_tasks + j)%iatom = iatom
933 tasks(ntasks - added_tasks + j)%jatom = jatom
934 tasks(ntasks - added_tasks + j)%iset = iset
935 tasks(ntasks - added_tasks + j)%jset = jset
936 tasks(ntasks - added_tasks + j)%ipgf = ipgf
937 tasks(ntasks - added_tasks + j)%jpgf = jpgf
938 tasks(ntasks - added_tasks + j)%rab = rab
939 tasks(ntasks - added_tasks + j)%radius = radius
942 END SUBROUTINE pgf_to_tasks
954 SUBROUTINE load_balance_distributed(tasks, ntasks, rs_descs, grid_level, natoms)
956 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
960 INTEGER :: grid_level, natoms
962 CHARACTER(LEN=*),
PARAMETER :: routinen =
'load_balance_distributed'
965 INTEGER,
DIMENSION(:, :, :),
POINTER ::
list
967 CALL timeset(routinen, handle)
972 CALL create_destination_list(
list, rs_descs, grid_level)
975 CALL compute_load_list(
list, rs_descs, grid_level, tasks, ntasks, natoms, create_list=.true.)
978 CALL optimize_load_list(
list, rs_descs(1)%rs_desc%group, rs_descs(1)%rs_desc%my_pos)
981 CALL compute_load_list(
list, rs_descs, grid_level, tasks, ntasks, natoms, create_list=.false.)
985 CALL timestop(handle)
987 END SUBROUTINE load_balance_distributed
996 SUBROUTINE balance_global_list(list_global)
997 INTEGER,
DIMENSION(:, :, 0:) :: list_global
999 CHARACTER(LEN=*),
PARAMETER :: routinen =
'balance_global_list'
1000 INTEGER,
PARAMETER :: max_iter = 100
1001 REAL(kind=
dp),
PARAMETER :: tolerance_factor = 0.005_dp
1003 INTEGER :: dest, handle, icpu, idest, iflux, &
1004 ilocal, k, maxdest, ncpu, nflux
1005 INTEGER,
ALLOCATABLE,
DIMENSION(:, :) :: flux_connections
1006 LOGICAL :: solution_optimal
1007 REAL(kind=
dp) :: average, load_shift, max_load_shift, &
1009 REAL(kind=
dp),
ALLOCATABLE,
DIMENSION(:) :: load, optimized_flux, optimized_load
1010 REAL(kind=
dp),
ALLOCATABLE,
DIMENSION(:, :) :: flux_limits
1012 CALL timeset(routinen, handle)
1014 ncpu =
SIZE(list_global, 3)
1015 maxdest =
SIZE(list_global, 2)
1016 ALLOCATE (load(0:ncpu - 1))
1018 ALLOCATE (optimized_load(0:ncpu - 1))
1023 DO icpu = 0, ncpu - 1
1024 DO idest = 1, maxdest
1025 dest = list_global(1, idest, icpu)
1026 IF (dest < ncpu .AND. dest > icpu) nflux = nflux + 1
1029 ALLOCATE (optimized_flux(nflux))
1030 ALLOCATE (flux_limits(2, nflux))
1031 ALLOCATE (flux_connections(2, nflux))
1036 DO icpu = 0, ncpu - 1
1037 load(icpu) = sum(list_global(2, :, icpu))
1038 DO idest = 1, maxdest
1039 dest = list_global(1, idest, icpu)
1040 IF (dest < ncpu)
THEN
1041 IF (dest /= icpu)
THEN
1042 IF (dest > icpu)
THEN
1044 flux_limits(2, nflux) = list_global(2, idest, icpu)
1045 flux_connections(1, nflux) = icpu
1046 flux_connections(2, nflux) = dest
1049 IF (flux_connections(1, iflux) == dest .AND. flux_connections(2, iflux) == icpu)
THEN
1050 flux_limits(1, iflux) = -list_global(2, idest, icpu)
1060 solution_optimal = .false.
1061 optimized_flux = 0.0_dp
1068 average = sum(load)/
SIZE(load)
1069 tolerance = tolerance_factor*average
1071 optimized_load(:) = load
1073 max_load_shift = 0.0_dp
1075 load_shift = (optimized_load(flux_connections(1, iflux)) - optimized_load(flux_connections(2, iflux)))/2
1076 load_shift = max(flux_limits(1, iflux) - optimized_flux(iflux), load_shift)
1077 load_shift = min(flux_limits(2, iflux) - optimized_flux(iflux), load_shift)
1078 max_load_shift = max(abs(load_shift), max_load_shift)
1079 optimized_load(flux_connections(1, iflux)) = optimized_load(flux_connections(1, iflux)) - load_shift
1080 optimized_load(flux_connections(2, iflux)) = optimized_load(flux_connections(2, iflux)) + load_shift
1081 optimized_flux(iflux) = optimized_flux(iflux) + load_shift
1083 IF (max_load_shift < tolerance)
THEN
1084 solution_optimal = .true.
1092 DO icpu = 0, ncpu - 1
1093 DO idest = 1, maxdest
1094 IF (list_global(1, idest, icpu) == icpu) ilocal = idest
1096 DO idest = 1, maxdest
1097 dest = list_global(1, idest, icpu)
1098 IF (dest < ncpu)
THEN
1099 IF (dest /= icpu)
THEN
1100 IF (dest > icpu)
THEN
1102 IF (optimized_flux(nflux) > 0)
THEN
1103 list_global(2, ilocal, icpu) = list_global(2, ilocal, icpu) + &
1104 list_global(2, idest, icpu) - nint(optimized_flux(nflux))
1105 list_global(2, idest, icpu) = nint(optimized_flux(nflux))
1107 list_global(2, ilocal, icpu) = list_global(2, ilocal, icpu) + &
1108 list_global(2, idest, icpu)
1109 list_global(2, idest, icpu) = 0
1113 IF (flux_connections(1, iflux) == dest .AND. flux_connections(2, iflux) == icpu)
THEN
1114 IF (optimized_flux(iflux) > 0)
THEN
1115 list_global(2, ilocal, icpu) = list_global(2, ilocal, icpu) + &
1116 list_global(2, idest, icpu)
1117 list_global(2, idest, icpu) = 0
1119 list_global(2, ilocal, icpu) = list_global(2, ilocal, icpu) + &
1120 list_global(2, idest, icpu) + nint(optimized_flux(iflux))
1121 list_global(2, idest, icpu) = -nint(optimized_flux(iflux))
1132 CALL timestop(handle)
1134 END SUBROUTINE balance_global_list
1147 SUBROUTINE optimize_load_list(list, group, my_pos)
1148 INTEGER,
DIMENSION(:, :, 0:) ::
list
1150 INTEGER,
INTENT(IN) :: my_pos
1152 CHARACTER(LEN=*),
PARAMETER :: routinen =
'optimize_load_list'
1153 INTEGER,
PARAMETER :: rank_of_root = 0
1155 INTEGER :: handle, icpu, idest, maxdest, ncpu
1156 INTEGER,
ALLOCATABLE,
DIMENSION(:) :: load_all
1157 INTEGER,
ALLOCATABLE,
DIMENSION(:, :) :: load_partial
1158 INTEGER,
ALLOCATABLE,
DIMENSION(:, :, :) :: list_global
1160 CALL timeset(routinen, handle)
1162 ncpu =
SIZE(
list, 3)
1163 maxdest =
SIZE(
list, 2)
1166 ALLOCATE (load_all(maxdest*ncpu))
1167 load_all(:) = reshape(
list(2, :, :), [maxdest*ncpu])
1168 CALL group%sum(load_all(:), rank_of_root)
1171 ALLOCATE (list_global(2, maxdest, ncpu))
1172 IF (rank_of_root == my_pos)
THEN
1173 list_global(1, :, :) =
list(1, :, :)
1174 list_global(2, :, :) = reshape(load_all, [maxdest, ncpu])
1175 CALL balance_global_list(list_global)
1177 CALL group%bcast(list_global, rank_of_root)
1180 ALLOCATE (load_partial(maxdest, ncpu))
1182 CALL group%sum_partial(reshape(load_all, [maxdest, ncpu]), load_partial(:, :))
1185 DO idest = 1, maxdest
1188 IF (load_partial(idest, icpu) > list_global(2, idest, icpu))
THEN
1189 IF (load_partial(idest, icpu) -
list(2, idest, icpu - 1) < list_global(2, idest, icpu))
THEN
1190 list(2, idest, icpu - 1) = list_global(2, idest, icpu) &
1191 - (load_partial(idest, icpu) -
list(2, idest, icpu - 1))
1193 list(2, idest, icpu - 1) = 0
1201 DEALLOCATE (load_all)
1202 DEALLOCATE (list_global)
1203 DEALLOCATE (load_partial)
1205 CALL timestop(handle)
1206 END SUBROUTINE optimize_load_list
1225 SUBROUTINE compute_load_list(list, rs_descs, grid_level, tasks, ntasks, natoms, create_list)
1226 INTEGER,
DIMENSION(:, :, 0:) ::
list
1229 INTEGER :: grid_level
1230 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
1231 INTEGER :: ntasks, natoms
1232 LOGICAL :: create_list
1234 CHARACTER(LEN=*),
PARAMETER :: routinen =
'compute_load_list'
1236 INTEGER :: cost, dest, handle, i, iatom, ilevel, img, img_old, iopt, ipgf, iset, itask, &
1237 itask_start, itask_stop, jatom, jpgf, jset, li, maxdest, ncpu, ndest_pair, nopt, nshort, &
1239 INTEGER(KIND=int_8) :: bit_pattern, ipair, ipair_old, natom8
1240 INTEGER(KIND=int_8),
ALLOCATABLE,
DIMENSION(:) :: loads
1241 INTEGER,
ALLOCATABLE,
DIMENSION(:) :: all_dests, index
1242 INTEGER,
DIMENSION(6) :: options
1244 CALL timeset(routinen, handle)
1246 ALLOCATE (loads(0:rs_descs(grid_level)%rs_desc%group_size - 1))
1247 CALL get_current_loads(loads, rs_descs, grid_level, ntasks, tasks, use_reordered_ranks=.false.)
1249 maxdest =
SIZE(
list, 2)
1250 ncpu =
SIZE(
list, 3)
1255 ipair_old = huge(ipair_old)
1257 ALLOCATE (all_dests(0))
1263 itask_start = itask_stop + 1
1264 itask_stop = itask_start
1265 IF (itask_stop > ntasks)
EXIT
1266 ilevel = tasks(itask_stop)%grid_level
1267 img_old = tasks(itask_stop)%image
1268 iatom = tasks(itask_stop)%iatom
1269 jatom = tasks(itask_stop)%jatom
1270 iset = tasks(itask_stop)%iset
1271 jset = tasks(itask_stop)%jset
1272 ipgf = tasks(itask_stop)%ipgf
1273 jpgf = tasks(itask_stop)%jpgf
1275 ipair_old = (iatom - 1)*natom8 + (jatom - 1)
1277 IF (itask_stop + 1 > ntasks)
EXIT
1278 ilevel = tasks(itask_stop + 1)%grid_level
1279 img = tasks(itask_stop + 1)%image
1280 iatom = tasks(itask_stop + 1)%iatom
1281 jatom = tasks(itask_stop + 1)%jatom
1282 iset = tasks(itask_stop + 1)%iset
1283 jset = tasks(itask_stop + 1)%jset
1284 ipgf = tasks(itask_stop + 1)%ipgf
1285 jpgf = tasks(itask_stop + 1)%jpgf
1287 ipair = (iatom - 1)*natom8 + (jatom - 1)
1288 IF (ipair == ipair_old .AND. img == img_old)
THEN
1289 itask_stop = itask_stop + 1
1295 nshort = itask_stop - itask_start + 1
1298 DEALLOCATE (all_dests)
1299 ALLOCATE (all_dests(nshort))
1301 ALLOCATE (index(nshort))
1303 ilevel = tasks(itask_start + i - 1)%grid_level
1304 img = tasks(itask_start + i - 1)%image
1305 iatom = tasks(itask_start + i - 1)%iatom
1306 jatom = tasks(itask_start + i - 1)%jatom
1307 iset = tasks(itask_start + i - 1)%iset
1308 jset = tasks(itask_start + i - 1)%jset
1309 ipgf = tasks(itask_start + i - 1)%ipgf
1310 jpgf = tasks(itask_start + i - 1)%jpgf
1312 IF (ilevel == grid_level)
THEN
1313 all_dests(i) = decode_rank(tasks(itask_start + i - 1)%destination,
SIZE(rs_descs))
1315 all_dests(i) = huge(all_dests(i))
1318 CALL sort(all_dests, nshort, index)
1321 IF ((all_dests(ndest_pair) /= all_dests(i)) .AND. (all_dests(i) /= huge(all_dests(i))))
THEN
1322 ndest_pair = ndest_pair + 1
1323 all_dests(ndest_pair) = all_dests(i)
1327 DO itask = itask_start, itask_stop
1329 dest = decode_rank(tasks(itask)%destination,
SIZE(rs_descs))
1330 ilevel = tasks(itask)%grid_level
1331 img = tasks(itask)%image
1332 iatom = tasks(itask)%iatom
1333 jatom = tasks(itask)%jatom
1334 iset = tasks(itask)%iset
1335 jset = tasks(itask)%jset
1336 ipgf = tasks(itask)%ipgf
1337 jpgf = tasks(itask)%jpgf
1340 IF (ilevel /= grid_level) cycle
1341 ipair = (iatom - 1)*natom8 + (jatom - 1)
1342 cost = int(tasks(itask)%cost)
1344 SELECT CASE (tasks(itask)%dist_type)
1346 bit_pattern = tasks(itask)%subpatch_pattern
1348 IF (btest(bit_pattern, 0))
THEN
1350 IF (any(all_dests(1:ndest_pair) == rank))
THEN
1352 options(nopt) = rank
1355 IF (btest(bit_pattern, 1))
THEN
1357 IF (any(all_dests(1:ndest_pair) == rank))
THEN
1359 options(nopt) = rank
1362 IF (btest(bit_pattern, 2))
THEN
1364 IF (any(all_dests(1:ndest_pair) == rank))
THEN
1366 options(nopt) = rank
1369 IF (btest(bit_pattern, 3))
THEN
1371 IF (any(all_dests(1:ndest_pair) == rank))
THEN
1373 options(nopt) = rank
1376 IF (btest(bit_pattern, 4))
THEN
1378 IF (any(all_dests(1:ndest_pair) == rank))
THEN
1380 options(nopt) = rank
1383 IF (btest(bit_pattern, 5))
THEN
1385 IF (any(all_dests(1:ndest_pair) == rank))
THEN
1387 options(nopt) = rank
1394 IF (loads(rank) > loads(options(iopt))) rank = options(iopt)
1399 li = list_index(
list, rank, dest)
1400 IF (create_list)
THEN
1401 list(2, li, dest) =
list(2, li, dest) + cost
1403 IF (
list(1, li, dest) == dest)
THEN
1404 tasks(itask)%destination = encode_rank(dest, ilevel,
SIZE(rs_descs))
1406 IF (
list(2, li, dest) >= cost)
THEN
1407 list(2, li, dest) =
list(2, li, dest) - cost
1408 tasks(itask)%destination = encode_rank(
list(1, li, dest), ilevel,
SIZE(rs_descs))
1410 tasks(itask)%destination = encode_rank(dest, ilevel,
SIZE(rs_descs))
1415 li = list_index(
list, dest, dest)
1416 IF (create_list)
THEN
1417 list(2, li, dest) =
list(2, li, dest) + cost
1419 IF (
list(1, li, dest) == dest)
THEN
1420 tasks(itask)%destination = encode_rank(dest, ilevel,
SIZE(rs_descs))
1422 IF (
list(2, li, dest) >= cost)
THEN
1423 list(2, li, dest) =
list(2, li, dest) - cost
1424 tasks(itask)%destination = encode_rank(
list(1, li, dest), ilevel,
SIZE(rs_descs))
1426 tasks(itask)%destination = encode_rank(dest, ilevel,
SIZE(rs_descs))
1431 cpabort(
"Unknown task list distribution type")
1438 CALL timestop(handle)
1440 END SUBROUTINE compute_load_list
1451 INTEGER FUNCTION list_index(list, rank, dest)
1452 INTEGER,
DIMENSION(:, :, 0:),
INTENT(IN) ::
list
1453 INTEGER,
INTENT(IN) :: rank, dest
1457 IF (
list(1, list_index, dest) == rank)
EXIT
1458 list_index = list_index + 1
1460 END FUNCTION list_index
1471 SUBROUTINE create_destination_list(list, rs_descs, grid_level)
1472 INTEGER,
DIMENSION(:, :, :),
POINTER ::
list
1475 INTEGER,
INTENT(IN) :: grid_level
1477 CHARACTER(LEN=*),
PARAMETER :: routinen =
'create_destination_list'
1479 INTEGER :: handle, i, icpu, j, maxcount, ncpu, &
1481 INTEGER,
ALLOCATABLE,
DIMENSION(:) :: index, sublist
1483 CALL timeset(routinen, handle)
1485 cpassert(.NOT.
ASSOCIATED(
list))
1486 ncpu = rs_descs(grid_level)%rs_desc%group_size
1489 ALLOCATE (
list(2, ultimate_max, 0:ncpu - 1))
1491 ALLOCATE (index(ultimate_max))
1492 ALLOCATE (sublist(ultimate_max))
1493 sublist = huge(sublist)
1496 DO icpu = 0, ncpu - 1
1505 CALL sort(sublist, ultimate_max, index)
1508 IF (sublist(i) /= sublist(j))
THEN
1510 sublist(j) = sublist(i)
1513 maxcount = max(maxcount, j)
1514 sublist(j + 1:ultimate_max) = huge(sublist)
1515 list(1, :, icpu) = sublist
1516 list(2, :, icpu) = 0
1521 CALL timestop(handle)
1523 END SUBROUTINE create_destination_list
1539 SUBROUTINE get_current_loads(loads, rs_descs, grid_level, ntasks, tasks, use_reordered_ranks)
1540 INTEGER(KIND=int_8),
DIMENSION(:) :: loads
1543 INTEGER :: grid_level, ntasks
1544 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
1545 LOGICAL,
INTENT(IN) :: use_reordered_ranks
1547 CHARACTER(LEN=*),
PARAMETER :: routinen =
'get_current_loads'
1549 INTEGER :: handle, i, iatom, ilevel, img, ipgf, &
1550 iset, jatom, jpgf, jset
1551 INTEGER(KIND=int_8) :: total_cost_local
1552 INTEGER(KIND=int_8),
ALLOCATABLE,
DIMENSION(:) :: recv_buf_i, send_buf_i
1555 CALL timeset(routinen, handle)
1557 desc => rs_descs(grid_level)%rs_desc
1560 ALLOCATE (send_buf_i(desc%group_size))
1561 ALLOCATE (recv_buf_i(desc%group_size))
1569 ilevel = tasks(i)%grid_level
1570 img = tasks(i)%image
1571 iatom = tasks(i)%iatom
1572 jatom = tasks(i)%jatom
1573 iset = tasks(i)%iset
1574 jset = tasks(i)%jset
1575 ipgf = tasks(i)%ipgf
1576 jpgf = tasks(i)%jpgf
1577 IF (ilevel /= grid_level) cycle
1578 IF (use_reordered_ranks)
THEN
1579 send_buf_i(rs_descs(ilevel)%rs_desc%virtual2real(decode_rank(tasks(i)%destination,
SIZE(rs_descs))) + 1) = &
1580 send_buf_i(rs_descs(ilevel)%rs_desc%virtual2real(decode_rank(tasks(i)%destination,
SIZE(rs_descs))) + 1) &
1583 send_buf_i(decode_rank(tasks(i)%destination,
SIZE(rs_descs)) + 1) = &
1584 send_buf_i(decode_rank(tasks(i)%destination,
SIZE(rs_descs)) + 1) &
1588 CALL desc%group%alltoall(send_buf_i, recv_buf_i, 1)
1591 total_cost_local = sum(recv_buf_i)
1594 CALL desc%group%allgather(total_cost_local, loads)
1596 CALL timestop(handle)
1598 END SUBROUTINE get_current_loads
1610 SUBROUTINE load_balance_replicated(rs_descs, ntasks, tasks)
1615 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
1617 CHARACTER(LEN=*),
PARAMETER :: routinen =
'load_balance_replicated'
1619 INTEGER :: handle, i, iatom, ilevel, img, ipgf, &
1620 iset, j, jatom, jpgf, jset, &
1621 no_overloaded, no_underloaded, &
1623 INTEGER(KIND=int_8) :: average_cost, cost_task_rep, count, &
1624 offset, total_cost_global
1625 INTEGER(KIND=int_8),
ALLOCATABLE,
DIMENSION(:) :: load_imbalance, loads, recv_buf_i
1626 INTEGER,
ALLOCATABLE,
DIMENSION(:) :: index
1629 CALL timeset(routinen, handle)
1631 desc => rs_descs(1)%rs_desc
1634 ALLOCATE (recv_buf_i(desc%group_size))
1635 ALLOCATE (loads(desc%group_size))
1638 DO i = 1,
SIZE(rs_descs)
1639 CALL get_current_loads(loads, rs_descs, i, ntasks, tasks, use_reordered_ranks=.true.)
1640 recv_buf_i(:) = recv_buf_i + loads
1643 total_cost_global = sum(recv_buf_i)
1644 average_cost = total_cost_global/desc%group_size
1652 ALLOCATE (load_imbalance(desc%group_size))
1653 ALLOCATE (index(desc%group_size))
1655 load_imbalance(:) = recv_buf_i - average_cost
1659 DO i = 1, desc%group_size
1660 IF (load_imbalance(i) > 0) no_overloaded = no_overloaded + 1
1661 IF (load_imbalance(i) < 0) no_underloaded = no_underloaded + 1
1666 CALL sort(recv_buf_i,
SIZE(recv_buf_i), index)
1672 IF (tasks(i)%dist_type == 0 &
1673 .AND. decode_rank(tasks(i)%destination,
SIZE(rs_descs)) == decode_rank(tasks(i)%source,
SIZE(rs_descs)))
THEN
1674 cost_task_rep = cost_task_rep + tasks(i)%cost
1680 CALL desc%group%allgather(cost_task_rep, recv_buf_i)
1682 DO i = 1, desc%group_size
1684 IF (load_imbalance(i) > 0) load_imbalance(i) = min(load_imbalance(i), recv_buf_i(i))
1693 IF (load_imbalance(desc%my_pos + 1) > 0)
THEN
1699 DO i = desc%group_size, desc%group_size - no_overloaded + 1, -1
1700 IF (index(i) == desc%my_pos + 1)
THEN
1703 offset = offset + load_imbalance(index(i))
1708 proc_receiving = huge(proc_receiving)
1709 DO i = 1, no_underloaded
1710 offset = offset + load_imbalance(index(i))
1711 IF (offset <= 0)
THEN
1721 IF (tasks(j)%dist_type == 0 &
1722 .AND. decode_rank(tasks(j)%destination,
SIZE(rs_descs)) == decode_rank(tasks(j)%source,
SIZE(rs_descs)))
THEN
1725 IF (proc_receiving > no_underloaded)
EXIT
1727 ilevel = tasks(j)%grid_level
1728 img = tasks(j)%image
1729 iatom = tasks(j)%iatom
1730 jatom = tasks(j)%jatom
1731 iset = tasks(j)%iset
1732 jset = tasks(j)%jset
1733 ipgf = tasks(j)%ipgf
1734 jpgf = tasks(j)%jpgf
1735 tasks(j)%destination = encode_rank(index(proc_receiving) - 1, ilevel,
SIZE(rs_descs))
1736 offset = offset + tasks(j)%cost
1737 count = count + tasks(j)%cost
1738 IF (count >= load_imbalance(desc%my_pos + 1))
EXIT
1739 IF (offset > 0)
THEN
1740 proc_receiving = proc_receiving + 1
1743 IF (proc_receiving > no_underloaded)
EXIT
1744 offset = load_imbalance(index(proc_receiving))
1751 DEALLOCATE (load_imbalance)
1753 CALL timestop(handle)
1755 END SUBROUTINE load_balance_replicated
1769 SUBROUTINE create_local_tasks(rs_descs, ntasks, tasks, ntasks_recv, tasks_recv)
1774 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
1775 INTEGER :: ntasks_recv
1776 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks_recv
1778 CHARACTER(LEN=*),
PARAMETER :: routinen =
'create_local_tasks'
1780 INTEGER :: handle, i, j, k, l, rank
1781 INTEGER(KIND=int_8),
ALLOCATABLE,
DIMENSION(:) :: recv_buf, send_buf
1782 INTEGER,
ALLOCATABLE,
DIMENSION(:) :: recv_disps, recv_sizes, send_disps, &
1786 CALL timeset(routinen, handle)
1788 desc => rs_descs(1)%rs_desc
1791 ALLOCATE (send_sizes(desc%group_size))
1792 ALLOCATE (recv_sizes(desc%group_size))
1793 ALLOCATE (send_disps(desc%group_size))
1794 ALLOCATE (recv_disps(desc%group_size))
1795 ALLOCATE (send_buf(desc%group_size))
1796 ALLOCATE (recv_buf(desc%group_size))
1801 rank = rs_descs(decode_level(tasks(i)%destination,
SIZE(rs_descs))) &
1802 %rs_desc%virtual2real(decode_rank(tasks(i)%destination,
SIZE(rs_descs)))
1803 send_buf(rank + 1) = send_buf(rank + 1) + 1
1806 CALL desc%group%alltoall(send_buf, recv_buf, 1)
1817 DO i = 2, desc%group_size
1820 send_disps(i) = send_disps(i - 1) + send_sizes(i - 1)
1821 recv_disps(i) = recv_disps(i - 1) + recv_sizes(i - 1)
1825 DEALLOCATE (send_buf)
1826 DEALLOCATE (recv_buf)
1829 ALLOCATE (send_buf(sum(send_sizes)))
1830 ALLOCATE (recv_buf(sum(recv_sizes)))
1836 i = rs_descs(decode_level(tasks(j)%destination,
SIZE(rs_descs))) &
1837 %rs_desc%virtual2real(decode_rank(tasks(j)%destination,
SIZE(rs_descs))) + 1
1838 l = send_disps(i) + send_sizes(i)
1844 CALL desc%group%alltoall(send_buf, send_sizes, send_disps, recv_buf, recv_sizes, recv_disps)
1846 DEALLOCATE (send_buf)
1849 ALLOCATE (tasks_recv(ntasks_recv))
1853 DO i = 1, desc%group_size
1861 DEALLOCATE (recv_buf)
1862 DEALLOCATE (send_sizes)
1863 DEALLOCATE (recv_sizes)
1864 DEALLOCATE (send_disps)
1865 DEALLOCATE (recv_disps)
1867 CALL timestop(handle)
1869 END SUBROUTINE create_local_tasks
1889 tasks, atom_pair_send, atom_pair_recv, &
1890 symmetric, reorder_rs_grid_ranks, skip_load_balance_distributed)
1894 INTEGER :: ntasks, natoms
1895 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
1896 TYPE(
atom_pair_type),
DIMENSION(:),
POINTER :: atom_pair_send, atom_pair_recv
1897 LOGICAL,
INTENT(IN) :: symmetric, reorder_rs_grid_ranks, &
1898 skip_load_balance_distributed
1900 CHARACTER(LEN=*),
PARAMETER :: routinen =
'distribute_tasks'
1902 INTEGER :: handle, igrid_level, irank, ntasks_recv
1903 INTEGER(KIND=int_8) :: load_gap, max_load, replicated_load
1904 INTEGER(KIND=int_8),
ALLOCATABLE,
DIMENSION(:) :: total_loads, total_loads_tmp, trial_loads
1905 INTEGER(KIND=int_8),
DIMENSION(:, :),
POINTER :: loads
1906 INTEGER,
ALLOCATABLE,
DIMENSION(:) :: indices, real2virtual, total_index
1907 LOGICAL :: distributed_grids, fixed_first_grid
1909 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks_recv
1911 CALL timeset(routinen, handle)
1913 cpassert(
ASSOCIATED(tasks))
1916 distributed_grids = .false.
1917 DO igrid_level = 1,
SIZE(rs_descs)
1918 IF (rs_descs(igrid_level)%rs_desc%distributed)
THEN
1919 distributed_grids = .true.
1922 desc => rs_descs(1)%rs_desc
1924 IF (distributed_grids)
THEN
1926 ALLOCATE (loads(0:desc%group_size - 1,
SIZE(rs_descs)))
1927 ALLOCATE (total_loads(0:desc%group_size - 1))
1933 DO igrid_level = 1,
SIZE(rs_descs)
1934 IF (rs_descs(igrid_level)%rs_desc%distributed)
THEN
1936 IF (.NOT. skip_load_balance_distributed)
THEN
1937 CALL load_balance_distributed(tasks, ntasks, rs_descs, igrid_level, natoms)
1940 CALL get_current_loads(loads(:, igrid_level), rs_descs, igrid_level, ntasks, &
1941 tasks, use_reordered_ranks=.false.)
1943 total_loads(:) = total_loads + loads(:, igrid_level)
1952 DO igrid_level = 1,
SIZE(rs_descs)
1953 IF (.NOT. rs_descs(igrid_level)%rs_desc%distributed)
THEN
1954 CALL get_current_loads(loads(:, igrid_level), rs_descs, igrid_level, ntasks, &
1955 tasks, use_reordered_ranks=.false.)
1956 replicated_load = replicated_load + sum(loads(:, igrid_level))
1962 IF (reorder_rs_grid_ranks)
THEN
1963 fixed_first_grid = .false.
1964 DO igrid_level = 1,
SIZE(rs_descs)
1965 IF (rs_descs(igrid_level)%rs_desc%distributed)
THEN
1966 IF (fixed_first_grid .EQV. .false.)
THEN
1967 total_loads(:) = loads(:, igrid_level)
1968 fixed_first_grid = .true.
1970 ALLOCATE (trial_loads(0:desc%group_size - 1))
1972 trial_loads(:) = total_loads + loads(:, igrid_level)
1973 max_load = maxval(trial_loads)
1975 DO irank = 0, desc%group_size - 1
1976 load_gap = load_gap + max_load - trial_loads(irank)
1981 IF (load_gap > replicated_load*1.05_dp)
THEN
1983 ALLOCATE (indices(0:desc%group_size - 1))
1984 ALLOCATE (total_index(0:desc%group_size - 1))
1985 ALLOCATE (total_loads_tmp(0:desc%group_size - 1))
1986 ALLOCATE (real2virtual(0:desc%group_size - 1))
1988 total_loads_tmp(:) = total_loads
1989 CALL sort(total_loads_tmp, desc%group_size, total_index)
1990 CALL sort(loads(:, igrid_level), desc%group_size, indices)
1994 DO irank = 0, desc%group_size - 1
1995 total_loads(total_index(irank) - 1) = total_loads(total_index(irank) - 1) + &
1996 loads(desc%group_size - irank - 1, igrid_level)
1997 real2virtual(total_index(irank) - 1) = indices(desc%group_size - irank - 1) - 1
2002 DEALLOCATE (indices)
2003 DEALLOCATE (total_index)
2004 DEALLOCATE (total_loads_tmp)
2005 DEALLOCATE (real2virtual)
2007 total_loads(:) = trial_loads
2010 DEALLOCATE (trial_loads)
2018 CALL load_balance_replicated(rs_descs, ntasks, tasks)
2021 CALL create_local_tasks(rs_descs, ntasks, tasks, ntasks_recv, tasks_recv)
2027 CALL get_atom_pair(atom_pair_send, tasks, ntasks=ntasks, send=.true., symmetric=symmetric, rs_descs=rs_descs)
2029 CALL get_atom_pair(atom_pair_recv, tasks_recv, ntasks=ntasks_recv, send=.false., symmetric=symmetric, rs_descs=rs_descs)
2034 DEALLOCATE (total_loads)
2038 ntasks_recv = ntasks
2039 CALL get_atom_pair(atom_pair_recv, tasks_recv, ntasks=ntasks_recv, send=.false., symmetric=symmetric, rs_descs=rs_descs)
2044 ALLOCATE (indices(ntasks_recv))
2045 CALL tasks_sort(tasks_recv, ntasks_recv, indices)
2046 DEALLOCATE (indices)
2053 ntasks = ntasks_recv
2055 CALL timestop(handle)
2069 SUBROUTINE get_atom_pair(atom_pair, tasks, ntasks, send, symmetric, rs_descs)
2072 TYPE(
task_type),
DIMENSION(:),
INTENT(INOUT) :: tasks
2073 INTEGER,
INTENT(IN) :: ntasks
2074 LOGICAL,
INTENT(IN) :: send, symmetric
2077 INTEGER :: i, ilevel, iatom, jatom, npairs, virt_rank
2078 INTEGER,
DIMENSION(:),
ALLOCATABLE :: indices
2081 cpassert(.NOT.
ASSOCIATED(atom_pair))
2082 IF (ntasks == 0)
THEN
2083 ALLOCATE (atom_pair(0))
2089 ALLOCATE (atom_pair_tmp(ntasks))
2091 atom_pair_tmp(i)%image = tasks(i)%image
2092 iatom = tasks(i)%iatom
2093 jatom = tasks(i)%jatom
2094 IF (symmetric .AND. iatom > jatom)
THEN
2096 atom_pair_tmp(i)%row = jatom
2097 atom_pair_tmp(i)%col = iatom
2099 atom_pair_tmp(i)%row = iatom
2100 atom_pair_tmp(i)%col = jatom
2106 ilevel = tasks(i)%grid_level
2107 virt_rank = decode_rank(tasks(i)%destination,
SIZE(rs_descs))
2108 atom_pair_tmp(i)%rank = rs_descs(ilevel)%rs_desc%virtual2real(virt_rank)
2112 atom_pair_tmp(i)%rank = decode_rank(tasks(i)%source,
SIZE(rs_descs))
2117 ALLOCATE (indices(ntasks))
2118 CALL atom_pair_sort(atom_pair_tmp, ntasks, indices)
2120 tasks(indices(1))%pair_index = 1
2122 IF (atom_pair_less_than(atom_pair_tmp(i - 1), atom_pair_tmp(i)))
THEN
2124 atom_pair_tmp(npairs) = atom_pair_tmp(i)
2126 tasks(indices(i))%pair_index = npairs
2128 DEALLOCATE (indices)
2131 ALLOCATE (atom_pair(npairs))
2132 atom_pair(:) = atom_pair_tmp(:npairs)
2133 DEALLOCATE (atom_pair_tmp)
2135 END SUBROUTINE get_atom_pair
2151 nimages, scatter, hmats)
2156 TYPE(
atom_pair_type),
DIMENSION(:),
POINTER :: atom_pair_send, atom_pair_recv
2162 CHARACTER(LEN=*),
PARAMETER :: routinen =
'rs_distribute_matrix'
2164 INTEGER :: acol, arow, handle, i, img, j, k, l, me, &
2165 nblkcols_total, nblkrows_total, ncol, &
2166 nrow, nthread, nthread_left
2167 INTEGER,
ALLOCATABLE,
DIMENSION(:) :: first_col, first_row, last_col, last_row, recv_disps, &
2168 recv_pair_count, recv_pair_disps, recv_sizes, send_disps, send_pair_count, &
2169 send_pair_disps, send_sizes
2170 INTEGER,
DIMENSION(:),
POINTER :: col_blk_size, row_blk_size
2172 REAL(kind=
dp),
ALLOCATABLE,
DIMENSION(:),
TARGET :: recv_buf_r, send_buf_r
2173 REAL(kind=
dp),
DIMENSION(:, :),
POINTER :: h_block, p_block
2176 REAL(kind=
dp),
DIMENSION(:),
POINTER :: vector
2180 CALL timeset(routinen, handle)
2182 IF (.NOT. scatter)
THEN
2183 cpassert(
PRESENT(hmats))
2186 desc => rs_descs(1)%rs_desc
2187 me = desc%my_pos + 1
2190 ALLOCATE (send_sizes(desc%group_size))
2191 ALLOCATE (recv_sizes(desc%group_size))
2192 ALLOCATE (send_disps(desc%group_size))
2193 ALLOCATE (recv_disps(desc%group_size))
2194 ALLOCATE (send_pair_count(desc%group_size))
2195 ALLOCATE (recv_pair_count(desc%group_size))
2196 ALLOCATE (send_pair_disps(desc%group_size))
2197 ALLOCATE (recv_pair_disps(desc%group_size))
2199 pmat => pmats(1)%matrix
2201 row_blk_size=row_blk_size, &
2202 col_blk_size=col_blk_size, &
2203 nblkrows_total=nblkrows_total, &
2204 nblkcols_total=nblkcols_total)
2205 ALLOCATE (first_row(nblkrows_total), last_row(nblkrows_total), &
2206 first_col(nblkcols_total), last_col(nblkcols_total))
2207 CALL dbcsr_convert_sizes_to_offsets(row_blk_size, first_row, last_row)
2208 CALL dbcsr_convert_sizes_to_offsets(col_blk_size, first_col, last_col)
2213 DO i = 1,
SIZE(atom_pair_send)
2214 k = atom_pair_send(i)%rank + 1
2215 arow = atom_pair_send(i)%row
2216 acol = atom_pair_send(i)%col
2217 nrow = last_row(arow) - first_row(arow) + 1
2218 ncol = last_col(acol) - first_col(acol) + 1
2219 send_sizes(k) = send_sizes(k) + nrow*ncol
2220 send_pair_count(k) = send_pair_count(k) + 1
2225 DO i = 2, desc%group_size
2226 send_disps(i) = send_disps(i - 1) + send_sizes(i - 1)
2227 send_pair_disps(i) = send_pair_disps(i - 1) + send_pair_count(i - 1)
2230 ALLOCATE (send_buf_r(sum(send_sizes)))
2236 DO i = 1,
SIZE(atom_pair_recv)
2237 k = atom_pair_recv(i)%rank + 1
2238 arow = atom_pair_recv(i)%row
2239 acol = atom_pair_recv(i)%col
2240 nrow = last_row(arow) - first_row(arow) + 1
2241 ncol = last_col(acol) - first_col(acol) + 1
2242 recv_sizes(k) = recv_sizes(k) + nrow*ncol
2243 recv_pair_count(k) = recv_pair_count(k) + 1
2248 DO i = 2, desc%group_size
2249 recv_disps(i) = recv_disps(i - 1) + recv_sizes(i - 1)
2250 recv_pair_disps(i) = recv_pair_disps(i - 1) + recv_pair_count(i - 1)
2252 ALLOCATE (recv_buf_r(sum(recv_sizes)))
2271 DO l = 1, desc%group_size
2274 DO i = 1, send_pair_count(l)
2275 arow = atom_pair_send(send_pair_disps(l) + i)%row
2276 acol = atom_pair_send(send_pair_disps(l) + i)%col
2277 img = atom_pair_send(send_pair_disps(l) + i)%image
2278 nrow = last_row(arow) - first_row(arow) + 1
2279 ncol = last_col(acol) - first_col(acol) + 1
2280 pmat => pmats(img)%matrix
2281 CALL dbcsr_get_block_p(matrix=pmat, row=arow, col=acol, block=p_block, found=found)
2286 send_buf_r(send_disps(l) + send_sizes(l) + j + (k - 1)*nrow) = p_block(j, k)
2289 send_sizes(l) = send_sizes(l) + nrow*ncol
2294 IF (.NOT. scatter)
THEN
2309 CALL desc%group%alltoall(send_buf_r, send_sizes, send_disps, &
2310 recv_buf_r, recv_sizes, recv_disps)
2315 IF (.NOT. scatter)
THEN
2319 DO i = 1, send_pair_count(me)
2320 arow = atom_pair_send(send_pair_disps(me) + i)%row
2321 acol = atom_pair_send(send_pair_disps(me) + i)%col
2322 img = atom_pair_send(send_pair_disps(me) + i)%image
2323 nrow = last_row(arow) - first_row(arow) + 1
2324 ncol = last_col(acol) - first_col(acol) + 1
2325 hmat => hmats(img)%matrix
2326 pmat => pmats(img)%matrix
2327 CALL dbcsr_get_block_p(matrix=hmat, row=arow, col=acol, block=h_block, found=found)
2329 CALL dbcsr_get_block_p(matrix=pmat, row=arow, col=acol, block=p_block, found=found)
2335 h_block(j, k) = h_block(j, k) + p_block(j, k)
2344 pmat => pmats(img)%matrix
2346 nblks_guess=
SIZE(atom_pair_recv)/nthread, sizedata_guess=
SIZE(recv_buf_r)/nthread, &
2356 DO l = 1, desc%group_size
2359 DO i = 1, recv_pair_count(l)
2360 arow = atom_pair_recv(recv_pair_disps(l) + i)%row
2361 acol = atom_pair_recv(recv_pair_disps(l) + i)%col
2362 img = atom_pair_recv(recv_pair_disps(l) + i)%image
2363 nrow = last_row(arow) - first_row(arow) + 1
2364 ncol = last_col(acol) - first_col(acol) + 1
2365 pmat => pmats(img)%matrix
2367 CALL dbcsr_get_block_p(matrix=pmat, row=arow, col=acol, block=p_block, found=found)
2369 IF (
PRESENT(hmats))
THEN
2370 hmat => hmats(img)%matrix
2371 CALL dbcsr_get_block_p(matrix=hmat, row=arow, col=acol, block=h_block, found=found)
2375 IF (scatter .AND. .NOT.
ASSOCIATED(p_block))
THEN
2376 vector => recv_buf_r(recv_disps(l) + recv_sizes(l) + 1:recv_disps(l) + recv_sizes(l) + nrow*ncol)
2377 CALL dbcsr_put_block(pmat, arow, acol, block=reshape(vector, [nrow, ncol]))
2379 IF (.NOT. scatter)
THEN
2383 h_block(j, k) = h_block(j, k) + recv_buf_r(recv_disps(l) + recv_sizes(l) + j + (k - 1)*nrow)
2388 recv_sizes(l) = recv_sizes(l) + nrow*ncol
2410 pmat => pmats(img)%matrix
2416 DEALLOCATE (send_buf_r)
2417 DEALLOCATE (recv_buf_r)
2419 DEALLOCATE (send_sizes)
2420 DEALLOCATE (recv_sizes)
2421 DEALLOCATE (send_disps)
2422 DEALLOCATE (recv_disps)
2423 DEALLOCATE (send_pair_count)
2424 DEALLOCATE (recv_pair_count)
2425 DEALLOCATE (send_pair_disps)
2426 DEALLOCATE (recv_pair_disps)
2428 DEALLOCATE (first_row, last_row, first_col, last_col)
2430 CALL timestop(handle)
2438 SUBROUTINE rs_calc_offsets(pairs, nsgf, group_size, &
2439 pair_offsets, rank_offsets, rank_sizes, buffer_size)
2441 INTEGER,
DIMENSION(:),
INTENT(IN) :: nsgf
2442 INTEGER,
INTENT(IN) :: group_size
2443 INTEGER,
DIMENSION(:),
POINTER :: pair_offsets, rank_offsets, rank_sizes
2444 INTEGER,
INTENT(INOUT) :: buffer_size
2446 INTEGER :: acol, arow, i, block_size, total_size, k, prev_k
2448 IF (
ASSOCIATED(pair_offsets))
DEALLOCATE (pair_offsets)
2449 IF (
ASSOCIATED(rank_offsets))
DEALLOCATE (rank_offsets)
2450 IF (
ASSOCIATED(rank_sizes))
DEALLOCATE (rank_sizes)
2453 ALLOCATE (pair_offsets(
SIZE(pairs)))
2455 DO i = 1,
SIZE(pairs)
2456 pair_offsets(i) = total_size
2459 block_size = nsgf(arow)*nsgf(acol)
2460 total_size = total_size + block_size
2462 buffer_size = total_size
2465 ALLOCATE (rank_offsets(group_size))
2466 ALLOCATE (rank_sizes(group_size))
2469 IF (
SIZE(pairs) > 0)
THEN
2470 prev_k = pairs(1)%rank + 1
2471 DO i = 1,
SIZE(pairs)
2472 k = pairs(i)%rank + 1
2473 cpassert(k >= prev_k)
2474 IF (k > prev_k)
THEN
2475 rank_offsets(k) = pair_offsets(i)
2476 rank_sizes(prev_k) = rank_offsets(k) - rank_offsets(prev_k)
2480 rank_sizes(k) = buffer_size - rank_offsets(k)
2483 END SUBROUTINE rs_calc_offsets
2490 TYPE(
dbcsr_p_type),
DIMENSION(:),
INTENT(IN) :: src_matrices
2495 CHARACTER(LEN=*),
PARAMETER :: routinen =
'rs_scatter_matrices'
2498 REAL(kind=
dp),
DIMENSION(:),
ALLOCATABLE :: buffer_send
2500 CALL timeset(routinen, handle)
2501 ALLOCATE (buffer_send(task_list%buffer_size_send))
2504 cpassert(
ASSOCIATED(task_list%atom_pair_send))
2505 CALL rs_pack_buffer(src_matrices=src_matrices, &
2506 dest_buffer=buffer_send, &
2507 atom_pair=task_list%atom_pair_send, &
2508 pair_offsets=task_list%pair_offsets_send)
2511 CALL group%alltoall(buffer_send, task_list%rank_sizes_send, task_list%rank_offsets_send, &
2512 dest_buffer%host_buffer, &
2513 task_list%rank_sizes_recv, task_list%rank_offsets_recv)
2515 DEALLOCATE (buffer_send)
2516 CALL timestop(handle)
2526 TYPE(
dbcsr_p_type),
DIMENSION(:),
INTENT(INOUT) :: dest_matrices
2530 CHARACTER(LEN=*),
PARAMETER :: routinen =
'rs_gather_matrices'
2533 REAL(kind=
dp),
DIMENSION(:),
ALLOCATABLE :: buffer_send
2535 CALL timeset(routinen, handle)
2538 ALLOCATE (buffer_send(task_list%buffer_size_send))
2541 CALL group%alltoall(src_buffer%host_buffer, task_list%rank_sizes_recv, task_list%rank_offsets_recv, &
2542 buffer_send, task_list%rank_sizes_send, task_list%rank_offsets_send)
2545 cpassert(
ASSOCIATED(task_list%atom_pair_send))
2546 CALL rs_unpack_buffer(src_buffer=buffer_send, &
2547 dest_matrices=dest_matrices, &
2548 atom_pair=task_list%atom_pair_send, &
2549 pair_offsets=task_list%pair_offsets_send)
2551 DEALLOCATE (buffer_send)
2552 CALL timestop(handle)
2561 TYPE(
dbcsr_p_type),
DIMENSION(:),
INTENT(IN) :: src_matrices
2565 CALL rs_pack_buffer(src_matrices=src_matrices, &
2566 dest_buffer=dest_buffer%host_buffer, &
2567 atom_pair=task_list%atom_pair_recv, &
2568 pair_offsets=task_list%pair_offsets_recv)
2578 TYPE(
dbcsr_p_type),
DIMENSION(:),
INTENT(INOUT) :: dest_matrices
2581 CALL rs_unpack_buffer(src_buffer=src_buffer%host_buffer, &
2582 dest_matrices=dest_matrices, &
2583 atom_pair=task_list%atom_pair_recv, &
2584 pair_offsets=task_list%pair_offsets_recv)
2592 SUBROUTINE rs_pack_buffer(src_matrices, dest_buffer, atom_pair, pair_offsets)
2593 TYPE(
dbcsr_p_type),
DIMENSION(:),
INTENT(IN) :: src_matrices
2594 REAL(kind=
dp),
DIMENSION(:),
INTENT(INOUT) :: dest_buffer
2596 INTEGER,
DIMENSION(:),
INTENT(IN) :: pair_offsets
2598 INTEGER :: acol, arow, img, i, offset, block_size
2600 REAL(kind=
dp),
DIMENSION(:, :),
POINTER :: block
2606 DO i = 1,
SIZE(atom_pair)
2607 arow = atom_pair(i)%row
2608 acol = atom_pair(i)%col
2609 img = atom_pair(i)%image
2611 block=block, found=found)
2613 block_size =
SIZE(block)
2614 offset = pair_offsets(i)
2615 dest_buffer(offset + 1:offset + block_size) = reshape(block, shape=[block_size])
2620 END SUBROUTINE rs_pack_buffer
2626 SUBROUTINE rs_unpack_buffer(src_buffer, dest_matrices, atom_pair, pair_offsets)
2627 REAL(kind=
dp),
DIMENSION(:),
INTENT(IN) :: src_buffer
2628 TYPE(
dbcsr_p_type),
DIMENSION(:),
INTENT(INOUT) :: dest_matrices
2630 INTEGER,
DIMENSION(:),
INTENT(IN) :: pair_offsets
2632 INTEGER :: acol, arow, img, i, offset, &
2633 nrows, ncols, lock_num
2635 REAL(kind=
dp),
DIMENSION(:, :),
POINTER :: block
2636 INTEGER(kind=omp_lock_kind),
ALLOCATABLE,
DIMENSION(:) :: locks
2639 ALLOCATE (locks(10*omp_get_max_threads()))
2640 DO i = 1,
SIZE(locks)
2641 CALL omp_init_lock(locks(i))
2648 DO i = 1,
SIZE(atom_pair)
2649 arow = atom_pair(i)%row
2650 acol = atom_pair(i)%col
2651 img = atom_pair(i)%image
2653 block=block, found=found)
2655 nrows =
SIZE(block, 1)
2656 ncols =
SIZE(block, 2)
2657 offset = pair_offsets(i)
2658 lock_num =
modulo(arow,
SIZE(locks)) + 1
2660 CALL omp_set_lock(locks(lock_num))
2661 block = block + reshape(src_buffer(offset + 1:offset + nrows*ncols), shape=[nrows, ncols])
2662 CALL omp_unset_lock(locks(lock_num))
2668 DO i = 1,
SIZE(locks)
2669 CALL omp_destroy_lock(locks(i))
2673 END SUBROUTINE rs_unpack_buffer
2691 SUBROUTINE rs_find_node(rs_desc, igrid_level, n_levels, cube_center, ntasks, tasks, &
2692 lb_cube, ub_cube, added_tasks)
2695 INTEGER,
INTENT(IN) :: igrid_level, n_levels
2696 INTEGER,
DIMENSION(3),
INTENT(IN) :: cube_center
2697 INTEGER,
INTENT(INOUT) :: ntasks
2698 TYPE(
task_type),
DIMENSION(:),
POINTER :: tasks
2699 INTEGER,
DIMENSION(3),
INTENT(IN) :: lb_cube, ub_cube
2700 INTEGER,
INTENT(OUT) :: added_tasks
2702 INTEGER,
PARAMETER :: add_tasks = 1000
2703 REAL(kind=
dp),
PARAMETER :: mult_tasks = 2.0_dp
2705 INTEGER :: bit_index, coord(3), curr_tasks, dest, i, icoord(3), idest, itask, ix, iy, iz, &
2706 lb_coord(3), lb_domain(3), lbc(3), ub_coord(3), ub_domain(3), ubc(3)
2707 INTEGER :: bit_pattern
2708 LOGICAL :: dir_periodic(3)
2710 coord(1) = rs_desc%x2coord(cube_center(1))
2711 coord(2) = rs_desc%y2coord(cube_center(2))
2712 coord(3) = rs_desc%z2coord(cube_center(3))
2713 dest = rs_desc%coord2rank(coord(1), coord(2), coord(3))
2716 lbc = lb_cube + cube_center
2717 ubc = ub_cube + cube_center
2719 IF (all((rs_desc%lb_global(:, dest) - rs_desc%border) <= lbc) .AND. &
2720 all((rs_desc%ub_global(:, dest) + rs_desc%border) >= ubc))
THEN
2722 tasks(ntasks)%destination = encode_rank(dest, igrid_level, n_levels)
2723 tasks(ntasks)%dist_type = 1
2724 tasks(ntasks)%subpatch_pattern = 0
2743 IF (rs_desc%perd(i) == 1)
THEN
2744 bit_pattern = ibclr(bit_pattern, bit_index)
2745 bit_index = bit_index + 1
2746 bit_pattern = ibclr(bit_pattern, bit_index)
2747 bit_index = bit_index + 1
2750 IF (ubc(i) <= rs_desc%lb_global(i, dest) - 1 + rs_desc%border)
THEN
2751 bit_pattern = ibset(bit_pattern, bit_index)
2752 bit_index = bit_index + 1
2754 bit_pattern = ibclr(bit_pattern, bit_index)
2755 bit_index = bit_index + 1
2758 IF (lbc(i) >= rs_desc%ub_global(i, dest) + 1 - rs_desc%border)
THEN
2759 bit_pattern = ibset(bit_pattern, bit_index)
2760 bit_index = bit_index + 1
2762 bit_pattern = ibclr(bit_pattern, bit_index)
2763 bit_index = bit_index + 1
2767 tasks(ntasks)%subpatch_pattern = bit_pattern
2777 lb_domain = rs_desc%lb_global(:, dest) - rs_desc%border
2778 ub_domain = rs_desc%ub_global(:, dest) + rs_desc%border
2781 IF (rs_desc%perd(i) == 0)
THEN
2784 IF (lb_domain(i) > lbc(i))
THEN
2785 lb_coord(i) = lb_coord(i) - 1
2786 icoord =
modulo(lb_coord, rs_desc%group_dim)
2787 idest = rs_desc%coord2rank(icoord(1), icoord(2), icoord(3))
2788 lb_domain(i) = lb_domain(i) - (rs_desc%ub_global(i, idest) - rs_desc%lb_global(i, idest) + 1)
2795 IF (ub_domain(i) < ubc(i))
THEN
2796 ub_coord(i) = ub_coord(i) + 1
2797 icoord =
modulo(ub_coord, rs_desc%group_dim)
2798 idest = rs_desc%coord2rank(icoord(1), icoord(2), icoord(3))
2799 ub_domain(i) = ub_domain(i) + (rs_desc%ub_global(i, idest) - rs_desc%lb_global(i, idest) + 1)
2809 IF (ub_domain(i) - lb_domain(i) + 1 >= rs_desc%npts(i))
THEN
2810 dir_periodic(i) = .true.
2812 ub_coord(i) = rs_desc%group_dim(i) - 1
2814 dir_periodic(i) = .false.
2818 added_tasks = product(ub_coord - lb_coord + 1)
2820 ntasks = ntasks + added_tasks - 1
2821 IF (ntasks >
SIZE(tasks))
THEN
2822 curr_tasks = int((
SIZE(tasks) + add_tasks)*mult_tasks)
2825 DO iz = lb_coord(3), ub_coord(3)
2826 DO iy = lb_coord(2), ub_coord(2)
2827 DO ix = lb_coord(1), ub_coord(1)
2828 icoord =
modulo([ix, iy, iz], rs_desc%group_dim)
2829 idest = rs_desc%coord2rank(icoord(1), icoord(2), icoord(3))
2830 tasks(itask)%destination = encode_rank(idest, igrid_level, n_levels)
2831 tasks(itask)%dist_type = 2
2832 tasks(itask)%subpatch_pattern = 0
2835 IF (ix == lb_coord(1) .AND. .NOT. dir_periodic(1))
THEN
2836 tasks(itask)%subpatch_pattern = ibset(tasks(itask)%subpatch_pattern, 0)
2838 IF (ix == ub_coord(1) .AND. .NOT. dir_periodic(1))
THEN
2839 tasks(itask)%subpatch_pattern = ibset(tasks(itask)%subpatch_pattern, 1)
2841 IF (iy == lb_coord(2) .AND. .NOT. dir_periodic(2))
THEN
2842 tasks(itask)%subpatch_pattern = ibset(tasks(itask)%subpatch_pattern, 2)
2844 IF (iy == ub_coord(2) .AND. .NOT. dir_periodic(2))
THEN
2845 tasks(itask)%subpatch_pattern = ibset(tasks(itask)%subpatch_pattern, 3)
2847 IF (iz == lb_coord(3) .AND. .NOT. dir_periodic(3))
THEN
2848 tasks(itask)%subpatch_pattern = ibset(tasks(itask)%subpatch_pattern, 4)
2850 IF (iz == ub_coord(3) .AND. .NOT. dir_periodic(3))
THEN
2851 tasks(itask)%subpatch_pattern = ibset(tasks(itask)%subpatch_pattern, 5)
2859 END SUBROUTINE rs_find_node
2873 FUNCTION encode_rank(rank, grid_level, n_levels)
RESULT(encoded_int)
2875 INTEGER,
INTENT(IN) :: rank, grid_level, n_levels
2876 INTEGER :: encoded_int
2880 encoded_int = rank*n_levels + grid_level - 1
2882 END FUNCTION encode_rank
2890 FUNCTION decode_rank(encoded_int, n_levels)
RESULT(rank)
2892 INTEGER,
INTENT(IN) :: encoded_int
2893 INTEGER,
INTENT(IN) :: n_levels
2896 rank = int(encoded_int/n_levels)
2898 END FUNCTION decode_rank
2906 FUNCTION decode_level(encoded_int, n_levels)
RESULT(grid_level)
2908 INTEGER,
INTENT(IN) :: encoded_int
2909 INTEGER,
INTENT(IN) :: n_levels
2910 INTEGER :: grid_level
2912 grid_level = int(
modulo(encoded_int, n_levels)) + 1
2914 END FUNCTION decode_level
2930 PURE FUNCTION tasks_less_than(a, b)
RESULT(res)
2934 IF (a%grid_level /= b%grid_level)
THEN
2935 res = a%grid_level < b%grid_level
2937 ELSE IF (a%image /= b%image)
THEN
2938 res = a%image < b%image
2940 ELSE IF (a%iatom /= b%iatom)
THEN
2941 res = a%iatom < b%iatom
2943 ELSE IF (a%jatom /= b%jatom)
THEN
2944 res = a%jatom < b%jatom
2946 ELSE IF (a%iset /= b%iset)
THEN
2947 res = a%iset < b%iset
2949 ELSE IF (a%jset /= b%jset)
THEN
2950 res = a%jset < b%jset
2952 ELSE IF (a%ipgf /= b%ipgf)
THEN
2953 res = a%ipgf < b%ipgf
2956 res = a%jpgf < b%jpgf
2959 END FUNCTION tasks_less_than
2972 SUBROUTINE tasks_sort(arr, n, indices)
2973 INTEGER,
INTENT(IN) :: n
2974 TYPE(
task_type),
DIMENSION(1:n),
INTENT(INOUT) :: arr
2975 integer,
DIMENSION(1:n),
INTENT(INOUT) :: indices
2978 TYPE(
task_type),
ALLOCATABLE :: tmp_arr(:)
2979 INTEGER,
ALLOCATABLE :: tmp_idx(:)
2983 ALLOCATE (tmp_arr((n + 1)/2), tmp_idx((n + 1)/2))
2985 indices = [(i, i=1, n)]
2987 CALL tasks_sort_low(arr(1:n), indices, tmp_arr, tmp_idx)
2989 DEALLOCATE (tmp_arr, tmp_idx)
2990 ELSE IF (n > 0)
THEN
2994 END SUBROUTINE tasks_sort
3006 RECURSIVE SUBROUTINE tasks_sort_low(arr, indices, tmp_arr, tmp_idx)
3007 TYPE(
task_type),
DIMENSION(:),
INTENT(INOUT) :: arr
3008 INTEGER,
DIMENSION(size(arr)),
INTENT(INOUT) :: indices
3009 TYPE(
task_type),
DIMENSION((size(arr) + 1)/2),
INTENT(INOUT) :: tmp_arr
3010 INTEGER,
DIMENSION((size(arr) + 1)/2),
INTENT(INOUT) :: tmp_idx
3012 INTEGER :: t, m, i, j, k
3019 IF (
size(arr) <= 7)
THEN
3020 DO j =
size(arr) - 1, 1, -1
3023 IF (tasks_less_than(arr(i + 1), arr(i)))
THEN
3030 indices(i) = indices(i + 1)
3035 IF (.NOT. swapped)
EXIT
3041 m = (
size(arr) + 1)/2
3042 CALL tasks_sort_low(arr(1:m), indices(1:m), tmp_arr, tmp_idx)
3043 CALL tasks_sort_low(arr(m + 1:), indices(m + 1:), tmp_arr, tmp_idx)
3047 IF (tasks_less_than(arr(m + 1), arr(m)))
THEN
3050 tmp_arr(1:m) = arr(1:m)
3051 tmp_idx(1:m) = indices(1:m)
3056 DO WHILE (i <= m .and. j <=
size(arr) - m)
3057 IF (tasks_less_than(arr(m + j), tmp_arr(i)))
THEN
3059 indices(k) = indices(m + j)
3063 indices(k) = tmp_idx(i)
3073 indices(k) = tmp_idx(i)
3080 END SUBROUTINE tasks_sort_low
3090 PURE FUNCTION atom_pair_less_than(a, b)
RESULT(res)
3094 IF (a%rank /= b%rank)
THEN
3095 res = a%rank < b%rank
3097 ELSE IF (a%row /= b%row)
THEN
3100 ELSE IF (a%col /= b%col)
THEN
3104 res = a%image < b%image
3107 END FUNCTION atom_pair_less_than
3120 SUBROUTINE atom_pair_sort(arr, n, indices)
3121 INTEGER,
INTENT(IN) :: n
3123 integer,
DIMENSION(1:n),
INTENT(INOUT) :: indices
3127 INTEGER,
ALLOCATABLE :: tmp_idx(:)
3131 ALLOCATE (tmp_arr((n + 1)/2), tmp_idx((n + 1)/2))
3133 indices = [(i, i=1, n)]
3135 CALL atom_pair_sort_low(arr(1:n), indices, tmp_arr, tmp_idx)
3137 DEALLOCATE (tmp_arr, tmp_idx)
3138 ELSE IF (n > 0)
THEN
3142 END SUBROUTINE atom_pair_sort
3154 RECURSIVE SUBROUTINE atom_pair_sort_low(arr, indices, tmp_arr, tmp_idx)
3156 INTEGER,
DIMENSION(size(arr)),
INTENT(INOUT) :: indices
3157 TYPE(
atom_pair_type),
DIMENSION((size(arr) + 1)/2),
INTENT(INOUT) :: tmp_arr
3158 INTEGER,
DIMENSION((size(arr) + 1)/2),
INTENT(INOUT) :: tmp_idx
3160 INTEGER :: t, m, i, j, k
3167 IF (
size(arr) <= 7)
THEN
3168 DO j =
size(arr) - 1, 1, -1
3171 IF (atom_pair_less_than(arr(i + 1), arr(i)))
THEN
3178 indices(i) = indices(i + 1)
3183 IF (.NOT. swapped)
EXIT
3189 m = (
size(arr) + 1)/2
3190 CALL atom_pair_sort_low(arr(1:m), indices(1:m), tmp_arr, tmp_idx)
3191 CALL atom_pair_sort_low(arr(m + 1:), indices(m + 1:), tmp_arr, tmp_idx)
3195 IF (atom_pair_less_than(arr(m + 1), arr(m)))
THEN
3198 tmp_arr(1:m) = arr(1:m)
3199 tmp_idx(1:m) = indices(1:m)
3204 DO WHILE (i <= m .and. j <=
size(arr) - m)
3205 IF (atom_pair_less_than(arr(m + j), tmp_arr(i)))
THEN
3207 indices(k) = indices(m + j)
3211 indices(k) = tmp_idx(i)
3221 indices(k) = tmp_idx(i)
3228 END SUBROUTINE atom_pair_sort_low
static GRID_HOST_DEVICE int modulo(int a, int m)
Equivalent of Fortran's MODULO, which always return a positive number. https://gcc....
All kind of helpful little routines.
real(kind=dp) function, public exp_radius_very_extended(la_min, la_max, lb_min, lb_max, pab, o1, o2, ra, rb, rp, zetp, eps, prefactor, cutoff, epsabs)
computes the radius of the Gaussian outside of which it is smaller than eps
subroutine, public get_gto_basis_set(gto_basis_set, name, aliases, norm_type, kind_radius, ncgf, nset, nsgf, cgf_symbol, sgf_symbol, norm_cgf, set_radius, lmax, lmin, lx, ly, lz, m, ncgf_set, npgf, nsgf_set, nshell, cphi, pgf_radius, sphi, scon, zet, first_cgf, first_sgf, l, last_cgf, last_sgf, n, gcc, maxco, maxl, maxpgf, maxsgf_set, maxshell, maxso, nco_sum, npgf_sum, nshell_sum, maxder, short_kind_radius, npgf_seg_sum, ccon)
...
Handles all functions related to the CELL.
Defines control structures, which contain the parameters and the settings for the DFT-based calculati...
subroutine, public dbcsr_get_block_p(matrix, row, col, block, found, row_size, col_size)
...
subroutine, public dbcsr_get_info(matrix, nblkrows_total, nblkcols_total, nfullrows_total, nfullcols_total, nblkrows_local, nblkcols_local, nfullrows_local, nfullcols_local, my_prow, my_pcol, local_rows, local_cols, proc_row_dist, proc_col_dist, row_blk_size, col_blk_size, row_blk_offset, col_blk_offset, distribution, name, matrix_type, group)
...
subroutine, public dbcsr_work_create(matrix, nblks_guess, sizedata_guess, n, work_mutable)
...
subroutine, public dbcsr_finalize(matrix)
...
subroutine, public dbcsr_put_block(matrix, row, col, block, summation)
...
for a given dr()/dh(r) this will provide the bounds to be used if one wants to go over a sphere-subre...
subroutine, public compute_cube_center(cube_center, rs_desc, zeta, zetb, ra, rab)
unifies the computation of the cube center, so that differences in implementation,...
subroutine, public return_cube(info, radius, lb_cube, ub_cube, sphere_bounds)
...
subroutine, public return_cube_nonortho(info, radius, lb, ub, rp)
...
integer function, public gaussian_gridlevel(gridlevel_info, exponent)
...
Fortran API for the grid package, which is written in C.
subroutine, public grid_create_task_list(ntasks, natoms, nkinds, nblocks, block_offsets, atom_positions, atom_kinds, basis_sets, level_list, iatom_list, jatom_list, iset_list, jset_list, ipgf_list, jpgf_list, border_mask_list, block_num_list, radius_list, rab_list, rs_grids, task_list)
Allocates a task list which can be passed to grid_collocate_task_list.
subroutine, public grid_create_basis_set(nset, nsgf, maxco, maxpgf, lmin, lmax, npgf, nsgf_set, first_sgf, sphi, zet, basis_set)
Allocates a basis set which can be passed to grid_create_task_list.
Defines the basic variable types.
integer, parameter, public int_8
integer, parameter, public dp
integer, parameter, public default_string_length
Types and basic routines needed for a kpoint calculation.
subroutine, public get_kpoint_info(kpoint, kp_scheme, nkp_grid, kp_shift, symmetry, verbose, full_grid, use_real_wfn, eps_geo, parallel_group_size, kp_range, nkp, xkp, wkp, para_env, blacs_env_all, para_env_kp, para_env_inter_kp, blacs_env, kp_env, kp_aux_env, mpools, iogrp, nkp_groups, kp_dist, cell_to_index, index_to_cell, sab_nl, sab_nl_nosym, inversion_symmetry_only, symmetry_backend, symmetry_reduction_method, gamma_centered, lattice_fft)
Retrieve information from a kpoint environment.
An array-based list which grows on demand. When the internal array is full, a new array of twice the ...
Utility routines for the memory handling.
Interface to the message passing library MPI.
Fortran API for the offload package, which is written in C.
subroutine, public offload_create_buffer(length, buffer)
Allocates a buffer of given length, ie. number of elements.
Define methods related to particle_type.
subroutine, public get_particle_set(particle_set, qs_kind_set, first_sgf, last_sgf, nsgf, nmao, basis, ncgf)
Get the components of a particle set.
Define the data structure for the particle information.
container for various plainwaves related things
subroutine, public pw_env_get(pw_env, pw_pools, cube_info, gridlevel_info, auxbas_pw_pool, auxbas_grid, auxbas_rs_desc, auxbas_rs_grid, rs_descs, rs_grids, xc_pw_pool, vdw_pw_pool, poisson_env, interp_section)
returns the various attributes of the pw env
Define the quickstep kind type and their sub types.
subroutine, public get_qs_kind(qs_kind, basis_set, basis_type, ncgf, nsgf, all_potential, tnadd_potential, gth_potential, sgp_potential, upf_potential, cneo_potential, se_parameter, dftb_parameter, xtb_parameter, dftb3_param, zatom, zeff, elec_conf, mao, lmax_dftb, alpha_core_charge, ccore_charge, core_charge, core_charge_radius, paw_proj_set, paw_atom, hard_radius, hard0_radius, max_rad_local, covalent_radius, vdw_radius, gpw_type_forced, harmonics, max_iso_not0, max_s_harm, grid_atom, ngrid_ang, ngrid_rad, lmax_rho0, dft_plus_u_atom, l_of_dft_plus_u, n_of_dft_plus_u, u_minus_j, hund_j, u_of_dft_plus_u, j_of_dft_plus_u, alpha_of_dft_plus_u, beta_of_dft_plus_u, j0_of_dft_plus_u, occupation_of_dft_plus_u, dispersion, bs_occupation, magnetization, no_optimize, addel, laddel, naddel, orbitals, max_scf, eps_scf, smear, u_ramping, u_minus_j_target, eps_u_ramping, proj_shell_charge, lr_atom, do_mtlr, u_j_loop, ao_coef, init_u_ramping_each_scf, reltmat, ghost, monovalent, floating, name, element_symbol, pao_basis_size, pao_model_file, pao_potentials, pao_descriptors, nelec)
Get attributes of an atomic kind.
subroutine, public get_ks_env(ks_env, v_hartree_rspace, s_mstruct_changed, rho_changed, exc_accint, potential_changed, forces_up_to_date, complex_ks, matrix_h, matrix_h_im, matrix_ks, matrix_ks_im, matrix_vxc, kinetic, matrix_s, matrix_s_ri_aux, matrix_w, matrix_p_mp2, matrix_p_mp2_admm, matrix_vhxc, matrix_h_kp, matrix_h_im_kp, matrix_ks_kp, matrix_vxc_kp, kinetic_kp, matrix_s_kp, matrix_w_kp, matrix_s_ri_aux_kp, matrix_ks_im_kp, rho, rho_xc, vppl, xcint_weights, rho_core, rho_nlcc, rho_nlcc_g, vee, neighbor_list_id, sab_orb, sab_all, sac_ae, sac_ppl, sac_lri, sap_ppnl, sap_oce, sab_lrc, sab_se, sab_xtbe, sab_tbe, sab_core, sab_xb, sab_xtb_pp, sab_xtb_nonbond, sab_vdw, sab_scp, sab_almo, sab_kp, sab_kp_nosym, sab_cneo, task_list, task_list_soft, kpoints, do_kpoints, atomic_kind_set, qs_kind_set, cell, cell_ref, use_ref_cell, particle_set, energy, force, local_particles, local_molecules, molecule_kind_set, molecule_set, subsys, cp_subsys, virial, results, atprop, nkind, natom, dft_control, dbcsr_dist, distribution_2d, pw_env, para_env, blacs_env, nelectron_total, nelectron_spin)
...
Define the neighbor list data types and the corresponding functionality.
subroutine, public rs_grid_create(rs, desc)
...
pure integer function, public rs_grid_locate_rank(rs_desc, rank_in, shift)
returns the 1D rank of the task which is a cartesian shift away from 1D rank rank_in only possible if...
pure subroutine, public rs_grid_reorder_ranks(desc, real2virtual)
Defines a new ordering of ranks on this realspace grid, recalculating the data bounds and reallocatin...
subroutine, public rs_grid_release(rs_grid)
releases the given rs grid (see doc/ReferenceCounting.html)
generate the tasks lists used by collocate and integrate routines
subroutine, public rs_copy_to_matrices(src_buffer, dest_matrices, task_list)
Copies from buffer into DBCSR matrics, replaces rs_gather_matrix for non-distributed grids.
subroutine, public rs_scatter_matrices(src_matrices, dest_buffer, task_list, group)
Scatters dbcsr matrix blocks and receives them into a buffer as needed before collocation.
subroutine, public rs_distribute_matrix(rs_descs, pmats, atom_pair_send, atom_pair_recv, nimages, scatter, hmats)
redistributes the matrix so that it can be used in realspace operations i.e. according to the task li...
subroutine, public distribute_tasks(rs_descs, ntasks, natoms, tasks, atom_pair_send, atom_pair_recv, symmetric, reorder_rs_grid_ranks, skip_load_balance_distributed)
Assembles tasks to be performed on local grid.
subroutine, public rs_gather_matrices(src_buffer, dest_matrices, task_list, group)
Gather the dbcsr matrix blocks and receives them into a buffer as needed after integration.
subroutine, public task_list_inner_loop(tasks, ntasks, curr_tasks, rs_descs, dft_control, cube_info, gridlevel_info, cindex, iatom, jatom, rpgfa, rpgfb, zeta, zetb, kind_radius_b, set_radius_a, set_radius_b, ra, rab, la_max, la_min, lb_max, lb_min, npgfa, npgfb, nseta, nsetb)
...
subroutine, public generate_qs_task_list(ks_env, task_list, basis_type, reorder_rs_grid_ranks, skip_load_balance_distributed, pw_env_external, sab_orb_external, ext_kpoints)
...
subroutine, public rs_copy_to_buffer(src_matrices, dest_buffer, task_list)
Copies the DBCSR blocks into buffer, replaces rs_scatter_matrix for non-distributed grids.
subroutine, public serialize_task(task, serialized_task)
Serialize a task into an integer array. Used for MPI communication.
subroutine, public deserialize_task(task, serialized_task)
De-serialize a task from an integer array. Used for MPI communication.
subroutine, public reallocate_tasks(tasks, new_size)
Grow an array of tasks while preserving the existing entries.
integer, parameter, public task_size_in_int8
All kind of helpful little routines.
Type defining parameters related to the simulation cell.
Contains information about kpoints.
contained for different pw related things
Provides all information about a quickstep kind.
calculation environment to calculate the ks matrix, holds all the needed vars. assumes that the core ...