mirror of
https://github.com/sfilippone/amg4psblas.git
synced 2026-10-06 22:55:12 +00:00
Compare commits
122
Commits
fixmatch
...
TestFerdous
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f523d0195f | ||
|
|
494b8b925f | ||
|
|
73e5d49913 | ||
|
|
a612cea167 | ||
|
|
ebe9b45177 | ||
|
|
32994c7ce8 | ||
|
|
d59c9e6c0a | ||
|
|
0d624df346 | ||
|
|
6414d3aef3 | ||
|
|
a259e8ab53 | ||
|
|
500403dbda | ||
|
|
066c1a5e62 | ||
|
|
1ab166b38b | ||
|
|
5efee20041 | ||
|
|
aa45e2fe93 | ||
|
|
e328f3969c | ||
|
|
9d1a416f99 | ||
|
|
9b065602a8 | ||
|
|
abf258e2e8 | ||
|
|
cdf92ea2b2 | ||
|
|
22d9baf296 | ||
|
|
44f174a571 | ||
|
|
3e945c75b4 | ||
|
|
a71fe82752 | ||
|
|
4f07a70ed1 | ||
|
|
cb660e044d | ||
|
|
d24c8c2d46 | ||
|
|
9ab54adf3f | ||
|
|
71d4cdc319 | ||
|
|
1374f21ba8 | ||
|
|
a9bb6b26fa | ||
|
|
561cadee0f | ||
|
|
5ca78fb871 | ||
|
|
f17082b337 | ||
|
|
1ea1be33ba | ||
|
|
47c6f4f2f8 | ||
|
|
dc1675766f | ||
|
|
ccac816f52 | ||
|
|
c7e8193514 | ||
|
|
36bd3a51a2 | ||
|
|
32777cc15c | ||
|
|
64c23f93f8 | ||
|
|
d19443052d | ||
|
|
df1e4a4616 | ||
|
|
3de1e607eb | ||
|
|
9b13aef1ce | ||
|
|
6dcae6d0c1 | ||
|
|
63b7602d3a | ||
|
|
b66de7f25c | ||
|
|
46047b2202 | ||
|
|
7cfe198d0f | ||
|
|
1aca17cd44 | ||
|
|
ea040ae5ee | ||
|
|
7741abd45d | ||
|
|
b5e52d31f5 | ||
|
|
deab695294 | ||
|
|
a54f084ffb | ||
|
|
bf0532867d | ||
|
|
2044c5c8eb | ||
|
|
f38f3cf09a | ||
|
|
6fd571ecb2 | ||
|
|
bf35c1659b | ||
|
|
b2230a6d6d | ||
|
|
6c20cd7819 | ||
|
|
f921aa47c4 | ||
|
|
532701031e | ||
|
|
b079d71f30 | ||
|
|
e2ca97ca47 | ||
|
|
5bc4f2a080 | ||
|
|
2c8dc2ffdd | ||
|
|
f3d7b3ab5e | ||
|
|
766ef320c2 | ||
|
|
002239f5b6 | ||
|
|
70b7c4db55 | ||
|
|
2cac21b345 | ||
|
|
6180f29f39 | ||
|
|
b4bfdd83e5 | ||
|
|
1140669ea7 | ||
|
|
919e2a2918 | ||
|
|
baffff3d93 | ||
|
|
25a603debe | ||
|
|
a20f0d47e7 | ||
|
|
76e04ee997 | ||
|
|
0a8debe43a | ||
|
|
8f6dc5fac2 | ||
|
|
7d40fde21d | ||
|
|
1760afbe97 | ||
|
|
60f90804d5 | ||
|
|
edea0caa63 | ||
|
|
bd8495794a | ||
|
|
0f223ca269 | ||
|
|
c4e584a8d2 | ||
|
|
9004e623fe | ||
|
|
653074995a | ||
|
|
09928af671 | ||
|
|
ba2c6aa721 | ||
|
|
c65304f255 | ||
|
|
cf67905788 | ||
|
|
19325ed390 | ||
|
|
c7b55a452e | ||
|
|
c80d2d8b1d | ||
|
|
8cf193ec5b | ||
|
|
8cf8c0fc7b | ||
|
|
b11a7db667 | ||
|
|
d700e4c7e4 | ||
|
|
43d3659a9e | ||
|
|
82e7e7d7e7 | ||
|
|
5993b79749 | ||
|
|
0523053c49 | ||
|
|
52eaa06f9e | ||
|
|
71156dbb3c | ||
|
|
9dab2a8c7c | ||
|
|
f2cf59c276 | ||
|
|
fa2f8f53d5 | ||
|
|
9609ea262f | ||
|
|
eb30e8be90 | ||
|
|
9779eeec9f | ||
|
|
6f419a2210 | ||
|
|
c4801635b8 | ||
|
|
c18e5720f5 | ||
|
|
9ab599367b | ||
|
|
2396123b44 |
+1
-1
@@ -75,7 +75,7 @@ CDEFINES=$(AMGCDEFINES)
|
||||
AMGFDEFINES=@AMGFDEFINES@ $(PSBFDEFINES)
|
||||
FDEFINES=$(AMGFDEFINES)
|
||||
|
||||
CXXDEFINES=@AMGCXXDEFINES@
|
||||
CXXDEFINES=@AMGCXXDEFINES@ $(PSBCXXDEFINES)
|
||||
|
||||
@COMPILERULES@
|
||||
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
|
||||
AMG4PSBLAS
|
||||
Algebraic Multigrid Package based on PSBLAS (Parallel Sparse BLAS version 3.8)
|
||||
|
||||
|
||||
+5
-2
@@ -16,7 +16,8 @@ DMODOBJS=amg_d_prec_type.o \
|
||||
amg_d_dec_aggregator_mod.o amg_d_symdec_aggregator_mod.o \
|
||||
amg_d_ainv_solver.o amg_d_base_ainv_mod.o \
|
||||
amg_d_invk_solver.o amg_d_invt_solver.o amg_d_krm_solver.o \
|
||||
amg_d_matchboxp_mod.o amg_d_parmatch_aggregator_mod.o
|
||||
amg_d_matchboxp_mod.o amg_d_parmatch_aggregator_mod.o \
|
||||
amg_d_newmatch_aggregator_mod.o amg_d_decmatch_mod.o
|
||||
|
||||
SMODOBJS=amg_s_prec_type.o amg_s_ilu_fact_mod.o \
|
||||
amg_s_inner_mod.o amg_s_ilu_solver.o amg_s_diag_solver.o amg_s_jac_smoother.o amg_s_as_smoother.o \
|
||||
@@ -116,7 +117,7 @@ amg_c_prec_type.o: amg_c_onelev_mod.o
|
||||
amg_z_prec_type.o: amg_z_onelev_mod.o
|
||||
|
||||
amg_s_onelev_mod.o: amg_s_base_smoother_mod.o amg_s_dec_aggregator_mod.o amg_s_parmatch_aggregator_mod.o
|
||||
amg_d_onelev_mod.o: amg_d_base_smoother_mod.o amg_d_dec_aggregator_mod.o amg_d_parmatch_aggregator_mod.o
|
||||
amg_d_onelev_mod.o: amg_d_base_smoother_mod.o amg_d_dec_aggregator_mod.o amg_d_parmatch_aggregator_mod.o amg_d_newmatch_aggregator_mod.o
|
||||
amg_c_onelev_mod.o: amg_c_base_smoother_mod.o amg_c_dec_aggregator_mod.o
|
||||
amg_z_onelev_mod.o: amg_z_base_smoother_mod.o amg_z_dec_aggregator_mod.o
|
||||
|
||||
@@ -129,6 +130,8 @@ amg_d_base_aggregator_mod.o: amg_base_prec_type.o
|
||||
amg_d_parmatch_aggregator_mod.o amg_d_dec_aggregator_mod.o: amg_d_base_aggregator_mod.o
|
||||
amg_d_hybrid_aggregator_mod.o amg_d_symdec_aggregator_mod.o: amg_d_dec_aggregator_mod.o
|
||||
amg_d_parmatch_aggregator_mod.o: amg_d_matchboxp_mod.o
|
||||
amg_d_newmatch_aggregator_mod.o: amg_d_base_aggregator_mod.o
|
||||
amg_d_newmatch_aggregator_mod.o: amg_d_decmatch_mod.o
|
||||
|
||||
amg_c_base_aggregator_mod.o: amg_base_prec_type.o
|
||||
amg_c_parmatch_aggregator_mod.o amg_c_dec_aggregator_mod.o: amg_c_base_aggregator_mod.o
|
||||
|
||||
@@ -275,7 +275,8 @@ module amg_base_prec_type
|
||||
integer(psb_ipk_), parameter :: amg_sym_dec_aggr_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_ext_aggr_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_coupled_aggr_ = 3
|
||||
integer(psb_ipk_), parameter :: amg_max_par_aggr_alg_ = amg_coupled_aggr_
|
||||
integer(psb_ipk_), parameter :: amg_newmtc_aggr_ = 4
|
||||
integer(psb_ipk_), parameter :: amg_max_par_aggr_alg_ = amg_newmtc_aggr_
|
||||
!
|
||||
! Legal values for entry: amg_aggr_type_
|
||||
!
|
||||
@@ -283,6 +284,7 @@ module amg_base_prec_type
|
||||
integer(psb_ipk_), parameter :: amg_soc1_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_soc2_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_matchboxp_ = 3
|
||||
integer(psb_ipk_), parameter :: amg_newmatch_ = 4
|
||||
!
|
||||
! Legal values for entry: amg_aggr_prol_
|
||||
!
|
||||
@@ -371,13 +373,14 @@ module amg_base_prec_type
|
||||
character(len=15), parameter, private :: &
|
||||
& matrix_names(0:1)=(/'distributed ','replicated '/)
|
||||
character(len=18), parameter, private :: &
|
||||
& aggr_type_names(0:3)=(/'None ',&
|
||||
& aggr_type_names(0:4)=(/'None ',&
|
||||
& 'SOC measure 1 ', 'SOC Measure 2 ',&
|
||||
& 'Parallel Matching '/)
|
||||
& 'Parallel Matching ','Decoupled Matching'/)
|
||||
character(len=18), parameter, private :: &
|
||||
& par_aggr_alg_names(0:3)=(/&
|
||||
& par_aggr_alg_names(0:4)=(/&
|
||||
& 'decoupled aggr. ', 'sym. dec. aggr. ',&
|
||||
& 'user defined aggr.', 'coupled aggr. '/)
|
||||
& 'user defined aggr.', 'coupled aggr. ',&
|
||||
& 'new matching aggr.'/)
|
||||
character(len=18), parameter, private :: &
|
||||
& ord_names(0:1)=(/'Natural ordering ','Desc. degree ord. '/)
|
||||
character(len=6), parameter, private :: &
|
||||
@@ -516,6 +519,10 @@ contains
|
||||
val = amg_soc2_
|
||||
case('SOC1')
|
||||
val = amg_soc1_
|
||||
case('NEWMATCH')
|
||||
val = amg_newmatch_
|
||||
case('NEWMTC')
|
||||
val = amg_newmtc_aggr_
|
||||
case('MATCHBOXP','PARMATCH')
|
||||
val = amg_matchboxp_
|
||||
case('COUPLED','COUP')
|
||||
|
||||
@@ -0,0 +1,529 @@
|
||||
!
|
||||
!
|
||||
! AMG4PSBLAS version 1.0
|
||||
! Algebraic Multigrid Package
|
||||
! based on PSBLAS (Parallel Sparse BLAS version 3.7)
|
||||
!
|
||||
! (C) Copyright 2021
|
||||
!
|
||||
! Salvatore Filippone
|
||||
! Pasqua D'Ambra
|
||||
! Fabio Durastante
|
||||
!
|
||||
! Redistribution and use in source and binary forms, with or without
|
||||
! modification, are permitted provided that the following conditions
|
||||
! are met:
|
||||
! 1. Redistributions of source code must retain the above copyright
|
||||
! notice, this list of conditions and the following disclaimer.
|
||||
! 2. Redistributions in binary form must reproduce the above copyright
|
||||
! notice, this list of conditions, and the following disclaimer in the
|
||||
! documentation and/or other materials provided with the distribution.
|
||||
! 3. The name of the AMG4PSBLAS group or the names of its contributors may
|
||||
! not be used to endorse or promote products derived from this
|
||||
! software without specific written permission.
|
||||
!
|
||||
! THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
! ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
|
||||
! TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
! PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AMG4PSBLAS GROUP OR ITS CONTRIBUTORS
|
||||
! BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
! CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
! SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
! INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
! CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
module amg_d_decmatch_mod
|
||||
|
||||
use iso_c_binding
|
||||
use psb_base_cbind_mod
|
||||
|
||||
interface new_Match_If
|
||||
function dnew_Match_If(ipar,matching,lambda,nr, irp, ja, val, diag, w, mate) &
|
||||
& bind(c,name="dnew_Match_If") result(res)
|
||||
use iso_c_binding
|
||||
import :: psb_c_ipk_, psb_c_lpk_, psb_c_mpk_, psb_c_epk_
|
||||
implicit none
|
||||
|
||||
integer(psb_c_ipk_) :: res
|
||||
integer(psb_c_ipk_), value :: nr,ipar,matching
|
||||
real(c_double), value :: lambda
|
||||
type(c_ptr), value :: irp, ja, mate
|
||||
type(c_ptr), value :: val, diag, w
|
||||
end function dnew_Match_If
|
||||
end interface new_Match_If
|
||||
|
||||
interface amg_build_decmatch
|
||||
module procedure amg_dbuild_decmatch
|
||||
end interface amg_build_decmatch
|
||||
|
||||
logical, parameter, private :: print_statistics=.false.
|
||||
contains
|
||||
|
||||
subroutine amg_ddecmatch_build_prol(w,a,desc_a,ilaggr,nlaggr,prol,info,&
|
||||
& symmetrize,reproducible,display_inp, display_out, print_out, &
|
||||
& parallel, matching,lambda)
|
||||
use psb_base_mod
|
||||
use psb_util_mod
|
||||
use iso_c_binding
|
||||
implicit none
|
||||
real(psb_dpk_), allocatable, intent(inout) :: w(:)
|
||||
type(psb_dspmat_type), intent(inout) :: a
|
||||
type(psb_desc_type) :: desc_a
|
||||
integer(psb_lpk_), allocatable, intent(out) :: ilaggr(:)
|
||||
integer(psb_lpk_), allocatable, intent(out) :: nlaggr(:)
|
||||
type(psb_ldspmat_type), intent(out) :: prol
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
logical, optional, intent(in) :: display_inp, display_out, reproducible
|
||||
logical, optional, intent(in) :: symmetrize, print_out, parallel
|
||||
integer(psb_ipk_), optional, intent(in) :: matching
|
||||
real(psb_dpk_), optional, intent(in) :: lambda
|
||||
|
||||
!
|
||||
!
|
||||
type(psb_ctxt_type) :: ictxt
|
||||
integer(psb_ipk_) :: iam, np, iown
|
||||
integer(psb_ipk_) :: nr, nc, sweep, nzl, ncsave, nct, idx
|
||||
integer(psb_lpk_) :: i, k, kg, idxg, ntaggr, naggrm1, naggrp1, &
|
||||
& ip, nlpairs, nlsingl, nunmatched, lnr
|
||||
real(psb_dpk_) :: wk, widx, wmax, nrmagg
|
||||
real(psb_dpk_), allocatable :: wtemp(:)
|
||||
integer(psb_ipk_), allocatable :: mate(:)
|
||||
integer(psb_lpk_), allocatable :: ilv(:)
|
||||
integer(psb_ipk_), save :: cnt=1
|
||||
character(len=256) :: aname
|
||||
type(psb_ld_coo_sparse_mat) :: tmpcoo
|
||||
logical :: display_out_, print_out_, reproducible_, parallel_
|
||||
integer(psb_ipk_) :: matching_
|
||||
real(psb_dpk_) :: lambda_
|
||||
logical, parameter :: dump=.false., debug=.false., dump_mate=.false., &
|
||||
& debug_ilaggr=.false., debug_sync=.false.
|
||||
integer(psb_ipk_), save :: idx_bldmtc=-1, idx_phase1=-1, idx_phase2=-1, idx_phase3=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
ictxt = desc_a%get_ctxt()
|
||||
call psb_info(ictxt,iam,np)
|
||||
|
||||
if ((do_timings).and.(idx_phase1==-1)) &
|
||||
& idx_phase1 = psb_get_timer_idx("MBP_BLDP: phase1 ")
|
||||
if ((do_timings).and.(idx_bldmtc==-1)) &
|
||||
& idx_bldmtc = psb_get_timer_idx("MBP_BLDP: buil_matching")
|
||||
if ((do_timings).and.(idx_phase2==-1)) &
|
||||
& idx_phase2 = psb_get_timer_idx("MBP_BLDP: phase2 ")
|
||||
if ((do_timings).and.(idx_phase3==-1)) &
|
||||
& idx_phase3 = psb_get_timer_idx("MBP_BLDP: phase3 ")
|
||||
|
||||
if (do_timings) call psb_tic(idx_phase1)
|
||||
|
||||
if (present(display_out)) then
|
||||
display_out_ = display_out
|
||||
else
|
||||
display_out_ = .false.
|
||||
end if
|
||||
if (present(print_out)) then
|
||||
print_out_ = print_out
|
||||
else
|
||||
print_out_ = .false.
|
||||
end if
|
||||
if (present(reproducible)) then
|
||||
reproducible_ = reproducible
|
||||
else
|
||||
reproducible_ = .false.
|
||||
end if
|
||||
|
||||
if (present(parallel)) then
|
||||
parallel_ = parallel
|
||||
else
|
||||
parallel_ = .true.
|
||||
end if
|
||||
|
||||
if (present(matching)) then
|
||||
matching_ = matching
|
||||
else
|
||||
matching_ = 2
|
||||
end if
|
||||
|
||||
if (present(lambda)) then
|
||||
lambda_ = lambda
|
||||
else
|
||||
lambda_ = 2.0
|
||||
end if
|
||||
|
||||
allocate(nlaggr(0:np-1),stat=info)
|
||||
if (info /= 0) then
|
||||
return
|
||||
end if
|
||||
|
||||
nlaggr = 0
|
||||
ilv = [(i,i=1,desc_a%get_local_cols())]
|
||||
call desc_a%l2gip(ilv,info,owned=.false.)
|
||||
|
||||
call psb_geall(ilaggr,desc_a,info)
|
||||
ilaggr = -1
|
||||
call psb_geasb(ilaggr,desc_a,info)
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
if (size(w) < nc) then
|
||||
call psb_realloc(nc,w,info)
|
||||
end if
|
||||
call psb_halo(w,desc_a,info)
|
||||
|
||||
if (debug) write(0,*) iam,' buildprol into buildmatching:',&
|
||||
& nr, nc
|
||||
if (debug_sync) then
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == 0) write(0,*)' buildprol into buildmatching:',&
|
||||
& nr, nc
|
||||
end if
|
||||
if (do_timings) call psb_toc(idx_phase1)
|
||||
if (do_timings) call psb_tic(idx_bldmtc)
|
||||
call amg_dbuild_decmatch(parallel_,matching_,lambda_,w,a,desc_a,mate,info)
|
||||
if (do_timings) call psb_toc(idx_bldmtc)
|
||||
if (debug) write(0,*) iam,' buildprol from buildmatching:',&
|
||||
& info
|
||||
if (debug_sync) then
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == 0) write(0,*)' out from buildmatching:', info
|
||||
end if
|
||||
|
||||
if (info == 0) then
|
||||
if (do_timings) call psb_tic(idx_phase2)
|
||||
if (debug_sync) then
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == 0) write(0,*)' Into building the tentative prol:'
|
||||
end if
|
||||
|
||||
call psb_geall(wtemp,desc_a,info)
|
||||
wtemp = dzero
|
||||
call psb_geasb(wtemp,desc_a,info)
|
||||
|
||||
nlaggr(iam) = 0
|
||||
nlpairs = 0
|
||||
nlsingl = 0
|
||||
nunmatched = 0
|
||||
!
|
||||
! First sweep
|
||||
! On return from build_matching, mate has been converted to local numbering,
|
||||
! so assigning to idx is OK.
|
||||
!
|
||||
do k=1, nr
|
||||
idx = mate(k)
|
||||
!
|
||||
! Figure out an allocation of aggregates to processes
|
||||
!
|
||||
if (idx < 0) then
|
||||
!
|
||||
! Unmatched vertex, potential singleton.
|
||||
!
|
||||
nunmatched = nunmatched + 1
|
||||
if (abs(w(k))<epsilon(nrmagg)) then
|
||||
! Keep it unaggregated
|
||||
wtemp(k) = dzero
|
||||
else
|
||||
! Create a singleton aggregate
|
||||
nlaggr(iam) = nlaggr(iam) + 1
|
||||
ilaggr(k) = nlaggr(iam)
|
||||
wtemp(k) = w(k)/abs(w(k))
|
||||
nlsingl = nlsingl + 1
|
||||
end if
|
||||
!!$ write(0,*) k,mate(k),ilaggr(k),' negative match ',abs(w(k)), epsilon(nrmagg)
|
||||
else if (idx > nc) then
|
||||
write(0,*) 'Impossible: mate(k) > nc'
|
||||
cycle
|
||||
else
|
||||
|
||||
if (ilaggr(k) == -1) then
|
||||
|
||||
wk = w(k)
|
||||
widx = w(idx)
|
||||
wmax = max(abs(wk),abs(widx))
|
||||
nrmagg = wmax*sqrt((wk/wmax)**2+(widx/wmax)**2)
|
||||
if (nrmagg > epsilon(nrmagg)) then
|
||||
if (idx <= nr) then
|
||||
if (ilaggr(idx) == -1) then
|
||||
! Now, if both vertices are local, the aggregate is local
|
||||
! (kinda obvious).
|
||||
nlaggr(iam) = nlaggr(iam) + 1
|
||||
ilaggr(k) = nlaggr(iam)
|
||||
ilaggr(idx) = nlaggr(iam)
|
||||
wtemp(k) = w(k)/nrmagg
|
||||
wtemp(idx) = w(idx)/nrmagg
|
||||
end if
|
||||
nlpairs = nlpairs+1
|
||||
else
|
||||
write(0,*) 'Really? mate(k) > nr? ',mate(k),nr
|
||||
end if
|
||||
else
|
||||
if (abs(w(k))<epsilon(nrmagg)) then
|
||||
! Keep it unaggregated
|
||||
wtemp(k) = dzero
|
||||
else
|
||||
! Create a singleton aggregate
|
||||
nlaggr(iam) = nlaggr(iam) + 1
|
||||
ilaggr(k) = nlaggr(iam)
|
||||
wtemp(k) = w(k)/abs(w(k))
|
||||
nlsingl = nlsingl + 1
|
||||
end if
|
||||
end if
|
||||
end if
|
||||
end if
|
||||
end do
|
||||
if (do_timings) call psb_toc(idx_phase2)
|
||||
if (do_timings) call psb_tic(idx_phase3)
|
||||
|
||||
! Ok, now compute offsets, gather halo and fix non-local
|
||||
! aggregates (those where ilaggr == -2)
|
||||
call psb_sum(ictxt,nlaggr)
|
||||
ntaggr = sum(nlaggr(0:np-1))
|
||||
naggrm1 = sum(nlaggr(0:iam-1))
|
||||
naggrp1 = sum(nlaggr(0:iam))
|
||||
!
|
||||
! Shift all indices already assigned (i.e. >0)
|
||||
!
|
||||
do k=1,nr
|
||||
if (ilaggr(k) > 0) then
|
||||
ilaggr(k) = ilaggr(k) + naggrm1
|
||||
!!$ else
|
||||
!!$ write(0,*) 'Leftover ILAGGR',k,ilaggr(k),mate(k),abs(w(k)),epsilon(nrmagg)
|
||||
end if
|
||||
end do
|
||||
call psb_halo(ilaggr,desc_a,info)
|
||||
call psb_halo(wtemp,desc_a,info)
|
||||
! Cleanup as yet unmarked entries
|
||||
do k=1,nr
|
||||
if (ilaggr(k) == -2) then
|
||||
idx = mate(k)
|
||||
if (idx > nr) then
|
||||
i = ilaggr(idx)
|
||||
if (i > 0) then
|
||||
ilaggr(k) = i
|
||||
else
|
||||
write(0,*) 'Error : unresolved (paired) index ',k,idx,i,nr,nc, ilv(k),ilv(idx)
|
||||
end if
|
||||
else
|
||||
write(0,*) 'Error : unresolved (paired) index ',k,idx,i,nr,nc, ilv(k),ilv(idx)
|
||||
end if
|
||||
end if
|
||||
if (ilaggr(k) <0) then
|
||||
write(0,*) 'Decmatch: Funny number: ',k,ilv(k),ilaggr(k),wtemp(k)
|
||||
end if
|
||||
end do
|
||||
if (debug_sync) then
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == 0) write(0,*)' Done building the tentative prol:'
|
||||
end if
|
||||
|
||||
|
||||
if (dump_mate) then
|
||||
block
|
||||
integer(psb_lpk_), allocatable :: glaggr(:)
|
||||
write(aname,'(a,i3.3,a,i3.3,a)') 'mateg-',cnt,'-p',iam,'.mtx'
|
||||
open(20,file=aname)
|
||||
write(20,'(a,I3,a)') '% sparse vector on process ',iam,' '
|
||||
do k=1, nr
|
||||
write(20,'(3(I8,1X))') ilv(k),ilv(mate(k))
|
||||
end do
|
||||
close(20)
|
||||
write(aname,'(a,i3.3,a,i3.3,a)') 'nloc-',cnt,'-p',iam,'.mtx'
|
||||
open(20,file=aname)
|
||||
write(20,'(a,I3,a)') '% sparse vector on process ',iam,' '
|
||||
write(20,'(a,I12,a)') 'nlpairs ',nlpairs
|
||||
write(20,'(a,I12,a)') 'nlsingl ',nlsingl
|
||||
write(20,'(a,I12,a)') 'nlaggr(iam) ',nlaggr(iam)
|
||||
close(20)
|
||||
write(aname,'(a,i3.3,a,i3.3,a,i3.3,a)') 'ilaggr-',cnt,'-i',iam,'-p',np,'.mtx'
|
||||
open(20,file=aname)
|
||||
write(20,'(a,I3,a)') '% sparse vector on process ',iam,' '
|
||||
do k=1, nr
|
||||
write(20,'(3(I8,1X))') ilv(k),ilaggr(k)
|
||||
end do
|
||||
close(20)
|
||||
|
||||
write(aname,'(a,i3.3,a,i3.3,a)') 'glaggr-',cnt,'-p',np,'.mtx'
|
||||
call psb_gather(glaggr,ilaggr,desc_a,info,root=izero)
|
||||
if (iam==0) call mm_array_write(glaggr,'Aggregates ',info,filename=aname)
|
||||
|
||||
cnt=cnt+1
|
||||
end block
|
||||
end if
|
||||
block
|
||||
integer(psb_lpk_) :: v(3)
|
||||
v(1) = nunmatched
|
||||
v(2) = nlsingl
|
||||
v(3) = nlpairs
|
||||
call psb_sum(ictxt,v)
|
||||
nunmatched = v(1)
|
||||
nlsingl = v(2)
|
||||
nlpairs = v(3)
|
||||
|
||||
end block
|
||||
if (print_statistics) then
|
||||
if (iam == 0) then
|
||||
write(0,*) 'Matching statistics: Unmatched nodes ',&
|
||||
& nunmatched,' Singletons:',nlsingl,' Pairs:',nlpairs
|
||||
end if
|
||||
end if
|
||||
|
||||
if (display_out_) then
|
||||
block
|
||||
integer(psb_ipk_) :: idx
|
||||
!
|
||||
! And finally print out
|
||||
!
|
||||
do i=0,np-1
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == i) then
|
||||
write(0,*) 'Process ', iam,' hosts aggregates: (',naggrm1+1,' : ',naggrp1-1,')'
|
||||
do k=1, nr
|
||||
idx = mate(k)
|
||||
kg = ilv(k)
|
||||
if (idx >0) then
|
||||
idxg = ilv(idx)
|
||||
else
|
||||
idxg = -1
|
||||
end if
|
||||
if (idx < 0) then
|
||||
write(0,*) kg,': singleton (',kg,' ( Proc',iam,') ) into aggregate => ', ilaggr(k)
|
||||
else if (idx <= nr) then
|
||||
write(0,*) kg,': paired with (',idxg,' ( Proc',iam,') ) into aggregate => ', ilaggr(k)
|
||||
else
|
||||
call desc_a%indxmap%qry_halo_owner(idx,iown,info)
|
||||
write(0,*) kg,': paired with (',idxg,' ( Proc',iown,') ) into aggregate => ', ilaggr(k)
|
||||
end if
|
||||
end do
|
||||
flush(0)
|
||||
end if
|
||||
end do
|
||||
end block
|
||||
end if
|
||||
|
||||
! Dirty trick: allocate tmpcoo with local
|
||||
! number of aggregates, then change to ntaggr.
|
||||
! Just to make sure the allocation is not global
|
||||
lnr = nr
|
||||
call tmpcoo%allocate(lnr,nlaggr(iam),lnr)
|
||||
k = 0
|
||||
do i=1,nr
|
||||
!
|
||||
! Note: at this point, a value ilaggr(i)<=0
|
||||
! tags an unaggregated row, and it has to be
|
||||
! left alone (i.e.: it should stay at fine level only)
|
||||
!
|
||||
if (ilaggr(i)>0) then
|
||||
k = k + 1
|
||||
tmpcoo%val(k) = wtemp(i)
|
||||
tmpcoo%ia(k) = i
|
||||
tmpcoo%ja(k) = ilaggr(i)
|
||||
end if
|
||||
end do
|
||||
call tmpcoo%set_nzeros(k)
|
||||
call tmpcoo%set_dupl(psb_dupl_add_)
|
||||
call tmpcoo%set_sorted() ! This is now in row-major
|
||||
|
||||
if (display_out_) then
|
||||
call psb_barrier(ictxt)
|
||||
flush(0)
|
||||
|
||||
if (iam == 0) write(0,*) 'Prolongator: '
|
||||
flush(0)
|
||||
do i=0,np-1
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == i) then
|
||||
do k=1, nr
|
||||
write(0,*) ilv(tmpcoo%ia(k)),tmpcoo%ja(k), tmpcoo%val(k)
|
||||
end do
|
||||
flush(0)
|
||||
end if
|
||||
end do
|
||||
end if
|
||||
|
||||
call prol%mv_from(tmpcoo)
|
||||
if (do_timings) call psb_toc(idx_phase3)
|
||||
|
||||
if (print_out_) then
|
||||
write(aname,'(a,i3.3,a)') 'prol-g-',iam,'.mtx'
|
||||
call prol%print(fname=aname,head='Test ',ivr=ilv)
|
||||
write(aname,'(a,i3.3,a)') 'prol-',iam,'.mtx'
|
||||
call prol%print(fname=aname,head='Test ')
|
||||
end if
|
||||
|
||||
else
|
||||
write(0,*) iam,' : error from Matching: ',info
|
||||
end if
|
||||
|
||||
end subroutine amg_ddecmatch_build_prol
|
||||
|
||||
subroutine amg_dbuild_decmatch(parallel,matching,lambda,w,a,desc_a,mate,info)
|
||||
use psb_base_mod
|
||||
use psb_util_mod
|
||||
use iso_c_binding
|
||||
implicit none
|
||||
logical, intent(in) :: parallel
|
||||
integer(psb_ipk_), intent(in) :: matching
|
||||
real(psb_dpk_), intent(in) :: lambda
|
||||
real(psb_dpk_), target :: w(:)
|
||||
type(psb_dspmat_type), intent(inout) :: a
|
||||
type(psb_desc_type) :: desc_a
|
||||
integer(psb_ipk_), allocatable, intent(out), target :: mate(:)
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
type(psb_d_csr_sparse_mat), target :: tcsr
|
||||
real(psb_dpk_), allocatable, target :: diag(:)
|
||||
real(psb_dpk_) :: ph0t,ph1t,ph2t
|
||||
type(psb_ctxt_type) :: ictxt
|
||||
integer(psb_ipk_) :: iam, np
|
||||
integer(psb_ipk_) :: nr, nc, nz, i, nunmatch, ipar
|
||||
integer(psb_ipk_), save :: cnt=2
|
||||
logical, parameter :: debug=.false., dump_ahat=.false., debug_sync=.false.
|
||||
logical, parameter :: old_style=.false., sort_minp=.true.
|
||||
character(len=40) :: name='build_matching', fname
|
||||
integer(psb_ipk_), save :: idx_cmboxp=-1, idx_bldahat=-1, idx_phase2=-1, idx_phase3=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
ictxt = desc_a%get_ctxt()
|
||||
call psb_info(ictxt,iam,np)
|
||||
|
||||
nr = a%get_nrows()
|
||||
call a%cp_to(tcsr)
|
||||
call psb_realloc(nr,mate,info)
|
||||
diag = a%get_diag(info)
|
||||
if (parallel) then
|
||||
ipar = 2
|
||||
else
|
||||
ipar = 1
|
||||
end if
|
||||
!
|
||||
! Now call matching!
|
||||
!
|
||||
if (debug) write(0,*) iam,' buildmatching into NewMatch:'
|
||||
if (do_timings) call psb_tic(idx_cmboxp)
|
||||
info = dnew_Match_If(ipar,matching,lambda,nr,c_loc(tcsr%irp),c_loc(tcsr%ja),&
|
||||
& c_loc(tcsr%val),c_loc(diag),c_loc(w),c_loc(mate))
|
||||
if (do_timings) call psb_toc(idx_cmboxp)
|
||||
if (debug) write(0,*) iam,' buildmatching from NewMatch:', info
|
||||
if (debug_sync) then
|
||||
call psb_max(ictxt,info)
|
||||
if (iam == 0) write(0,*)' done NewMatch', info
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_phase3)
|
||||
nunmatch = count(mate(1:nr)<=0)
|
||||
! call psb_sum(ictxt,nunmatch)
|
||||
!if (nunmatch /= 0) write(0,*) iam,' Unmatched nodes local imbalance ',nunmatch
|
||||
! if (count(mate(1:nr)<0) /= nunmatch) write(0,*) iam,' Matching results ?',&
|
||||
! & nunmatch, count(mate(1:nr)<0)
|
||||
if (debug_sync) then
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == 0) write(0,*)' done build_matching '
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_phase3)
|
||||
return
|
||||
|
||||
9999 continue
|
||||
call psb_error(ictxt)
|
||||
|
||||
end subroutine amg_dbuild_decmatch
|
||||
|
||||
end module amg_d_decmatch_mod
|
||||
@@ -143,9 +143,10 @@ contains
|
||||
type(psb_ld_coo_sparse_mat) :: tmpcoo
|
||||
logical :: display_out_, print_out_, reproducible_
|
||||
logical, parameter :: dump=.false., debug=.false., dump_mate=.false., &
|
||||
& debug_ilaggr=.false., debug_sync=.false.
|
||||
& debug_ilaggr=.false., debug_sync=.false., debug_mate=.false.
|
||||
integer(psb_ipk_), save :: idx_bldmtc=-1, idx_phase1=-1, idx_phase2=-1, idx_phase3=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
integer, parameter :: ilaggr_neginit=-1, ilaggr_nonlocal=-2
|
||||
|
||||
ictxt = desc_a%get_ctxt()
|
||||
call psb_info(ictxt,iam,np)
|
||||
@@ -187,7 +188,7 @@ contains
|
||||
call desc_a%l2gip(ilv,info,owned=.false.)
|
||||
|
||||
call psb_geall(ilaggr,desc_a,info)
|
||||
ilaggr = -1
|
||||
ilaggr = ilaggr_neginit
|
||||
call psb_geasb(ilaggr,desc_a,info)
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -213,7 +214,20 @@ contains
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == 0) write(0,*)' out from buildmatching:', info
|
||||
end if
|
||||
|
||||
if (debug_mate) then
|
||||
block
|
||||
integer(psb_lpk_), allocatable :: ckmate(:)
|
||||
allocate(ckmate(nr))
|
||||
ckmate(1:nr) = mate(1:nr)
|
||||
call psb_msort(ckmate(1:nr))
|
||||
do i=1,nr-1
|
||||
if ((ckmate(i)>0) .and. (ckmate(i) == ckmate(i+1))) then
|
||||
write(0,*) iam,' Duplicate mate entry at',i,' :',ckmate(i)
|
||||
end if
|
||||
end do
|
||||
end block
|
||||
end if
|
||||
|
||||
if (info == 0) then
|
||||
if (do_timings) call psb_tic(idx_phase2)
|
||||
if (debug_sync) then
|
||||
@@ -259,7 +273,7 @@ contains
|
||||
cycle
|
||||
else
|
||||
|
||||
if (ilaggr(k) == -1) then
|
||||
if (ilaggr(k) == ilaggr_neginit) then
|
||||
|
||||
wk = w(k)
|
||||
widx = w(idx)
|
||||
@@ -267,7 +281,7 @@ contains
|
||||
nrmagg = wmax*sqrt((wk/wmax)**2+(widx/wmax)**2)
|
||||
if (nrmagg > epsilon(nrmagg)) then
|
||||
if (idx <= nr) then
|
||||
if (ilaggr(idx) == -1) then
|
||||
if (ilaggr(idx) == ilaggr_neginit) then
|
||||
! Now, if both vertices are local, the aggregate is local
|
||||
! (kinda obvious).
|
||||
nlaggr(iam) = nlaggr(iam) + 1
|
||||
@@ -275,6 +289,9 @@ contains
|
||||
ilaggr(idx) = nlaggr(iam)
|
||||
wtemp(k) = w(k)/nrmagg
|
||||
wtemp(idx) = w(idx)/nrmagg
|
||||
else
|
||||
write(0,*) iam,' Inconsistent mate? ',k,mate(k),idx,&
|
||||
&mate(idx),ilaggr(idx)
|
||||
end if
|
||||
nlpairs = nlpairs+1
|
||||
else if (idx <= nc) then
|
||||
@@ -294,7 +311,7 @@ contains
|
||||
ilaggr(k) = nlaggr(iam)
|
||||
nlpairs = nlpairs+1
|
||||
else
|
||||
ilaggr(k) = -2
|
||||
ilaggr(k) = ilaggr_nonlocal
|
||||
end if
|
||||
else
|
||||
! Use a statistically unbiased tie-breaking rule,
|
||||
@@ -309,7 +326,7 @@ contains
|
||||
ilaggr(k) = nlaggr(iam)
|
||||
nlpairs = nlpairs+1
|
||||
else
|
||||
ilaggr(k) = -2
|
||||
ilaggr(k) = ilaggr_nonlocal
|
||||
end if
|
||||
end if
|
||||
end if
|
||||
@@ -325,6 +342,12 @@ contains
|
||||
nlsingl = nlsingl + 1
|
||||
end if
|
||||
end if
|
||||
if (ilaggr(k) == ilaggr_neginit) then
|
||||
write(0,*) iam,' Error: no update to ',k,mate(k),&
|
||||
& abs(w(k)),nrmagg,epsilon(nrmagg),wtemp(k)
|
||||
end if
|
||||
else
|
||||
if (ilaggr(k)<0) write(0,*) 'Strange? ',k,ilaggr(k)
|
||||
end if
|
||||
end if
|
||||
end do
|
||||
@@ -332,7 +355,7 @@ contains
|
||||
if (do_timings) call psb_tic(idx_phase3)
|
||||
|
||||
! Ok, now compute offsets, gather halo and fix non-local
|
||||
! aggregates (those where ilaggr == -2)
|
||||
! aggregates (those where ilaggr == ilaggr_nonlocal)
|
||||
call psb_sum(ictxt,nlaggr)
|
||||
ntaggr = sum(nlaggr(0:np-1))
|
||||
naggrm1 = sum(nlaggr(0:iam-1))
|
||||
@@ -347,7 +370,7 @@ contains
|
||||
call psb_halo(wtemp,desc_a,info)
|
||||
! Cleanup as yet unmarked entries
|
||||
do k=1,nr
|
||||
if (ilaggr(k) == -2) then
|
||||
if (ilaggr(k) == ilaggr_nonlocal) then
|
||||
idx = mate(k)
|
||||
if (idx > nr) then
|
||||
i = ilaggr(idx)
|
||||
@@ -359,9 +382,14 @@ contains
|
||||
else
|
||||
write(0,*) 'Error : unresolved (paired) index ',k,idx,i,nr,nc, ilv(k),ilv(idx)
|
||||
end if
|
||||
end if
|
||||
if (ilaggr(k) <0) then
|
||||
write(0,*) 'Matchboxp: Funny number: ',k,ilv(k),ilaggr(k),wtemp(k)
|
||||
else if (ilaggr(k) <0) then
|
||||
write(0,*) iam,'Matchboxp: Funny number: ',k,ilv(k),ilaggr(k),wtemp(k)
|
||||
write(0,*) iam,' : : ',nr,nc,mate(k)
|
||||
if (mate(k) <= nr) then
|
||||
write(0,*) iam,' : : ',ilaggr(mate(k)),mate(mate(k)),&
|
||||
& ilv(k),ilv(mate(k)), ilv(mate(mate(k))),ilaggr(mate(mate(k)))
|
||||
end if
|
||||
flush(0)
|
||||
end if
|
||||
end do
|
||||
if (debug_sync) then
|
||||
@@ -414,7 +442,7 @@ contains
|
||||
|
||||
end block
|
||||
if (iam == 0) then
|
||||
write(0,*) 'Matching statistics: Unmatched nodes ',&
|
||||
write(0,*) iam,'Matching statistics: Unmatched nodes ',&
|
||||
& nunmatched,' Singletons:',nlsingl,' Pairs:',nlpairs
|
||||
end if
|
||||
|
||||
|
||||
@@ -0,0 +1,585 @@
|
||||
!
|
||||
!
|
||||
! The aggregator object hosts the aggregation method for building
|
||||
! the multilevel hierarchy. This variant is based on the hybrid method
|
||||
! presented in
|
||||
!
|
||||
!
|
||||
! sm - class(amg_T_base_smoother_type), allocatable
|
||||
! The current level preconditioner (aka smoother).
|
||||
! parms - type(amg_RTml_parms)
|
||||
! The parameters defining the multilevel strategy.
|
||||
! ac - The local part of the current-level matrix, built by
|
||||
! coarsening the previous-level matrix.
|
||||
! desc_ac - type(psb_desc_type).
|
||||
! The communication descriptor associated to the matrix
|
||||
! stored in ac.
|
||||
! base_a - type(psb_Tspmat_type), pointer.
|
||||
! Pointer (really a pointer!) to the local part of the current
|
||||
! matrix (so we have a unified treatment of residuals).
|
||||
! We need this to avoid passing explicitly the current matrix
|
||||
! to the routine which applies the preconditioner.
|
||||
! base_desc - type(psb_desc_type), pointer.
|
||||
! Pointer to the communication descriptor associated to the
|
||||
! matrix pointed by base_a.
|
||||
! map - Stores the maps (restriction and prolongation) between the
|
||||
! vector spaces associated to the index spaces of the previous
|
||||
! and current levels.
|
||||
!
|
||||
! Methods:
|
||||
! Most methods follow the encapsulation hierarchy: they take whatever action
|
||||
! is appropriate for the current object, then call the corresponding method for
|
||||
! the contained object.
|
||||
! As an example: the descr() method prints out a description of the
|
||||
! level. It starts by invoking the descr() method of the parms object,
|
||||
! then calls the descr() method of the smoother object.
|
||||
!
|
||||
! descr - Prints a description of the object.
|
||||
! default - Set default values
|
||||
! dump - Dump to file object contents
|
||||
! set - Sets various parameters; when a request is unknown
|
||||
! it is passed to the smoother object for further processing.
|
||||
! check - Sanity checks.
|
||||
! sizeof - Total memory occupation in bytes
|
||||
! get_nzeros - Number of nonzeros
|
||||
!
|
||||
!
|
||||
|
||||
module amg_d_newmatch_aggregator_mod
|
||||
use amg_d_base_aggregator_mod
|
||||
use iso_c_binding
|
||||
|
||||
type, bind(c):: nwm_Vector
|
||||
type(c_ptr) :: data
|
||||
integer(c_int) :: size
|
||||
integer(c_int) :: owns_data
|
||||
end type nwm_Vector
|
||||
|
||||
type, bind(c):: nwm_CSRMatrix
|
||||
type(c_ptr) :: i
|
||||
type(c_ptr) :: j
|
||||
integer(c_int) :: num_rows
|
||||
integer(c_int) :: num_cols
|
||||
integer(c_int) :: num_nonzeros
|
||||
integer(c_int) :: owns_data
|
||||
type(c_ptr) :: data
|
||||
end type nwm_CSRMatrix
|
||||
|
||||
type, extends(amg_d_base_aggregator_type) :: amg_d_newmatch_aggregator_type
|
||||
integer(psb_ipk_) :: matching_alg
|
||||
integer(psb_ipk_) :: n_sweeps
|
||||
!
|
||||
! Note: the BootCMatch kernel we invoke overwrites
|
||||
! the W argument with its update. Hence, copy it in w_nxt
|
||||
! before passing it to the matching
|
||||
!
|
||||
integer(psb_ipk_) :: orig_aggr_size
|
||||
integer(psb_ipk_) :: jacobi_sweeps
|
||||
real(psb_dpk_), allocatable :: w(:), w_nxt(:)
|
||||
type(psb_dspmat_type), allocatable :: prol, restr
|
||||
type(psb_dspmat_type), allocatable :: ac, base_a, rwa
|
||||
type(psb_desc_type), allocatable :: desc_ac, desc_ax, base_desc, rwdesc
|
||||
type(nwm_Vector) :: w_c_nxt
|
||||
integer(psb_ipk_) :: max_csize
|
||||
integer(psb_ipk_) :: max_nlevels
|
||||
real(psb_dpk_) :: lambda
|
||||
logical :: reproducible_matching = .false.
|
||||
logical :: need_symmetrize = .false.
|
||||
logical :: unsmoothed_hierarchy = .true.
|
||||
logical :: parallel_matching = .true.
|
||||
contains
|
||||
procedure, pass(ag) :: bld_tprol => amg_d_newmatch_aggregator_build_tprol
|
||||
procedure, pass(ag) :: csetc => amg_d_newmatch_aggr_csetc
|
||||
procedure, pass(ag) :: csetr => amg_d_newmatch_aggr_csetr
|
||||
procedure, pass(ag) :: cseti => d_newmatch_aggr_cseti
|
||||
procedure, pass(ag) :: default => d_newmatch_aggr_set_default
|
||||
procedure, pass(ag) :: mat_asb => amg_d_newmatch_aggregator_mat_asb
|
||||
procedure, pass(ag) :: mat_bld => amg_d_newmatch_aggregator_mat_bld
|
||||
procedure, pass(ag) :: inner_mat_asb => amg_d_newmatch_aggregator_inner_mat_asb
|
||||
procedure, pass(ag) :: update_next => d_newmatch_aggregator_update_next
|
||||
procedure, pass(ag) :: bld_wnxt => d_newmatch_bld_wnxt
|
||||
procedure, pass(ag) :: bld_default_w => d_bld_default_w
|
||||
procedure, pass(ag) :: set_c_default_w => d_set_default_nwm_w
|
||||
procedure, pass(ag) :: descr => d_newmatch_aggregator_descr
|
||||
procedure, pass(ag) :: clone => d_newmatch_aggregator_clone
|
||||
procedure, pass(ag) :: free => d_newmatch_aggregator_free
|
||||
procedure, nopass :: fmt => d_newmatch_aggregator_fmt
|
||||
end type amg_d_newmatch_aggregator_type
|
||||
|
||||
|
||||
interface
|
||||
subroutine amg_d_newmatch_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
& a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
import :: amg_d_newmatch_aggregator_type, psb_desc_type, &
|
||||
& psb_dspmat_type, psb_ldspmat_type, psb_dpk_, &
|
||||
& psb_ipk_, psb_lpk_, psb_epk_, amg_dml_parms, amg_daggr_data
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(amg_daggr_data), intent(in) :: ag_data
|
||||
type(psb_dspmat_type), intent(inout) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), allocatable, intent(out) :: ilaggr(:),nlaggr(:)
|
||||
type(psb_ldspmat_type), intent(out) :: t_prol
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_newmatch_aggregator_build_tprol
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_newmatch_aggregator_mat_bld(ag,parms,a,desc_a,ilaggr,nlaggr,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: amg_d_newmatch_aggregator_type, psb_desc_type, &
|
||||
& psb_dspmat_type, psb_ldspmat_type, psb_dpk_, &
|
||||
& psb_ipk_, psb_lpk_, psb_epk_, amg_dml_parms
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
type(psb_dspmat_type), intent(out) :: op_prol,ac,op_restr
|
||||
type(psb_ldspmat_type), intent(inout) :: t_prol
|
||||
type(psb_desc_type), intent(inout) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_newmatch_aggregator_mat_bld
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_newmatch_aggregator_mat_asb(ag,parms,a,desc_a,&
|
||||
& ac,desc_ac, op_prol,op_restr,info)
|
||||
import :: amg_d_newmatch_aggregator_type, psb_desc_type, &
|
||||
& psb_dspmat_type, psb_ldspmat_type, psb_dpk_, &
|
||||
& psb_ipk_, psb_lpk_, psb_epk_, amg_dml_parms
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol,ac,op_restr
|
||||
type(psb_desc_type), intent(inout) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_newmatch_aggregator_mat_asb
|
||||
end interface
|
||||
|
||||
|
||||
interface
|
||||
subroutine amg_d_newmatch_map_to_tprol(desc_a,ilaggr,nlaggr,valaggr, op_prol,info)
|
||||
import :: amg_d_newmatch_aggregator_type, psb_desc_type, &
|
||||
& psb_dspmat_type, psb_ldspmat_type, psb_dpk_, &
|
||||
& psb_ipk_, psb_lpk_, psb_epk_, amg_dml_parms
|
||||
implicit none
|
||||
type(psb_desc_type), intent(in) :: desc_a
|
||||
integer(psb_lpk_), allocatable, intent(inout) :: ilaggr(:),nlaggr(:)
|
||||
real(psb_dpk_), allocatable, intent(inout) :: valaggr(:)
|
||||
type(psb_ldspmat_type), intent(out) :: op_prol
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_newmatch_map_to_tprol
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_daggrmat_unsmth_spmm_asb(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,op_prol,op_restr,info)
|
||||
import :: amg_d_newmatch_aggregator_type, psb_desc_type, &
|
||||
& psb_dspmat_type, psb_ldspmat_type, psb_dpk_, &
|
||||
& psb_ipk_, psb_lpk_, psb_epk_, amg_dml_parms
|
||||
implicit none
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(in) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_ldspmat_type), intent(inout) :: op_prol
|
||||
type(psb_ldspmat_type), intent(out) :: ac,op_restr
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_daggrmat_unsmth_spmm_asb
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_newmatch_aggregator_inner_mat_asb(ag,parms,a,desc_a,&
|
||||
& ac,desc_ac, op_prol,op_restr,info)
|
||||
import :: amg_d_newmatch_aggregator_type, psb_desc_type, psb_dspmat_type,&
|
||||
& psb_ldspmat_type, psb_dpk_, psb_ipk_, psb_lpk_, amg_dml_parms, amg_daggr_data
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(in) :: desc_a
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol,op_restr
|
||||
type(psb_dspmat_type), intent(inout) :: ac
|
||||
type(psb_desc_type), intent(inout) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_newmatch_aggregator_inner_mat_asb
|
||||
end interface
|
||||
|
||||
|
||||
!!$ interface
|
||||
!!$ subroutine amg_d_newmatch_unsmth_spmm_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ & ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
!!$ import :: amg_d_newmatch_aggregator_type, psb_desc_type, &
|
||||
!!$ & psb_dspmat_type, psb_ldspmat_type, psb_dpk_, &
|
||||
!!$ & psb_ipk_, psb_lpk_, psb_epk_, amg_dml_parms
|
||||
!!$
|
||||
!!$ implicit none
|
||||
!!$
|
||||
!!$ ! Arguments
|
||||
!!$ type(psb_dspmat_type), intent(in) :: a
|
||||
!!$ type(psb_desc_type), intent(in) :: desc_a
|
||||
!!$ integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
!!$ type(amg_dml_parms), intent(inout) :: parms
|
||||
!!$ type(psb_ldspmat_type), intent(inout) :: t_prol
|
||||
!!$ type(psb_dspmat_type), intent(inout) :: op_prol,ac,op_restr
|
||||
!!$ type(psb_desc_type), intent(inout) :: desc_ac
|
||||
!!$ integer(psb_ipk_), intent(out) :: info
|
||||
!!$ end subroutine amg_d_newmatch_unsmth_spmm_bld
|
||||
!!$ end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_newmatch_spmm_bld_ov(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: amg_d_newmatch_aggregator_type, psb_desc_type, psb_dspmat_type,&
|
||||
& psb_ldspmat_type, psb_dpk_, psb_ipk_, psb_lpk_, amg_dml_parms, amg_daggr_data
|
||||
implicit none
|
||||
type(psb_dspmat_type), intent(inout) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_ldspmat_type), intent(inout) :: t_prol
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol,ac, op_restr
|
||||
type(psb_desc_type), intent(out) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_newmatch_spmm_bld_ov
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_newmatch_spmm_bld_inner(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: amg_d_newmatch_aggregator_type, psb_desc_type, psb_dspmat_type,&
|
||||
& psb_ldspmat_type, psb_dpk_, psb_ipk_, psb_lpk_, amg_dml_parms, amg_daggr_data,&
|
||||
& psb_d_csr_sparse_mat, psb_ld_csr_sparse_mat
|
||||
implicit none
|
||||
type(psb_d_csr_sparse_mat), intent(inout) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_ldspmat_type), intent(inout) :: t_prol
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol,ac, op_restr
|
||||
type(psb_desc_type), intent(out) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_newmatch_spmm_bld_inner
|
||||
end interface
|
||||
|
||||
private :: is_legal_malg, is_legal_csize, is_legal_nsweeps, is_legal_nlevels
|
||||
|
||||
|
||||
contains
|
||||
|
||||
subroutine d_bld_default_w(ag,nr)
|
||||
use psb_realloc_mod
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
integer(psb_ipk_), intent(in) :: nr
|
||||
integer(psb_ipk_) :: info
|
||||
call psb_realloc(nr,ag%w,info)
|
||||
if (info /= psb_success_) return
|
||||
ag%w = done
|
||||
call ag%set_c_default_w()
|
||||
end subroutine d_bld_default_w
|
||||
|
||||
subroutine d_set_default_nwm_w(ag)
|
||||
use psb_realloc_mod
|
||||
use iso_c_binding
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
integer(psb_ipk_) :: info
|
||||
|
||||
call psb_safe_ab_cpy(ag%w,ag%w_nxt,info)
|
||||
ag%w_c_nxt%size = psb_size(ag%w_nxt)
|
||||
ag%w_c_nxt%owns_data = 0
|
||||
if (ag%w_c_nxt%size > 0) call set_cloc(ag%w_nxt, ag%w_c_nxt)
|
||||
|
||||
end subroutine d_set_default_nwm_w
|
||||
|
||||
subroutine set_cloc(vect,w_c_nxt)
|
||||
use iso_c_binding
|
||||
real(psb_dpk_), target :: vect(:)
|
||||
type(nwm_Vector) :: w_c_nxt
|
||||
|
||||
w_c_nxt%data = c_loc(vect)
|
||||
end subroutine set_cloc
|
||||
|
||||
|
||||
subroutine d_newmatch_bld_wnxt(ag,ilaggr,valaggr,nx)
|
||||
use psb_realloc_mod
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
integer(psb_lpk_), intent(in) :: ilaggr(:)
|
||||
real(psb_dpk_), intent(in) :: valaggr(:)
|
||||
integer(psb_ipk_), intent(in) :: nx
|
||||
|
||||
integer(psb_ipk_) :: info,i,j
|
||||
|
||||
! The vector was already fixed in the call to Newmatch.
|
||||
call psb_realloc(nx,ag%w_nxt,info)
|
||||
|
||||
end subroutine d_newmatch_bld_wnxt
|
||||
|
||||
function d_newmatch_aggregator_fmt() result(val)
|
||||
implicit none
|
||||
character(len=32) :: val
|
||||
|
||||
val = "new matching aggregation"
|
||||
end function d_newmatch_aggregator_fmt
|
||||
|
||||
subroutine d_newmatch_aggregator_descr(ag,parms,iout,info,prefix)
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), intent(in) :: ag
|
||||
type(amg_dml_parms), intent(in) :: parms
|
||||
integer(psb_ipk_), intent(in) :: iout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
character(len=*), intent(in), optional :: prefix
|
||||
character(1024) :: prefix_
|
||||
if (present(prefix)) then
|
||||
prefix_ = prefix
|
||||
else
|
||||
prefix_ = ""
|
||||
end if
|
||||
|
||||
write(iout,*) trim(prefix_),' ','NewMatch Aggregator'
|
||||
write(iout,*) trim(prefix_),' ',' Number of Matching sweeps: ',ag%n_sweeps
|
||||
write(iout,*) trim(prefix_),' ',' Matching algorithm : ',ag%matching_alg
|
||||
write(iout,*) trim(prefix_),' ','Aggregator object type: ',ag%fmt()
|
||||
call parms%mldescr(iout,info)
|
||||
|
||||
return
|
||||
end subroutine d_newmatch_aggregator_descr
|
||||
|
||||
function is_legal_malg(alg) result(val)
|
||||
logical :: val
|
||||
integer(psb_ipk_) :: alg
|
||||
|
||||
val = ((0<=alg).and.(alg<=2))
|
||||
end function is_legal_malg
|
||||
|
||||
function is_legal_csize(csize) result(val)
|
||||
logical :: val
|
||||
integer(psb_ipk_) :: csize
|
||||
|
||||
val = ((-1==csize).or.(csize >0))
|
||||
end function is_legal_csize
|
||||
|
||||
function is_legal_nsweeps(nsw) result(val)
|
||||
logical :: val
|
||||
integer(psb_ipk_) :: nsw
|
||||
|
||||
val = (1<=nsw)
|
||||
end function is_legal_nsweeps
|
||||
|
||||
function is_legal_nlevels(nlv) result(val)
|
||||
logical :: val
|
||||
integer(psb_ipk_) :: nlv
|
||||
|
||||
val = (1<=nlv)
|
||||
end function is_legal_nlevels
|
||||
|
||||
subroutine d_newmatch_aggregator_update_next(ag,agnext,info)
|
||||
use psb_realloc_mod
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
class(amg_d_base_aggregator_type), target, intent(inout) :: agnext
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
!
|
||||
!
|
||||
select type(agnext)
|
||||
class is (amg_d_newmatch_aggregator_type)
|
||||
if (.not.is_legal_malg(agnext%matching_alg)) &
|
||||
& agnext%matching_alg = ag%matching_alg
|
||||
if (.not.is_legal_nsweeps(agnext%n_sweeps))&
|
||||
& agnext%n_sweeps = ag%n_sweeps
|
||||
if (.not.is_legal_csize(agnext%max_csize))&
|
||||
& agnext%max_csize = ag%max_csize
|
||||
if (.not.is_legal_nlevels(agnext%max_nlevels))&
|
||||
& agnext%max_nlevels = ag%max_nlevels
|
||||
! Is this going to generate shallow copies/memory leaks/double frees?
|
||||
! To be investigated further.
|
||||
call psb_safe_ab_cpy(ag%w_nxt,agnext%w,info)
|
||||
call agnext%set_c_default_w()
|
||||
class default
|
||||
! What should we do here?
|
||||
end select
|
||||
info = 0
|
||||
end subroutine d_newmatch_aggregator_update_next
|
||||
|
||||
|
||||
subroutine amg_d_newmatch_aggr_csetr(ag,what,val,info,idx)
|
||||
|
||||
Implicit None
|
||||
|
||||
! Arguments
|
||||
class(amg_d_newmatch_aggregator_type), intent(inout) :: ag
|
||||
character(len=*), intent(in) :: what
|
||||
real(psb_dpk_), intent(in) :: val
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
integer(psb_ipk_), intent(in), optional :: idx
|
||||
integer(psb_ipk_) :: err_act, iwhat
|
||||
character(len=20) :: name='d_newmatch_aggr_csetr'
|
||||
info = psb_success_
|
||||
|
||||
! For now we ignore IDX
|
||||
|
||||
select case(psb_toupper(trim(what)))
|
||||
case('NWM_LAMBDA')
|
||||
ag%lambda = val
|
||||
case default
|
||||
! Do nothing
|
||||
end select
|
||||
return
|
||||
end subroutine amg_d_newmatch_aggr_csetr
|
||||
|
||||
subroutine amg_d_newmatch_aggr_csetc(ag,what,val,info,idx)
|
||||
|
||||
Implicit None
|
||||
|
||||
! Arguments
|
||||
class(amg_d_newmatch_aggregator_type), intent(inout) :: ag
|
||||
character(len=*), intent(in) :: what
|
||||
character(len=*), intent(in) :: val
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
integer(psb_ipk_), intent(in), optional :: idx
|
||||
integer(psb_ipk_) :: err_act, iwhat
|
||||
character(len=20) :: name='d_newmatch_aggr_csetc'
|
||||
info = psb_success_
|
||||
|
||||
! For now we ignore IDX
|
||||
|
||||
select case(psb_toupper(trim(what)))
|
||||
case('NWM_PARALLEL_MATCHING')
|
||||
select case(psb_toupper(trim(val)))
|
||||
case('SEQUENTIAL','F','FALSE')
|
||||
ag%parallel_matching = .false.
|
||||
case('PARALLEL','TRUE','T')
|
||||
ag%parallel_matching =.true.
|
||||
end select
|
||||
case('NWM_REPRODUCIBLE_MATCHING')
|
||||
select case(psb_toupper(trim(val)))
|
||||
case('F','FALSE')
|
||||
ag%reproducible_matching = .false.
|
||||
case('REPRODUCIBLE','TRUE','T')
|
||||
ag%reproducible_matching =.true.
|
||||
end select
|
||||
case('NWM_NEED_SYMMETRIZE')
|
||||
select case(psb_toupper(trim(val)))
|
||||
case('FALSE','F')
|
||||
ag%need_symmetrize = .false.
|
||||
case('SYMMETRIZE','TRUE','T')
|
||||
ag%need_symmetrize =.true.
|
||||
end select
|
||||
case('NWM_UNSMOOTHED_HIERARCHY')
|
||||
select case(psb_toupper(trim(val)))
|
||||
case('F','FALSE')
|
||||
ag%unsmoothed_hierarchy = .false.
|
||||
case('T','TRUE')
|
||||
ag%unsmoothed_hierarchy =.true.
|
||||
end select
|
||||
case default
|
||||
! Do nothing
|
||||
end select
|
||||
return
|
||||
end subroutine amg_d_newmatch_aggr_csetc
|
||||
|
||||
subroutine d_newmatch_aggr_cseti(ag,what,val,info,idx)
|
||||
|
||||
Implicit None
|
||||
|
||||
! Arguments
|
||||
class(amg_d_newmatch_aggregator_type), intent(inout) :: ag
|
||||
character(len=*), intent(in) :: what
|
||||
integer(psb_ipk_), intent(in) :: val
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
integer(psb_ipk_), intent(in), optional :: idx
|
||||
integer(psb_ipk_) :: err_act, iwhat
|
||||
character(len=20) :: name='d_newmatch_aggr_cseti'
|
||||
info = psb_success_
|
||||
|
||||
! For now we ignore IDX
|
||||
|
||||
select case(psb_toupper(trim(what)))
|
||||
case('NWM_MATCH_ALG','NWM_MATCHING_ALG')
|
||||
ag%matching_alg = val
|
||||
case('NWM_SWEEPS')
|
||||
ag%n_sweeps=val
|
||||
case('NWM_MAX_CSIZE')
|
||||
ag%max_csize = val
|
||||
case('NWM_MAX_NLEVELS')
|
||||
ag%max_nlevels = val
|
||||
case('NWM_W_SIZE')
|
||||
call ag%bld_default_w(val)
|
||||
case('AGGR_SIZE')
|
||||
ag%orig_aggr_size = val
|
||||
ag%n_sweeps=max(1,ceiling(log(val*1.0)/log(2.0)))
|
||||
case default
|
||||
! Do nothing
|
||||
end select
|
||||
return
|
||||
end subroutine d_newmatch_aggr_cseti
|
||||
|
||||
subroutine d_newmatch_aggr_set_default(ag)
|
||||
|
||||
Implicit None
|
||||
|
||||
! Arguments
|
||||
class(amg_d_newmatch_aggregator_type), intent(inout) :: ag
|
||||
character(len=20) :: name='d_newmatch_aggr_set_default'
|
||||
ag%matching_alg = 1
|
||||
ag%n_sweeps = 1
|
||||
ag%max_nlevels = 36
|
||||
ag%max_csize = -1
|
||||
ag%lambda = -1
|
||||
!
|
||||
! Apparently newMatch works better
|
||||
! by keeping all entries
|
||||
!
|
||||
ag%do_clean_zeros = .false.
|
||||
|
||||
return
|
||||
|
||||
end subroutine d_newmatch_aggr_set_default
|
||||
|
||||
subroutine d_newmatch_aggregator_free(ag,info)
|
||||
use iso_c_binding
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), intent(inout) :: ag
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(ag%w)) deallocate(ag%w,stat=info)
|
||||
if (info /= 0) return
|
||||
if (allocated(ag%w_nxt)) deallocate(ag%w_nxt,stat=info)
|
||||
if (info /= 0) return
|
||||
ag%w_c_nxt%size = 0
|
||||
ag%w_c_nxt%data = c_null_ptr
|
||||
ag%w_c_nxt%owns_data = 0
|
||||
end subroutine d_newmatch_aggregator_free
|
||||
|
||||
subroutine d_newmatch_aggregator_clone(ag,agnext,info)
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), intent(inout) :: ag
|
||||
class(amg_d_base_aggregator_type), allocatable, intent(inout) :: agnext
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(agnext)) then
|
||||
call agnext%free(info)
|
||||
if (info == 0) deallocate(agnext,stat=info)
|
||||
end if
|
||||
if (info /= 0) return
|
||||
allocate(agnext,source=ag,stat=info)
|
||||
select type(agnext)
|
||||
class is (amg_d_newmatch_aggregator_type)
|
||||
call agnext%set_c_default_w()
|
||||
class default
|
||||
! Should never ever get here
|
||||
info = -1
|
||||
end select
|
||||
end subroutine d_newmatch_aggregator_clone
|
||||
|
||||
end module amg_d_newmatch_aggregator_mod
|
||||
@@ -57,6 +57,7 @@ module amg_d_onelev_mod
|
||||
use amg_d_base_smoother_mod
|
||||
use amg_d_dec_aggregator_mod
|
||||
use amg_d_parmatch_aggregator_mod
|
||||
use amg_d_newmatch_aggregator_mod
|
||||
|
||||
use psb_base_mod, only : psb_dspmat_type, psb_d_vect_type, &
|
||||
& psb_d_base_vect_type, psb_ldspmat_type, psb_dlinmap_type, psb_dpk_, &
|
||||
|
||||
@@ -143,9 +143,10 @@ contains
|
||||
type(psb_ls_coo_sparse_mat) :: tmpcoo
|
||||
logical :: display_out_, print_out_, reproducible_
|
||||
logical, parameter :: dump=.false., debug=.false., dump_mate=.false., &
|
||||
& debug_ilaggr=.false., debug_sync=.false.
|
||||
& debug_ilaggr=.false., debug_sync=.false., debug_mate=.false.
|
||||
integer(psb_ipk_), save :: idx_bldmtc=-1, idx_phase1=-1, idx_phase2=-1, idx_phase3=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
integer, parameter :: ilaggr_neginit=-1, ilaggr_nonlocal=-2
|
||||
|
||||
ictxt = desc_a%get_ctxt()
|
||||
call psb_info(ictxt,iam,np)
|
||||
@@ -187,7 +188,7 @@ contains
|
||||
call desc_a%l2gip(ilv,info,owned=.false.)
|
||||
|
||||
call psb_geall(ilaggr,desc_a,info)
|
||||
ilaggr = -1
|
||||
ilaggr = ilaggr_neginit
|
||||
call psb_geasb(ilaggr,desc_a,info)
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -213,7 +214,20 @@ contains
|
||||
call psb_barrier(ictxt)
|
||||
if (iam == 0) write(0,*)' out from buildmatching:', info
|
||||
end if
|
||||
|
||||
if (debug_mate) then
|
||||
block
|
||||
integer(psb_lpk_), allocatable :: ckmate(:)
|
||||
allocate(ckmate(nr))
|
||||
ckmate(1:nr) = mate(1:nr)
|
||||
call psb_msort(ckmate(1:nr))
|
||||
do i=1,nr-1
|
||||
if ((ckmate(i)>0) .and. (ckmate(i) == ckmate(i+1))) then
|
||||
write(0,*) iam,' Duplicate mate entry at',i,' :',ckmate(i)
|
||||
end if
|
||||
end do
|
||||
end block
|
||||
end if
|
||||
|
||||
if (info == 0) then
|
||||
if (do_timings) call psb_tic(idx_phase2)
|
||||
if (debug_sync) then
|
||||
@@ -259,7 +273,7 @@ contains
|
||||
cycle
|
||||
else
|
||||
|
||||
if (ilaggr(k) == -1) then
|
||||
if (ilaggr(k) == ilaggr_neginit) then
|
||||
|
||||
wk = w(k)
|
||||
widx = w(idx)
|
||||
@@ -267,7 +281,7 @@ contains
|
||||
nrmagg = wmax*sqrt((wk/wmax)**2+(widx/wmax)**2)
|
||||
if (nrmagg > epsilon(nrmagg)) then
|
||||
if (idx <= nr) then
|
||||
if (ilaggr(idx) == -1) then
|
||||
if (ilaggr(idx) == ilaggr_neginit) then
|
||||
! Now, if both vertices are local, the aggregate is local
|
||||
! (kinda obvious).
|
||||
nlaggr(iam) = nlaggr(iam) + 1
|
||||
@@ -275,6 +289,9 @@ contains
|
||||
ilaggr(idx) = nlaggr(iam)
|
||||
wtemp(k) = w(k)/nrmagg
|
||||
wtemp(idx) = w(idx)/nrmagg
|
||||
else
|
||||
write(0,*) iam,' Inconsistent mate? ',k,mate(k),idx,&
|
||||
&mate(idx),ilaggr(idx)
|
||||
end if
|
||||
nlpairs = nlpairs+1
|
||||
else if (idx <= nc) then
|
||||
@@ -294,7 +311,7 @@ contains
|
||||
ilaggr(k) = nlaggr(iam)
|
||||
nlpairs = nlpairs+1
|
||||
else
|
||||
ilaggr(k) = -2
|
||||
ilaggr(k) = ilaggr_nonlocal
|
||||
end if
|
||||
else
|
||||
! Use a statistically unbiased tie-breaking rule,
|
||||
@@ -309,7 +326,7 @@ contains
|
||||
ilaggr(k) = nlaggr(iam)
|
||||
nlpairs = nlpairs+1
|
||||
else
|
||||
ilaggr(k) = -2
|
||||
ilaggr(k) = ilaggr_nonlocal
|
||||
end if
|
||||
end if
|
||||
end if
|
||||
@@ -325,6 +342,12 @@ contains
|
||||
nlsingl = nlsingl + 1
|
||||
end if
|
||||
end if
|
||||
if (ilaggr(k) == ilaggr_neginit) then
|
||||
write(0,*) iam,' Error: no update to ',k,mate(k),&
|
||||
& abs(w(k)),nrmagg,epsilon(nrmagg),wtemp(k)
|
||||
end if
|
||||
else
|
||||
if (ilaggr(k)<0) write(0,*) 'Strange? ',k,ilaggr(k)
|
||||
end if
|
||||
end if
|
||||
end do
|
||||
@@ -332,7 +355,7 @@ contains
|
||||
if (do_timings) call psb_tic(idx_phase3)
|
||||
|
||||
! Ok, now compute offsets, gather halo and fix non-local
|
||||
! aggregates (those where ilaggr == -2)
|
||||
! aggregates (those where ilaggr == ilaggr_nonlocal)
|
||||
call psb_sum(ictxt,nlaggr)
|
||||
ntaggr = sum(nlaggr(0:np-1))
|
||||
naggrm1 = sum(nlaggr(0:iam-1))
|
||||
@@ -347,7 +370,7 @@ contains
|
||||
call psb_halo(wtemp,desc_a,info)
|
||||
! Cleanup as yet unmarked entries
|
||||
do k=1,nr
|
||||
if (ilaggr(k) == -2) then
|
||||
if (ilaggr(k) == ilaggr_nonlocal) then
|
||||
idx = mate(k)
|
||||
if (idx > nr) then
|
||||
i = ilaggr(idx)
|
||||
@@ -359,9 +382,14 @@ contains
|
||||
else
|
||||
write(0,*) 'Error : unresolved (paired) index ',k,idx,i,nr,nc, ilv(k),ilv(idx)
|
||||
end if
|
||||
end if
|
||||
if (ilaggr(k) <0) then
|
||||
write(0,*) 'Matchboxp: Funny number: ',k,ilv(k),ilaggr(k),wtemp(k)
|
||||
else if (ilaggr(k) <0) then
|
||||
write(0,*) iam,'Matchboxp: Funny number: ',k,ilv(k),ilaggr(k),wtemp(k)
|
||||
write(0,*) iam,' : : ',nr,nc,mate(k)
|
||||
if (mate(k) <= nr) then
|
||||
write(0,*) iam,' : : ',ilaggr(mate(k)),mate(mate(k)),&
|
||||
& ilv(k),ilv(mate(k)), ilv(mate(mate(k))),ilaggr(mate(mate(k)))
|
||||
end if
|
||||
flush(0)
|
||||
end if
|
||||
end do
|
||||
if (debug_sync) then
|
||||
@@ -414,7 +442,7 @@ contains
|
||||
|
||||
end block
|
||||
if (iam == 0) then
|
||||
write(0,*) 'Matching statistics: Unmatched nodes ',&
|
||||
write(0,*) iam,'Matching statistics: Unmatched nodes ',&
|
||||
& nunmatched,' Singletons:',nlsingl,' Pairs:',nlpairs
|
||||
end if
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@ MODDIR=../../../modules
|
||||
HERE=../..
|
||||
|
||||
FINCLUDES=$(FMFLAG)$(HERE) $(FMFLAG)$(MODDIR) $(FMFLAG)$(INCDIR) $(PSBLAS_INCLUDES)
|
||||
CXXINCLUDES=$(FMFLAG)$(HERE) $(FMFLAG)$(INCDIR) $(FMFLAG)/.
|
||||
CXXINCLUDES=$(FIFLAG)$(HERE) $(FIFLAG)$(INCDIR) $(FIFLAG)/. -I../../../../ParallelRomaF-main/include -I$(PSBLAS_INCDIR)
|
||||
|
||||
#CINCLUDES= -I${SUPERLU_INCDIR} -I${HSL_INCDIR} -I${SPRAL_INCDIR} -I/home/users/pasqua/Ambra/BootCMatch/include -lBCM -L/home/users/pasqua/Ambra/BootCMatch/lib -lm
|
||||
|
||||
@@ -59,12 +59,36 @@ amg_s_parmatch_spmm_bld.o \
|
||||
amg_s_parmatch_spmm_bld_ov.o \
|
||||
amg_s_parmatch_unsmth_bld.o \
|
||||
amg_s_parmatch_smth_bld.o \
|
||||
amg_s_parmatch_spmm_bld_inner.o
|
||||
amg_s_parmatch_spmm_bld_inner.o \
|
||||
amg_d_newmatch_aggregator_inner_mat_asb.o\
|
||||
amg_d_newmatch_aggregator_mat_asb.o \
|
||||
amg_d_newmatch_aggregator_mat_bld.o \
|
||||
amg_d_newmatch_aggregator_tprol.o \
|
||||
amg_d_newmatch_map_to_tprol.o \
|
||||
amg_d_newmatch_spmm_bld_inner.o \
|
||||
amg_d_newmatch_spmm_bld_ov.o
|
||||
|
||||
MPCOBJS=MatchBoxPC.o \
|
||||
algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC.o
|
||||
MPCXXOBJS=MatchBoxPC.o \
|
||||
algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC.o \
|
||||
newmatch_interface.o \
|
||||
sendBundledMessages.o \
|
||||
initialize.o \
|
||||
extractUChunk.o \
|
||||
isAlreadyMatched.o \
|
||||
findOwnerOfGhost.o \
|
||||
clean.o \
|
||||
computeCandidateMate.o \
|
||||
parallelComputeCandidateMateB.o \
|
||||
processMatchedVertices.o \
|
||||
processMatchedVerticesAndSendMessages.o \
|
||||
processCrossEdge.o \
|
||||
queueTransfer.o \
|
||||
processMessages.o \
|
||||
processExposedVertex.o \
|
||||
algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP.o \
|
||||
MatchingAlgorithms.o
|
||||
|
||||
OBJS = $(FOBJS) $(MPCOBJS)
|
||||
OBJS = $(FOBJS) $(MPCOBJS) $(MPCXXOBJS)
|
||||
|
||||
LIBNAME=libamg_prec.a
|
||||
|
||||
@@ -74,10 +98,6 @@ lib: objs
|
||||
$(AR) $(HERE)/$(LIBNAME) $(OBJS)
|
||||
$(RANLIB) $(HERE)/$(LIBNAME)
|
||||
|
||||
mpobjs:
|
||||
(make $(MPFOBJS) F90="$(MPF90)" F90COPT="$(F90COPT)")
|
||||
(make $(MPCOBJS) CC="$(MPCC)" CCOPT="$(CCOPT)")
|
||||
|
||||
veryclean: clean
|
||||
/bin/rm -f $(LIBNAME)
|
||||
|
||||
|
||||
@@ -60,17 +60,43 @@ void dMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card ) {
|
||||
#if !defined(SERIAL_MPI)
|
||||
MPI_Comm C_comm=MPI_Comm_f2c(icomm);
|
||||
|
||||
#ifdef DEBUG
|
||||
fprintf(stderr,"MatchBoxPC: rank %d nlver %ld nledge %ld [ %ld %ld ]\n",
|
||||
myRank,NLVer, NLEdge,verDistance[0],verDistance[1]);
|
||||
#endif
|
||||
dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(NLVer, NLEdge,
|
||||
|
||||
|
||||
#define TIME_TRACKER
|
||||
#ifdef TIME_TRACKER
|
||||
double tmr = MPI_Wtime();
|
||||
#endif
|
||||
|
||||
#define OMP
|
||||
#ifdef OMP
|
||||
dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(NLVer, NLEdge,
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
myRank, numProcs, C_comm,
|
||||
msgIndSent, msgActualSent, msgPercent,
|
||||
ph0_time, ph1_time, ph2_time,
|
||||
ph1_card, ph2_card );
|
||||
#else
|
||||
dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(NLVer, NLEdge,
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
myRank, numProcs, C_comm,
|
||||
msgIndSent, msgActualSent, msgPercent,
|
||||
ph0_time, ph1_time, ph2_time,
|
||||
ph1_card, ph2_card );
|
||||
#endif
|
||||
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
tmr = MPI_Wtime() - tmr;
|
||||
fprintf(stderr, "Elaboration time: %f for %ld nodes\n", tmr, NLVer);
|
||||
#endif
|
||||
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -52,145 +52,412 @@
|
||||
|
||||
#ifndef _matchboxpC_H_
|
||||
#define _matchboxpC_H_
|
||||
//Turn on a lot of debugging information with this switch:
|
||||
// Turn on a lot of debugging information with this switch:
|
||||
//#define PRINT_DEBUG_INFO_
|
||||
#include <stdio.h>
|
||||
#include <iostream>
|
||||
#include <assert.h>
|
||||
#include <map>
|
||||
#include <vector>
|
||||
// #include "matchboxp.h"
|
||||
#include "omp.h"
|
||||
#include "primitiveDataTypeDefinitions.h"
|
||||
#include "dataStrStaticQueue.h"
|
||||
|
||||
using namespace std;
|
||||
|
||||
const int NUM_THREAD = 4;
|
||||
const int UCHUNK = 10;
|
||||
|
||||
const MilanLongInt REQUEST = 1;
|
||||
const MilanLongInt SUCCESS = 2;
|
||||
const MilanLongInt FAILURE = 3;
|
||||
const MilanLongInt SIZEINFO = 4;
|
||||
|
||||
const int ComputeTag = 7; // Predefined tag
|
||||
const int BundleTag = 9; // Predefined tag
|
||||
|
||||
static vector<MilanLongInt> DEFAULT_VECTOR;
|
||||
|
||||
// MPI type map
|
||||
template <typename T>
|
||||
MPI_Datatype TypeMap();
|
||||
template <>
|
||||
inline MPI_Datatype TypeMap<int64_t>() { return MPI_LONG_LONG; }
|
||||
template <>
|
||||
inline MPI_Datatype TypeMap<int>() { return MPI_INT; }
|
||||
template <>
|
||||
inline MPI_Datatype TypeMap<double>() { return MPI_DOUBLE; }
|
||||
template <>
|
||||
inline MPI_Datatype TypeMap<float>() { return MPI_FLOAT; }
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
extern "C"
|
||||
{
|
||||
#endif
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
#define MilanMpiLongInt MPI_LONG_LONG
|
||||
|
||||
#define MilanMpiLongInt MPI_LONG_LONG
|
||||
|
||||
#ifndef _primitiveDataType_Definition_
|
||||
#define _primitiveDataType_Definition_
|
||||
//Regular integer:
|
||||
#ifndef INTEGER_H
|
||||
#define INTEGER_H
|
||||
typedef int32_t MilanInt;
|
||||
#endif
|
||||
// Regular integer:
|
||||
#ifndef INTEGER_H
|
||||
#define INTEGER_H
|
||||
typedef int32_t MilanInt;
|
||||
#endif
|
||||
|
||||
//Regular long integer:
|
||||
#ifndef LONG_INT_H
|
||||
#define LONG_INT_H
|
||||
#ifdef BIT64
|
||||
typedef int64_t MilanLongInt;
|
||||
typedef MPI_LONG MilanMpiLongInt;
|
||||
#else
|
||||
typedef int32_t MilanLongInt;
|
||||
typedef MPI_INT MilanMpiLongInt;
|
||||
#endif
|
||||
#endif
|
||||
// Regular long integer:
|
||||
#ifndef LONG_INT_H
|
||||
#define LONG_INT_H
|
||||
#ifdef BIT64
|
||||
typedef int64_t MilanLongInt;
|
||||
typedef MPI_LONG MilanMpiLongInt;
|
||||
#else
|
||||
typedef int32_t MilanLongInt;
|
||||
typedef MPI_INT MilanMpiLongInt;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
//Regular boolean
|
||||
#ifndef BOOL_H
|
||||
#define BOOL_H
|
||||
typedef bool MilanBool;
|
||||
#endif
|
||||
// Regular boolean
|
||||
#ifndef BOOL_H
|
||||
#define BOOL_H
|
||||
typedef bool MilanBool;
|
||||
#endif
|
||||
|
||||
//Regular double and absolute value computation:
|
||||
#ifndef REAL_H
|
||||
#define REAL_H
|
||||
typedef double MilanReal;
|
||||
typedef MPI_DOUBLE MilanMpiReal;
|
||||
inline MilanReal MilanAbs(MilanReal value)
|
||||
{
|
||||
return fabs(value);
|
||||
}
|
||||
#endif
|
||||
// Regular double and absolute value computation:
|
||||
#ifndef REAL_H
|
||||
#define REAL_H
|
||||
typedef double MilanReal;
|
||||
typedef MPI_DOUBLE MilanMpiReal;
|
||||
inline MilanReal MilanAbs(MilanReal value)
|
||||
{
|
||||
return fabs(value);
|
||||
}
|
||||
#endif
|
||||
|
||||
//Regular float and absolute value computation:
|
||||
#ifndef FLOAT_H
|
||||
#define FLOAT_H
|
||||
typedef float MilanFloat;
|
||||
typedef MPI_FLOAT MilanMpiFloat;
|
||||
inline MilanFloat MilanAbsFloat(MilanFloat value)
|
||||
{
|
||||
return fabs(value);
|
||||
}
|
||||
#endif
|
||||
// Regular float and absolute value computation:
|
||||
#ifndef FLOAT_H
|
||||
#define FLOAT_H
|
||||
typedef float MilanFloat;
|
||||
typedef MPI_FLOAT MilanMpiFloat;
|
||||
inline MilanFloat MilanAbsFloat(MilanFloat value)
|
||||
{
|
||||
return fabs(value);
|
||||
}
|
||||
#endif
|
||||
|
||||
//// Define the limits:
|
||||
#ifndef LIMITS_H
|
||||
#define LIMITS_H
|
||||
//Integer Maximum and Minimum:
|
||||
// #define MilanIntMax INT_MAX
|
||||
// #define MilanIntMin INT_MIN
|
||||
#define MilanIntMax INT32_MAX
|
||||
#define MilanIntMin INT32_MIN
|
||||
//// Define the limits:
|
||||
#ifndef LIMITS_H
|
||||
#define LIMITS_H
|
||||
// Integer Maximum and Minimum:
|
||||
// #define MilanIntMax INT_MAX
|
||||
// #define MilanIntMin INT_MIN
|
||||
#define MilanIntMax INT32_MAX
|
||||
#define MilanIntMin INT32_MIN
|
||||
|
||||
#ifdef BIT64
|
||||
#define MilanLongIntMax INT64_MAX
|
||||
#define MilanLongIntMin -INT64_MAX
|
||||
#else
|
||||
#define MilanLongIntMax INT32_MAX
|
||||
#define MilanLongIntMin -INT32_MAX
|
||||
#endif
|
||||
#ifdef BIT64
|
||||
#define MilanLongIntMax INT64_MAX
|
||||
#define MilanLongIntMin -INT64_MAX
|
||||
#else
|
||||
#define MilanLongIntMax INT32_MAX
|
||||
#define MilanLongIntMin -INT32_MAX
|
||||
#endif
|
||||
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// +INFINITY
|
||||
const double PLUS_INFINITY = numeric_limits<int>::infinity();
|
||||
const double MINUS_INFINITY = -PLUS_INFINITY;
|
||||
//#define MilanRealMax LDBL_MAX
|
||||
#define MilanRealMax PLUS_INFINITY
|
||||
#define MilanRealMin MINUS_INFINITY
|
||||
//#define MilanRealMax LDBL_MAX
|
||||
#define MilanRealMax PLUS_INFINITY
|
||||
#define MilanRealMin MINUS_INFINITY
|
||||
#endif
|
||||
|
||||
//Function of find the owner of a ghost vertex using binary search:
|
||||
inline MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs);
|
||||
// Function of find the owner of a ghost vertex using binary search:
|
||||
MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs);
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC
|
||||
(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* verLocPtr, MilanLongInt* verLocInd, MilanReal* edgeLocWeight,
|
||||
MilanLongInt* verDistance,
|
||||
MilanLongInt* Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt* msgIndSent, MilanLongInt* msgActualSent, MilanReal* msgPercent,
|
||||
MilanReal* ph0_time, MilanReal* ph1_time, MilanReal* ph2_time,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card );
|
||||
MilanLongInt firstComputeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight);
|
||||
|
||||
void salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC
|
||||
(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* verLocPtr, MilanLongInt* verLocInd, MilanFloat* edgeLocWeight,
|
||||
MilanLongInt* verDistance,
|
||||
MilanLongInt* Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt* msgIndSent, MilanLongInt* msgActualSent, MilanReal* msgPercent,
|
||||
MilanReal* ph0_time, MilanReal* ph1_time, MilanReal* ph2_time,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card );
|
||||
void queuesTransfer(vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void dMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* verLocPtr, MilanLongInt* verLocInd, MilanReal* edgeLocWeight,
|
||||
MilanLongInt* verDistance,
|
||||
MilanLongInt* Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MilanInt icomm,
|
||||
MilanLongInt* msgIndSent, MilanLongInt* msgActualSent, MilanReal* msgPercent,
|
||||
MilanReal* ph0_time, MilanReal* ph1_time, MilanReal* ph2_time,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card );
|
||||
bool isAlreadyMatched(MilanLongInt node,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
void sMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* verLocPtr, MilanLongInt* verLocInd, MilanFloat* edgeLocWeight,
|
||||
MilanLongInt* verDistance,
|
||||
MilanLongInt* Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MilanInt icomm,
|
||||
MilanLongInt* msgIndSent, MilanLongInt* msgActualSent, MilanReal* msgPercent,
|
||||
MilanReal* ph0_time, MilanReal* ph1_time, MilanReal* ph2_time,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card );
|
||||
MilanLongInt computeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt StartIndex, MilanLongInt EndIndex,
|
||||
MilanLongInt *numGhostEdgesPtr,
|
||||
MilanLongInt *numGhostVerticesPtr,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &Counter,
|
||||
vector<MilanLongInt> &verGhostPtr,
|
||||
vector<MilanLongInt> &verGhostInd,
|
||||
vector<MilanLongInt> &tempCounter,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Message,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
MilanLongInt *&candidateMate,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void clean(MilanLongInt NLVer,
|
||||
MilanInt myRank,
|
||||
MilanLongInt MessageIndex,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus,
|
||||
MilanInt BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
MilanLongInt msgActual,
|
||||
MilanLongInt *msgActualSent,
|
||||
MilanLongInt msgInd,
|
||||
MilanLongInt *msgIndSent,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanReal *msgPercent);
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_B(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *candidateMate);
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
MilanLongInt *SPtr);
|
||||
|
||||
void processMatchedVertices(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void processMatchedVerticesAndSendMessages(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message);
|
||||
|
||||
void sendBundledMessages(MilanLongInt *numGhostEdgesPtr,
|
||||
MilanInt *BufferSizePtr,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanLongInt *MessageIndexPtr,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus);
|
||||
|
||||
void processMessages(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *SPtr,
|
||||
vector<MilanLongInt> &U);
|
||||
|
||||
void extractUChunk(
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU);
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
|
||||
void salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
|
||||
void dMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MilanInt icomm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
|
||||
void sMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MilanInt icomm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
|
||||
#endif
|
||||
#ifdef __cplusplus
|
||||
|
||||
@@ -72,12 +72,6 @@
|
||||
|
||||
#ifdef SERIAL_MPI
|
||||
#else
|
||||
//MPI type map
|
||||
template<typename T> MPI_Datatype TypeMap();
|
||||
template<> inline MPI_Datatype TypeMap<int64_t>() { return MPI_LONG_LONG; }
|
||||
template<> inline MPI_Datatype TypeMap<int>() { return MPI_INT; }
|
||||
template<> inline MPI_Datatype TypeMap<double>() { return MPI_DOUBLE; }
|
||||
template<> inline MPI_Datatype TypeMap<float>() { return MPI_FLOAT; }
|
||||
|
||||
// DOUBLE PRECISION VERSION
|
||||
//WARNING: The vertex block on a given rank is contiguous
|
||||
|
||||
+554
@@ -0,0 +1,554 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
// ***********************************************************************
|
||||
//
|
||||
// MatchboxP: A C++ library for approximate weighted matching
|
||||
// Mahantesh Halappanavar (hala@pnnl.gov)
|
||||
// Pacific Northwest National Laboratory
|
||||
//
|
||||
// ***********************************************************************
|
||||
//
|
||||
// Copyright (2021) Battelle Memorial Institute
|
||||
// All rights reserved.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without
|
||||
// modification, are permitted provided that the following conditions
|
||||
// are met:
|
||||
//
|
||||
// 1. Redistributions of source code must retain the above copyright
|
||||
// notice, this list of conditions and the following disclaimer.
|
||||
//
|
||||
// 2. Redistributions in binary form must reproduce the above copyright
|
||||
// notice, this list of conditions and the following disclaimer in the
|
||||
// documentation and/or other materials provided with the distribution.
|
||||
//
|
||||
// 3. Neither the name of the copyright holder nor the names of its
|
||||
// contributors may be used to endorse or promote products derived from
|
||||
// this software without specific prior written permission.
|
||||
//
|
||||
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
|
||||
// FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE
|
||||
// COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT,
|
||||
// INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
// BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
// LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
// CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
|
||||
// LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
|
||||
// ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
// POSSIBILITY OF SUCH DAMAGE.
|
||||
//
|
||||
// ************************************************************************
|
||||
//////////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// DOMINATING EDGES MODEL ///////////////////////////////////
|
||||
//////////////////////////////////////////////////////////////////////////////////////
|
||||
/* Function : algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMate()
|
||||
*
|
||||
* Date : New update: Feb 17, 2019, Richland, Washington.
|
||||
* Date : Original development: May 17, 2009, E&CS Bldg.
|
||||
*
|
||||
* Purpose : Compute Approximate Maximum Weight Matching in Linear Time
|
||||
*
|
||||
* Args : inputMatrix - instance of Compressed-Col format of Matrix
|
||||
* Mate - The Mate array
|
||||
*
|
||||
* Returns : By Value: (void)
|
||||
* By Reference: Mate
|
||||
*
|
||||
* Comments : 1/2 Approx Algorithm. Picks the locally available heaviest edge.
|
||||
* Assumption: The Mate Array is empty.
|
||||
*/
|
||||
|
||||
/*
|
||||
NLVer = #of vertices, NLEdge = #of edges
|
||||
CSR/CSC/Compressed format: verLocPtr = Pointer, verLocInd = Index, edgeLocWeight = edge weights (positive real numbers)
|
||||
verDistance = A vector of size |P|+1 containing the cumulative number of vertices per process
|
||||
Mate = A vector of size |V_p| (local subgraph) to store the output (matching)
|
||||
MPI: myRank, numProcs, comm,
|
||||
Statistics: msgIndSent, msgActualSent, msgPercent : Size: |P| number of processes in the comm-world
|
||||
Statistics: ph0_time, ph1_time, ph2_time: Runtimes
|
||||
Statistics: ph1_card, ph2_card : Size: |P| number of processes in the comm-world (number of matched edges in Phase 1 and Phase 2)
|
||||
*/
|
||||
//#define DEBUG_HANG_
|
||||
#ifdef SERIAL_MPI
|
||||
#else
|
||||
|
||||
// DOUBLE PRECISION VERSION
|
||||
// WARNING: The vertex block on a given rank is contiguous
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent,
|
||||
MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card)
|
||||
{
|
||||
|
||||
/*
|
||||
* verDistance: it's a vector long as the number of processors.
|
||||
* verDistance[i] contains the first node index of the i-th processor
|
||||
* verDistance[i + 1] contains the last node index of the i-th processor
|
||||
* NLVer: number of elements in the LocPtr
|
||||
* NLEdge: number of edges assigned to the current processor
|
||||
*
|
||||
* Contains the portion of matrix assigned to the processor in
|
||||
* Yale notation
|
||||
* verLocInd: contains the positions on row of the matrix
|
||||
* verLocPtr: i-th value is the position of the first element on the i-th row and
|
||||
* i+1-th value is the position of the first element on the i+1-th row
|
||||
*/
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Within algoEdgeApproxDominatingEdgesLinearSearchMessageBundling()";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ") verDistance [" ;
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
cout << verDistance[i] << "," << verDistance[i+1];
|
||||
cout << "]\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0) {
|
||||
cout << "\n(" << myRank << ") verDistance [" ;
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
cout << verDistance[i] << "," ;
|
||||
cout << verDistance[numProcs]<< "]\n";
|
||||
}
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
MilanLongInt StartIndex = verDistance[myRank]; // The starting vertex owned by the current rank
|
||||
MilanLongInt EndIndex = verDistance[myRank + 1] - 1; // The ending vertex owned by the current rank
|
||||
|
||||
MPI_Status computeStatus;
|
||||
|
||||
MilanLongInt msgActual = 0, msgInd = 0;
|
||||
MilanReal heaviestEdgeWt = 0.0f; // Assumes positive weight
|
||||
MilanReal startTime, finishTime;
|
||||
|
||||
startTime = MPI_Wtime();
|
||||
|
||||
// Data structures for sending and receiving messages:
|
||||
vector<MilanLongInt> Message; // [ u, v, message_type ]
|
||||
Message.resize(3, -1);
|
||||
// Data structures for Message Bundling:
|
||||
// Although up to two messages can be sent along any cross edge,
|
||||
// only one message will be sent in the initialization phase -
|
||||
// one of: REQUEST/FAILURE/SUCCESS
|
||||
vector<MilanLongInt> QLocalVtx, QGhostVtx, QMsgType;
|
||||
vector<MilanInt> QOwner; // Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
|
||||
MilanLongInt *PCounter = new MilanLongInt[numProcs];
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
PCounter[i] = 0;
|
||||
|
||||
MilanLongInt NumMessagesBundled = 0;
|
||||
// TODO when the last computational section will be refactored this could be eliminated
|
||||
MilanInt ghostOwner = 0; // Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
MilanLongInt *candidateMate = nullptr;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")NV: " << NLVer << " Edges: " << NLEdge;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")StartIndex: " << StartIndex << " EndIndex: " << EndIndex;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Other Variables:
|
||||
MilanLongInt u = -1, v = -1, w = -1, i = 0;
|
||||
MilanLongInt k = -1, adj1 = -1, adj2 = -1;
|
||||
MilanLongInt k1 = -1, adj11 = -1, adj12 = -1;
|
||||
MilanLongInt myCard = 0;
|
||||
|
||||
// Build the Ghost Vertex Set: Vg
|
||||
map<MilanLongInt, MilanLongInt> Ghost2LocalMap; // Map each ghost vertex to a local vertex
|
||||
vector<MilanLongInt> Counter; // Store the edge count for each ghost vertex
|
||||
MilanLongInt numGhostVertices = 0, numGhostEdges = 0; // Number of Ghost vertices
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")About to compute Ghost Vertices...";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0)
|
||||
cout << "\n(" << myRank << ")About to compute Ghost Vertices...";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Define Adjacency Lists for Ghost Vertices:
|
||||
// cout<<"Building Ghost data structures ... \n\n";
|
||||
vector<MilanLongInt> verGhostPtr, verGhostInd, tempCounter;
|
||||
// Mate array for ghost vertices:
|
||||
vector<MilanLongInt> GMate; // Proportional to the number of ghost vertices
|
||||
MilanLongInt S;
|
||||
MilanLongInt privateMyCard = 0;
|
||||
vector<MilanLongInt> PCumulative, PMessageBundle, PSizeInfoMessages;
|
||||
vector<MPI_Request> SRequest; // Requests that are used for each send message
|
||||
vector<MPI_Status> SStatus; // Status of sent messages, used in MPI_Wait
|
||||
MilanLongInt MessageIndex = 0; // Pointer for current message
|
||||
MilanInt BufferSize;
|
||||
MilanLongInt *Buffer;
|
||||
|
||||
vector<MilanLongInt> privateQLocalVtx, privateQGhostVtx, privateQMsgType;
|
||||
vector<MilanInt> privateQOwner;
|
||||
vector<MilanLongInt> U, privateU;
|
||||
|
||||
initialize(NLVer, NLEdge, StartIndex,
|
||||
EndIndex, &numGhostEdges,
|
||||
&numGhostVertices, &S,
|
||||
verLocInd, verLocPtr,
|
||||
Ghost2LocalMap, Counter,
|
||||
verGhostPtr, verGhostInd,
|
||||
tempCounter, GMate,
|
||||
Message, QLocalVtx,
|
||||
QGhostVtx, QMsgType, QOwner,
|
||||
candidateMate, U,
|
||||
privateU,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph0_time = finishTime - startTime; // Time taken for Phase-0: Initialization
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished initialization" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
startTime = MPI_Wtime();
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
//////////////////////////////////// INITIALIZATION /////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
// Compute the Initial Matching Set:
|
||||
|
||||
/*
|
||||
* OMP PARALLEL_COMPUTE_CANDIDATE_MATE_B has been splitted from
|
||||
* PARALLEL_PROCESS_EXPOSED_VERTEX_B in order to better parallelize
|
||||
* the two.
|
||||
* PARALLEL_COMPUTE_CANDIDATE_MATE_B is now totally parallel.
|
||||
*/
|
||||
|
||||
PARALLEL_COMPUTE_CANDIDATE_MATE_B(NLVer,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
myRank,
|
||||
edgeLocWeight,
|
||||
candidateMate);
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Exposed Vertex" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " candidateMate after parallelCompute " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << candidateMate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
/*
|
||||
* PARALLEL_PROCESS_EXPOSED_VERTEX_B
|
||||
* TODO: write comment
|
||||
*
|
||||
* TODO: Test when it's actually more efficient to execute this code
|
||||
* in parallel.
|
||||
*/
|
||||
PARALLEL_PROCESS_EXPOSED_VERTEX_B(NLVer,
|
||||
candidateMate,
|
||||
verLocInd,
|
||||
verLocPtr,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
Mate,
|
||||
GMate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
U,
|
||||
privateU,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
tempCounter.clear(); // Do not need this any more
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Exposed Vertex" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " Mate after Exposed Vertices " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MATCHED VERTICES //////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
// TODO what would be the optimal UCHUNK
|
||||
vector<MilanLongInt> UChunkBeingProcessed;
|
||||
UChunkBeingProcessed.reserve(UCHUNK);
|
||||
|
||||
processMatchedVertices(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
candidateMate,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Process Vertices" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " Mate after Matched Vertices " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
///////////////////////////// SEND BUNDLED MESSAGES /////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
sendBundledMessages(&numGhostEdges,
|
||||
&BufferSize,
|
||||
Buffer,
|
||||
PCumulative,
|
||||
PMessageBundle,
|
||||
PSizeInfoMessages,
|
||||
PCounter,
|
||||
NumMessagesBundled,
|
||||
&msgActual,
|
||||
&MessageIndex,
|
||||
numProcs,
|
||||
myRank,
|
||||
comm,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
SRequest,
|
||||
SStatus);
|
||||
|
||||
///////////////////////// END OF SEND BUNDLED MESSAGES //////////////////////////////////
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph1_time = finishTime - startTime; // Time taken for Phase-1
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished sendBundles" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
*ph1_card = myCard; // Cardinality at the end of Phase-1
|
||||
startTime = MPI_Wtime();
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
//////////////////////////////////////// MAIN LOOP //////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
// Main While Loop:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Entering While(true) loop..";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
while (true) {
|
||||
#ifdef DEBUG_HANG_
|
||||
//if (myRank == 0)
|
||||
cout << "\n(" << myRank << ") Main loop" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MATCHED VERTICES //////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMatchedVerticesAndSendMessages(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
candidateMate,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner,
|
||||
comm,
|
||||
&msgActual,
|
||||
Message);
|
||||
|
||||
///////////////////////// END OF PROCESS MATCHED VERTICES /////////////////////////
|
||||
|
||||
//// BREAK IF NO MESSAGES EXPECTED /////////
|
||||
#ifdef DEBUG_HANG_
|
||||
#if 0
|
||||
cout << myRank << " Mate after ProcessMatchedAndSend phase "<<S <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Deciding whether to break: S= " << S << endl;
|
||||
#endif
|
||||
|
||||
if (S == 0) {
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << "\n(" << myRank << ") Breaking out" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MESSAGES //////////////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMessages(NLVer,
|
||||
Mate,
|
||||
candidateMate,
|
||||
Ghost2LocalMap,
|
||||
GMate,
|
||||
Counter,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&msgActual,
|
||||
edgeLocWeight,
|
||||
verDistance,
|
||||
verLocPtr,
|
||||
k,
|
||||
verLocInd,
|
||||
numProcs,
|
||||
myRank,
|
||||
comm,
|
||||
Message,
|
||||
numGhostEdges,
|
||||
u,
|
||||
v,
|
||||
&S,
|
||||
U);
|
||||
|
||||
///////////////////////// END OF PROCESS MESSAGES /////////////////////////////////
|
||||
#ifdef DEBUG_HANG_
|
||||
#if 0
|
||||
cout << myRank << " Mate after ProcessMessages phase "<<S <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Finished Message processing phase: S= " << S;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")** SENT : ACTUAL= " << msgActual;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")** SENT : INDIVIDUAL= " << msgInd << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of while (true)
|
||||
|
||||
clean(NLVer,
|
||||
myRank,
|
||||
MessageIndex,
|
||||
SRequest,
|
||||
SStatus,
|
||||
BufferSize,
|
||||
Buffer,
|
||||
msgActual,
|
||||
msgActualSent,
|
||||
msgInd,
|
||||
msgIndSent,
|
||||
NumMessagesBundled,
|
||||
msgPercent);
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph2_time = finishTime - startTime; // Time taken for Phase-2
|
||||
*ph2_card = myCard; // Cardinality at the end of Phase-2
|
||||
}
|
||||
// End of algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMate
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -97,6 +97,8 @@ subroutine amg_c_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
integer(psb_lpk_) :: ntaggr
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
logical :: clean_zeros
|
||||
integer(psb_ipk_), save :: idx_map_bld=-1, idx_map_tprol=-1
|
||||
logical, parameter :: do_timings=.false.
|
||||
|
||||
name='amg_c_dec_aggregator_tprol'
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -108,6 +110,10 @@ subroutine amg_c_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
if ((do_timings).and.(idx_map_bld==-1)) &
|
||||
& idx_map_bld = psb_get_timer_idx("DEC_TPROL: map_bld")
|
||||
if ((do_timings).and.(idx_map_tprol==-1)) &
|
||||
& idx_map_tprol = psb_get_timer_idx("DEC_TPROL: map_tprol")
|
||||
|
||||
call amg_check_def(parms%ml_cycle,'Multilevel cycle',&
|
||||
& amg_mult_ml_,is_legal_ml_cycle)
|
||||
@@ -121,10 +127,14 @@ subroutine amg_c_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
! The decoupled aggregator based on SOC measures ignores
|
||||
! ag_data except for clean_zeros; soc_map_bld is a procedure pointer.
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_map_bld)
|
||||
clean_zeros = ag%do_clean_zeros
|
||||
call ag%soc_map_bld(parms%aggr_ord,parms%aggr_thresh,clean_zeros,a,desc_a,nlaggr,ilaggr,info)
|
||||
if (do_timings) call psb_toc(idx_map_bld)
|
||||
if (do_timings) call psb_tic(idx_map_tprol)
|
||||
|
||||
if (info==psb_success_) call amg_map_to_tprol(desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
if (do_timings) call psb_toc(idx_map_tprol)
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
call psb_errpush(info,name,a_err='soc_map_bld/map_to_tprol')
|
||||
|
||||
@@ -140,6 +140,9 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
real(psb_spk_) :: anorm, omega, tmp, dg, theta
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
name='amg_aggrmat_smth_bld'
|
||||
info=psb_success_
|
||||
@@ -153,6 +156,23 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
ctxt = desc_a%get_context()
|
||||
|
||||
call psb_info(ctxt, me, np)
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("DEC_SMTH_BLD: par_spspmm")
|
||||
if ((do_timings).and.(idx_phase1==-1)) &
|
||||
& idx_phase1 = psb_get_timer_idx("DEC_SMTH_BLD: phase1 ")
|
||||
if ((do_timings).and.(idx_phase2==-1)) &
|
||||
& idx_phase2 = psb_get_timer_idx("DEC_SMTH_BLD: phase2 ")
|
||||
if ((do_timings).and.(idx_phase3==-1)) &
|
||||
& idx_phase3 = psb_get_timer_idx("DEC_SMTH_BLD: phase3 ")
|
||||
if ((do_timings).and.(idx_gtrans==-1)) &
|
||||
& idx_gtrans = psb_get_timer_idx("DEC_SMTH_BLD: gtrans ")
|
||||
if ((do_timings).and.(idx_refine==-1)) &
|
||||
& idx_refine = psb_get_timer_idx("DEC_SMTH_BLD: refine ")
|
||||
if ((do_timings).and.(idx_cdasb==-1)) &
|
||||
& idx_cdasb = psb_get_timer_idx("DEC_SMTH_BLD: cdasb ")
|
||||
if ((do_timings).and.(idx_ptap==-1)) &
|
||||
& idx_ptap = psb_get_timer_idx("DEC_SMTH_BLD: ptap_bld ")
|
||||
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
@@ -171,6 +191,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
! naggr: number of local aggregates
|
||||
! nrow: local rows.
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_phase1)
|
||||
|
||||
! Get the diagonal D
|
||||
adiag = a%get_diag(info)
|
||||
@@ -196,7 +217,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!
|
||||
! Build the filtered matrix Af from A
|
||||
!
|
||||
|
||||
!$OMP parallel do private(i,j,tmp,jd) schedule(static)
|
||||
do i=1, nrow
|
||||
tmp = czero
|
||||
jd = -1
|
||||
@@ -214,11 +235,13 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
enddo
|
||||
!$OMP end parallel do
|
||||
! Take out zeroed terms
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= czero) then
|
||||
adiag(i) = cone / adiag(i)
|
||||
@@ -226,7 +249,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
adiag(i) = cone
|
||||
end if
|
||||
end do
|
||||
|
||||
!$OMP end parallel do
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
@@ -252,8 +275,9 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_omega_alg_')
|
||||
goto 9999
|
||||
end if
|
||||
if (do_timings) call psb_toc(idx_phase1)
|
||||
|
||||
|
||||
if (do_timings) call psb_tic(idx_phase2)
|
||||
call acsrf%scal(adiag,info)
|
||||
if (info /= psb_success_) goto 9999
|
||||
|
||||
@@ -267,6 +291,8 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
call psb_cdasb(desc_ac,info)
|
||||
call psb_cd_reinit(desc_ac,info)
|
||||
if (do_timings) call psb_toc(idx_phase2)
|
||||
if (do_timings) call psb_tic(idx_phase3)
|
||||
!
|
||||
! Build the smoothed prolongator using either A or Af
|
||||
! acsr1 = (I-w*D*A) Prol acsr1 = (I-w*D*Af) Prol
|
||||
@@ -279,8 +305,8 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spspmm 1')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (do_timings) call psb_toc(idx_phase3)
|
||||
if (do_timings) call psb_tic(idx_ptap)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done SPSPMM 1'
|
||||
@@ -292,7 +318,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
call op_prol%mv_from(coo_prol)
|
||||
call op_restr%mv_from(coo_restr)
|
||||
|
||||
if (do_timings) call psb_toc(idx_ptap)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
|
||||
@@ -97,6 +97,8 @@ subroutine amg_d_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
integer(psb_lpk_) :: ntaggr
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
logical :: clean_zeros
|
||||
integer(psb_ipk_), save :: idx_map_bld=-1, idx_map_tprol=-1
|
||||
logical, parameter :: do_timings=.false.
|
||||
|
||||
name='amg_d_dec_aggregator_tprol'
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -108,6 +110,10 @@ subroutine amg_d_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
if ((do_timings).and.(idx_map_bld==-1)) &
|
||||
& idx_map_bld = psb_get_timer_idx("DEC_TPROL: map_bld")
|
||||
if ((do_timings).and.(idx_map_tprol==-1)) &
|
||||
& idx_map_tprol = psb_get_timer_idx("DEC_TPROL: map_tprol")
|
||||
|
||||
call amg_check_def(parms%ml_cycle,'Multilevel cycle',&
|
||||
& amg_mult_ml_,is_legal_ml_cycle)
|
||||
@@ -121,10 +127,14 @@ subroutine amg_d_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
! The decoupled aggregator based on SOC measures ignores
|
||||
! ag_data except for clean_zeros; soc_map_bld is a procedure pointer.
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_map_bld)
|
||||
clean_zeros = ag%do_clean_zeros
|
||||
call ag%soc_map_bld(parms%aggr_ord,parms%aggr_thresh,clean_zeros,a,desc_a,nlaggr,ilaggr,info)
|
||||
if (do_timings) call psb_toc(idx_map_bld)
|
||||
if (do_timings) call psb_tic(idx_map_tprol)
|
||||
|
||||
if (info==psb_success_) call amg_map_to_tprol(desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
if (do_timings) call psb_toc(idx_map_tprol)
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
call psb_errpush(info,name,a_err='soc_map_bld/map_to_tprol')
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
!
|
||||
!
|
||||
! AMG4PSBLAS version 1.0
|
||||
! Algebraic Multigrid Package
|
||||
! based on PSBLAS (Parallel Sparse BLAS version 3.7)
|
||||
!
|
||||
! (C) Copyright 2021
|
||||
!
|
||||
! Salvatore Filippone
|
||||
! Pasqua D'Ambra
|
||||
! Fabio Durastante
|
||||
!
|
||||
! Redistribution and use in source and binary forms, with or without
|
||||
! modification, are permitted provided that the following conditions
|
||||
! are met:
|
||||
! 1. Redistributions of source code must retain the above copyright
|
||||
! notice, this list of conditions and the following disclaimer.
|
||||
! 2. Redistributions in binary form must reproduce the above copyright
|
||||
! notice, this list of conditions, and the following disclaimer in the
|
||||
! documentation and/or other materials provided with the distribution.
|
||||
! 3. The name of the AMG4PSBLAS group or the names of its contributors may
|
||||
! not be used to endorse or promote products derived from this
|
||||
! software without specific written permission.
|
||||
!
|
||||
! THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
! ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
|
||||
! TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
! PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AMG4PSBLAS GROUP OR ITS CONTRIBUTORS
|
||||
! BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
! CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
! SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
! INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
! CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
! File: amg_d_newmatch_aggregator_mat_asb.f90
|
||||
!
|
||||
! Subroutine: amg_d_newmatch_aggregator_mat_asb
|
||||
! Version: real
|
||||
!
|
||||
!
|
||||
! From a given AC to final format, generating DESC_AC.
|
||||
! This is quite involved, because in the context of aggregation based
|
||||
! on parallel matching we are building the matrix hierarchy within BLD_TPROL
|
||||
! as we go, especially if we have multiple sweeps, hence this code is called
|
||||
! in two completely different contexts:
|
||||
! 1. Within bld_tprol for the internal hierarchy
|
||||
! 2. Outside, from amg_hierarchy_bld
|
||||
! The solution we have found is for bld_tprol to copy its output
|
||||
! into special components ag%ac ag%desc_ac etc so that:
|
||||
! 1. if they are allocated, it means that bld_tprol has been already invoked, we are in
|
||||
! amg_hierarchy_bld and we only need to copy them
|
||||
! 2. If they are not allocated, we are within bld_tprol, and we need to actually
|
||||
! perform the various needed steps.
|
||||
!
|
||||
! Arguments:
|
||||
! ag - type(amg_d_newmatch_aggregator_type), input/output.
|
||||
! The aggregator object
|
||||
! parms - type(amg_dml_parms), input
|
||||
! The aggregation parameters
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
! desc_a - type(psb_desc_type), input.
|
||||
! The communication descriptor of the fine-level matrix.
|
||||
! The 'one-level' data structure that will contain the local
|
||||
! part of the matrix to be built as well as the information
|
||||
! concerning the prolongator and its transpose.
|
||||
! ilaggr - integer, dimension(:), input
|
||||
! The mapping between the row indices of the coarse-level
|
||||
! matrix and the row indices of the fine-level matrix.
|
||||
! ilaggr(i)=j means that node i in the adjacency graph
|
||||
! of the fine-level matrix is mapped onto node j in the
|
||||
! adjacency graph of the coarse-level matrix. Note that the indices
|
||||
! are assumed to be shifted so as to make sure the ranges on
|
||||
! the various processes do not overlap.
|
||||
! nlaggr - integer, dimension(:) input
|
||||
! nlaggr(i) contains the aggregates held by process i.
|
||||
! ac - type(psb_dspmat_type), inout
|
||||
! The coarse matrix
|
||||
! desc_ac - type(psb_desc_type), output.
|
||||
! The communication descriptor of the fine-level matrix.
|
||||
! The 'one-level' data structure that will contain the local
|
||||
! part of the matrix to be built as well as the information
|
||||
! concerning the prolongator and its transpose.
|
||||
!
|
||||
! op_prol - type(psb_dspmat_type), input/output
|
||||
! The tentative prolongator on input, the computed prolongator on output
|
||||
!
|
||||
! op_restr - type(psb_dspmat_type), input/output
|
||||
! The restrictor operator; normally, it is the transpose of the prolongator.
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_d_newmatch_aggregator_inner_mat_asb(ag,parms,a,desc_a,&
|
||||
& ac,desc_ac, op_prol,op_restr,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
#if defined(SERIAL_MPI)
|
||||
use amg_d_newmatch_aggregator_mod
|
||||
#else
|
||||
use amg_d_newmatch_aggregator_mod, amg_protect_name => amg_d_newmatch_aggregator_inner_mat_asb
|
||||
#endif
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(in) :: desc_a
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol,op_restr
|
||||
type(psb_dspmat_type), intent(inout) :: ac
|
||||
type(psb_desc_type), intent(inout) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
!
|
||||
type(psb_ctxt_type) :: ictxt
|
||||
integer(psb_ipk_) :: np, me
|
||||
type(psb_ld_coo_sparse_mat) :: acoo, bcoo
|
||||
type(psb_ld_csr_sparse_mat) :: acsr1
|
||||
integer(psb_ipk_) :: nzl, inl
|
||||
integer(psb_lpk_) :: ntaggr
|
||||
integer(psb_ipk_) :: err_act, debug_level, debug_unit
|
||||
character(len=20) :: name='d_newmatch_inner_mat_asb'
|
||||
character(len=80) :: aname
|
||||
logical, parameter :: debug=.false., dump_prol_restr=.false.
|
||||
|
||||
|
||||
if (psb_get_errstatus().ne.0) return
|
||||
call psb_erractionsave(err_act)
|
||||
debug_unit = psb_get_debug_unit()
|
||||
debug_level = psb_get_debug_level()
|
||||
info = psb_success_
|
||||
ictxt = desc_a%get_context()
|
||||
call psb_info(ictxt,me,np)
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
if (debug) write(0,*) me,' ',trim(name),' Start:',&
|
||||
& allocated(ag%ac),allocated(ag%desc_ac), allocated(ag%prol),allocated(ag%restr)
|
||||
|
||||
select case(parms%coarse_mat)
|
||||
|
||||
case(amg_distr_mat_)
|
||||
! Do nothing, it has already been done in spmm_bld_ov.
|
||||
|
||||
case(amg_repl_mat_)
|
||||
!
|
||||
!
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='no repl coarse_mat_ here')
|
||||
goto 9999
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='invalid amg_coarse_mat_')
|
||||
goto 9999
|
||||
end select
|
||||
#endif
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
|
||||
end subroutine amg_d_newmatch_aggregator_inner_mat_asb
|
||||
@@ -0,0 +1,150 @@
|
||||
!
|
||||
!
|
||||
! File: amg_d_newmatch_aggregator_mat_asb.f90
|
||||
!
|
||||
! Subroutine: amg_d_newmatch_aggregator_mat_asb
|
||||
! Version: real
|
||||
!
|
||||
!
|
||||
! From a given AC to final format, generating DESC_AC
|
||||
!
|
||||
! Arguments:
|
||||
! ag - type(amg_d_newmatch_aggregator_type), input/output.
|
||||
! The aggregator object
|
||||
! parms - type(amg_dml_parms), input
|
||||
! The aggregation parameters
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
! desc_a - type(psb_desc_type), input.
|
||||
! The communication descriptor of the fine-level matrix.
|
||||
! The 'one-level' data structure that will contain the local
|
||||
! part of the matrix to be built as well as the information
|
||||
! concerning the prolongator and its transpose.
|
||||
! ilaggr - integer, dimension(:), input
|
||||
! The mapping between the row indices of the coarse-level
|
||||
! matrix and the row indices of the fine-level matrix.
|
||||
! ilaggr(i)=j means that node i in the adjacency graph
|
||||
! of the fine-level matrix is mapped onto node j in the
|
||||
! adjacency graph of the coarse-level matrix. Note that the indices
|
||||
! are assumed to be shifted so as to make sure the ranges on
|
||||
! the various processes do not overlap.
|
||||
! nlaggr - integer, dimension(:) input
|
||||
! nlaggr(i) contains the aggregates held by process i.
|
||||
! ac - type(psb_dspmat_type), inout
|
||||
! The coarse matrix
|
||||
! desc_ac - type(psb_desc_type), output.
|
||||
! The communication descriptor of the fine-level matrix.
|
||||
! The 'one-level' data structure that will contain the local
|
||||
! part of the matrix to be built as well as the information
|
||||
! concerning the prolongator and its transpose.
|
||||
!
|
||||
! op_prol - type(psb_dspmat_type), input/output
|
||||
! The tentative prolongator on input, the computed prolongator on output
|
||||
!
|
||||
! op_restr - type(psb_dspmat_type), input/output
|
||||
! The restrictor operator; normally, it is the transpose of the prolongator.
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_d_newmatch_aggregator_mat_asb(ag,parms,a,desc_a,&
|
||||
& ac,desc_ac, op_prol,op_restr,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_newmatch_aggregator_mod, amg_protect_name => amg_d_newmatch_aggregator_mat_asb
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol, ac,op_restr
|
||||
type(psb_desc_type), intent(inout) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
!
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me
|
||||
type(psb_ld_coo_sparse_mat) :: tmpcoo
|
||||
type(psb_ldspmat_type) :: tmp_ac
|
||||
integer(psb_ipk_) :: i_nr, i_nc, i_nl, nzl
|
||||
integer(psb_lpk_) :: ntaggr
|
||||
integer(psb_ipk_) :: err_act, debug_level, debug_unit
|
||||
character(len=20) :: name='d_newmatch_aggregator_mat_asb'
|
||||
|
||||
|
||||
if (psb_get_errstatus().ne.0) return
|
||||
call psb_erractionsave(err_act)
|
||||
debug_unit = psb_get_debug_unit()
|
||||
debug_level = psb_get_debug_level()
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
|
||||
select case(parms%coarse_mat)
|
||||
|
||||
case(amg_distr_mat_)
|
||||
|
||||
call ac%cscnv(info,type='csr')
|
||||
call op_prol%cscnv(info,type='csr')
|
||||
call op_restr%cscnv(info,type='csr')
|
||||
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done ac '
|
||||
|
||||
case(amg_repl_mat_)
|
||||
!
|
||||
! We are assuming here that an d matrix
|
||||
! can hold all entries
|
||||
!
|
||||
if (desc_ac%get_global_rows() < huge(1_psb_ipk_) ) then
|
||||
ntaggr = desc_ac%get_global_rows()
|
||||
i_nr = ntaggr
|
||||
else
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='invalid amg_coarse_mat_')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
call op_prol%mv_to(tmpcoo)
|
||||
nzl = tmpcoo%get_nzeros()
|
||||
call psb_loc_to_glob(tmpcoo%ja(1:nzl),desc_ac,info,'I')
|
||||
call op_prol%mv_from(tmpcoo)
|
||||
|
||||
call op_restr%mv_to(tmpcoo)
|
||||
nzl = tmpcoo%get_nzeros()
|
||||
call psb_loc_to_glob(tmpcoo%ia(1:nzl),desc_ac,info,'I')
|
||||
call op_restr%mv_from(tmpcoo)
|
||||
|
||||
call op_prol%set_ncols(i_nr)
|
||||
call op_restr%set_nrows(i_nr)
|
||||
|
||||
call psb_gather(tmp_ac,ac,desc_ac,info,root=-ione,&
|
||||
& dupl=psb_dupl_add_,keeploc=.false.)
|
||||
call tmp_ac%mv_to(tmpcoo)
|
||||
call ac%mv_from(tmpcoo)
|
||||
|
||||
call psb_cdall(ctxt,desc_ac,info,mg=ntaggr,repl=.true.)
|
||||
if (info == psb_success_) call psb_cdasb(desc_ac,info)
|
||||
!
|
||||
! Now that we have the descriptors and the restrictor, we should
|
||||
! update the W. But we don't, because REPL is only valid
|
||||
! at the coarsest level, so no need to carry over.
|
||||
!
|
||||
|
||||
if (info /= psb_success_) goto 9999
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='invalid amg_coarse_mat_')
|
||||
goto 9999
|
||||
end select
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
|
||||
end subroutine amg_d_newmatch_aggregator_mat_asb
|
||||
@@ -0,0 +1,183 @@
|
||||
!
|
||||
!
|
||||
! File: amg_d_base_aggregator_mat_bld.f90
|
||||
!
|
||||
! Subroutine: amg_d_base_aggregator_mat_bld
|
||||
! Version: real
|
||||
!
|
||||
! This routine builds the matrix associated to the current level of the
|
||||
! multilevel preconditioner from the matrix associated to the previous level,
|
||||
! by using the user-specified aggregation technique (therefore, it also builds the
|
||||
! prolongation and restriction operators mapping the current level to the
|
||||
! previous one and vice versa).
|
||||
! The current level is regarded as the coarse one, while the previous as
|
||||
! the fine one. This is in agreement with the fact that the routine is called,
|
||||
! by amg_mlprec_bld, only on levels >=2.
|
||||
! The coarse-level matrix A_C is built from a fine-level matrix A
|
||||
! by using the Galerkin approach, i.e.
|
||||
!
|
||||
! A_C = P_C^T A P_C,
|
||||
!
|
||||
! where P_C is a prolongator from the coarse level to the fine one.
|
||||
!
|
||||
! A mapping from the nodes of the adjacency graph of A to the nodes of the
|
||||
! adjacency graph of A_C has been computed by the amg_aggrmap_bld subroutine.
|
||||
! The prolongator P_C is built here from this mapping, according to the
|
||||
! value of p%iprcparm(amg_aggr_kind_), specified by the user through
|
||||
! amg_dprecinit and amg_zprecset.
|
||||
! On output from this routine the entries of AC, op_prol, op_restr
|
||||
! are still in "global numbering" mode; this is fixed in the calling routine
|
||||
! amg_d_lev_aggrmat_bld.
|
||||
!
|
||||
! Currently four different prolongators are implemented, corresponding to
|
||||
! four aggregation algorithms:
|
||||
! 1. un-smoothed aggregation,
|
||||
! 2. smoothed aggregation,
|
||||
! 3. "bizarre" aggregation.
|
||||
! 4. minimum energy
|
||||
! 1. The non-smoothed aggregation uses as prolongator the piecewise constant
|
||||
! interpolation operator corresponding to the fine-to-coarse level mapping built
|
||||
! by p%aggr%bld_tprol. This is called tentative prolongator.
|
||||
! 2. The smoothed aggregation uses as prolongator the operator obtained by applying
|
||||
! a damped Jacobi smoother to the tentative prolongator.
|
||||
! 3. The "bizarre" aggregation uses a prolongator proposed by the authors of MLD2P4.
|
||||
! This prolongator still requires a deep analysis and testing and its use is
|
||||
! not recommended.
|
||||
! 4. Minimum energy aggregation
|
||||
!
|
||||
! For more details see
|
||||
! M. Brezina and P. Vanek, A black-box iterative solver based on a two-level
|
||||
! Schwarz method, Computing, 63 (1999), 233-263.
|
||||
! P. D'Ambra, D. di Serafino and S. Filippone, On the development of PSBLAS-based
|
||||
! parallel two-level Schwarz preconditioners, Appl. Num. Math., 57 (2007),
|
||||
! 1181-1196.
|
||||
! M. Sala, R. Tuminaro: A new Petrov-Galerkin smoothed aggregation preconditioner
|
||||
! for nonsymmetric linear systems, SIAM J. Sci. Comput., 31(1):143-166 (2008)
|
||||
!
|
||||
!
|
||||
! The main structure is:
|
||||
! 1. Perform sanity checks;
|
||||
! 2. Compute prolongator/restrictor/AC
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! ag - type(amg_d_base_aggregator_type), input/output.
|
||||
! The aggregator object
|
||||
! parms - type(amg_dml_parms), input
|
||||
! The aggregation parameters
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
! desc_a - type(psb_desc_type), input.
|
||||
! The communication descriptor of the fine-level matrix.
|
||||
! The 'one-level' data structure that will contain the local
|
||||
! part of the matrix to be built as well as the information
|
||||
! concerning the prolongator and its transpose.
|
||||
! ilaggr - integer, dimension(:), input
|
||||
! The mapping between the row indices of the coarse-level
|
||||
! matrix and the row indices of the fine-level matrix.
|
||||
! ilaggr(i)=j means that node i in the adjacency graph
|
||||
! of the fine-level matrix is mapped onto node j in the
|
||||
! adjacency graph of the coarse-level matrix. Note that the indices
|
||||
! are assumed to be shifted so as to make sure the ranges on
|
||||
! the various processes do not overlap.
|
||||
! nlaggr - integer, dimension(:) input
|
||||
! nlaggr(i) contains the aggregates held by process i.
|
||||
! ac - type(psb_dspmat_type), output
|
||||
! The coarse matrix on output
|
||||
!
|
||||
! op_prol - type(psb_dspmat_type), input/output
|
||||
! The tentative prolongator on input, the computed prolongator on output
|
||||
!
|
||||
! op_restr - type(psb_dspmat_type), output
|
||||
! The restrictor operator; normally, it is the transpose of the prolongator.
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_d_newmatch_aggregator_mat_bld(ag,parms,a,desc_a,ilaggr,nlaggr,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_d_inner_mod
|
||||
use amg_d_prec_type, amg_protect_name => amg_d_newmatch_aggregator_mat_bld
|
||||
!use amg_d_newmatch_aggregator_mod, amg_protect_name => amg_d_newmatch_aggregator_mat_bld
|
||||
implicit none
|
||||
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
type(psb_ldspmat_type), intent(inout) :: t_prol
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol,ac,op_restr
|
||||
type(psb_desc_type), intent(inout) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
character(len=20) :: name
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_mpk_) :: np, me
|
||||
type(psb_ld_coo_sparse_mat) :: acoo, bcoo
|
||||
type(psb_ld_csr_sparse_mat) :: acsr1
|
||||
integer(psb_lpk_) :: nzl,ntaggr
|
||||
integer(psb_ipk_) :: err_act
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
|
||||
name='amg_d_newmatch_aggregator_mat_bld'
|
||||
if (psb_get_errstatus().ne.0) return
|
||||
call psb_erractionsave(err_act)
|
||||
debug_unit = psb_get_debug_unit()
|
||||
debug_level = psb_get_debug_level()
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
|
||||
!
|
||||
! Build the coarse-level matrix from the fine-level one, starting from
|
||||
! the mapping defined by amg_aggrmap_bld and applying the aggregation
|
||||
! algorithm specified by
|
||||
!
|
||||
select case (parms%aggr_prol)
|
||||
case (amg_no_smooth_)
|
||||
|
||||
!!$ call amg_d_newmatch_unsmth_spmm_bld(a,desc_a,ilaggr,nlaggr,&
|
||||
!!$ & parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_daggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
|
||||
case(amg_smooth_prol_)
|
||||
|
||||
call amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
!!$ case(amg_biz_prol_)
|
||||
!!$
|
||||
!!$ call amg_daggrmat_biz_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
!!$ & parms,ac,desc_ac,op_prol,op_restr,info)
|
||||
|
||||
case(amg_min_energy_)
|
||||
|
||||
call amg_daggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Invalid aggr kind')
|
||||
goto 9999
|
||||
|
||||
end select
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='Inner aggrmat bld')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
|
||||
end subroutine amg_d_newmatch_aggregator_mat_bld
|
||||
@@ -0,0 +1,448 @@
|
||||
!
|
||||
!
|
||||
! File: amg_d_newmatch_aggregator_tprol.f90
|
||||
!
|
||||
! Subroutine: amg_d_newmatch_aggregator_tprol
|
||||
! Version: real
|
||||
!
|
||||
!
|
||||
|
||||
subroutine amg_d_newmatch_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
& a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_inner_mod
|
||||
use amg_d_decmatch_mod
|
||||
#if defined(SERIAL_MPI)
|
||||
use amg_d_newmatch_aggregator_mod
|
||||
#else
|
||||
use amg_d_newmatch_aggregator_mod, amg_protect_name => amg_d_newmatch_aggregator_build_tprol
|
||||
#endif
|
||||
use iso_c_binding
|
||||
implicit none
|
||||
class(amg_d_newmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(amg_daggr_data), intent(in) :: ag_data
|
||||
type(psb_dspmat_type), intent(inout) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), allocatable, intent(out) :: ilaggr(:),nlaggr(:)
|
||||
type(psb_ldspmat_type), intent(out) :: t_prol
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
|
||||
! Local variables
|
||||
real(psb_dpk_), allocatable :: tmpw(:), tmpwnxt(:)
|
||||
integer(psb_lpk_), allocatable :: ixaggr(:), nxaggr(:), tlaggr(:), ivr(:)
|
||||
type(psb_dspmat_type) :: a_tmp
|
||||
type(nwm_CSRMatrix) :: C, P
|
||||
integer(c_int) :: match_algorithm, n_sweeps, max_csize, max_nlevels
|
||||
character(len=40) :: name, ch_err
|
||||
character(len=80) :: fname, prefix_
|
||||
type(psb_ctxt_type) :: ictxt
|
||||
integer(psb_ipk_) :: np, me
|
||||
integer(psb_ipk_) :: err_act, ierr
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
integer(psb_ipk_) :: i, j, k, nr, nc
|
||||
integer(psb_lpk_) :: isz, num_pcols, nrac, ncac, lname, nz, x_sweeps, csz
|
||||
integer(psb_lpk_) :: psz, sizes(4)
|
||||
type(psb_d_csr_sparse_mat), target :: csr_prol, csr_pvi, csr_prod_res, acsr
|
||||
type(psb_ld_csr_sparse_mat), target :: lcsr_prol
|
||||
type(psb_desc_type), allocatable :: desc_acv(:)
|
||||
type(psb_ld_coo_sparse_mat) :: tmpcoo, transp_coo
|
||||
type(psb_dspmat_type), allocatable :: acv(:)
|
||||
type(psb_dspmat_type), allocatable :: prolv(:), restrv(:)
|
||||
type(psb_ldspmat_type) :: tmp_prol, tmp_pg, tmp_restr
|
||||
type(psb_desc_type) :: tmp_desc_ac, tmp_desc_ax, tmp_desc_p
|
||||
integer(psb_ipk_), save :: idx_mboxp=-1, idx_spmmbld=-1, idx_sweeps_mult=-1
|
||||
logical, parameter :: dump=.false., do_timings=.true., debug=.false., &
|
||||
& dump_prol_restr=.false.
|
||||
name='d_newmatch_tprol'
|
||||
ictxt = desc_a%get_context()
|
||||
call psb_info(ictxt,me,np)
|
||||
if (psb_get_errstatus().ne.0) then
|
||||
write(0,*) me,trim(name),' Err_status :',psb_get_errstatus()
|
||||
return
|
||||
end if
|
||||
if (debug) write(0,*) me,trim(name),' Start '
|
||||
call psb_erractionsave(err_act)
|
||||
debug_unit = psb_get_debug_unit()
|
||||
debug_level = psb_get_debug_level()
|
||||
info = psb_success_
|
||||
|
||||
if ((do_timings).and.(idx_mboxp==-1)) &
|
||||
& idx_mboxp = psb_get_timer_idx("PMC_TPROL: MatchBoxP")
|
||||
if ((do_timings).and.(idx_spmmbld==-1)) &
|
||||
& idx_spmmbld = psb_get_timer_idx("PMC_TPROL: spmm_bld")
|
||||
if ((do_timings).and.(idx_sweeps_mult==-1)) &
|
||||
& idx_sweeps_mult = psb_get_timer_idx("PMC_TPROL: sweeps_mult")
|
||||
|
||||
|
||||
call amg_check_def(parms%ml_cycle,'Multilevel cycle',&
|
||||
& amg_mult_ml_,is_legal_ml_cycle)
|
||||
call amg_check_def(parms%par_aggr_alg,'Aggregation',&
|
||||
& amg_coupled_aggr_,is_legal_decoupled_par_aggr_alg)
|
||||
call amg_check_def(parms%aggr_ord,'Ordering',&
|
||||
& amg_aggr_ord_nat_,is_legal_ml_aggr_ord)
|
||||
call amg_check_def(parms%aggr_thresh,'Aggr_Thresh',dzero,is_legal_d_aggr_thrs)
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
match_algorithm = ag%matching_alg
|
||||
n_sweeps = ag%n_sweeps
|
||||
if (2**n_sweeps /= ag%orig_aggr_size) then
|
||||
if (me == 0) then
|
||||
write(debug_unit, *) 'Warning: AGGR_SIZE reset to value ',2**n_sweeps
|
||||
end if
|
||||
end if
|
||||
if (ag%max_csize > 0) then
|
||||
max_csize = ag%max_csize
|
||||
else
|
||||
max_csize = ag_data%min_coarse_size
|
||||
end if
|
||||
if (ag%max_nlevels > 0) then
|
||||
max_nlevels = ag%max_nlevels
|
||||
else
|
||||
max_nlevels = ag_data%max_levs
|
||||
end if
|
||||
if (.true.) then
|
||||
block
|
||||
integer(psb_ipk_) :: ipv(2)
|
||||
ipv(1) = max_csize
|
||||
ipv(2) = n_sweeps
|
||||
call psb_bcast(ictxt,ipv)
|
||||
max_csize = ipv(1)
|
||||
n_sweeps = ipv(2)
|
||||
end block
|
||||
else
|
||||
call psb_bcast(ictxt,max_csize)
|
||||
call psb_bcast(ictxt,n_sweeps)
|
||||
end if
|
||||
if (n_sweeps /= ag%n_sweeps) then
|
||||
write(0,*) me,' Inconsistent N_SWEEPS ',n_sweeps,ag%n_sweeps
|
||||
end if
|
||||
n_sweeps = max(1,n_sweeps)
|
||||
|
||||
if (debug) write(0,*) me,' Copies, with n_sweeps: ',n_sweeps,max_csize
|
||||
if (ag%unsmoothed_hierarchy.and.allocated(ag%base_a)) then
|
||||
call ag%base_a%cp_to(acsr)
|
||||
if (ag%do_clean_zeros) call acsr%clean_zeros(info)
|
||||
nr = acsr%get_nrows()
|
||||
if (psb_size(ag%w) < nr) call ag%bld_default_w(nr)
|
||||
isz = acsr%get_ncols()
|
||||
|
||||
call psb_realloc(isz,ixaggr,info)
|
||||
if (info == psb_success_) &
|
||||
& allocate(acv(0:n_sweeps), desc_acv(0:n_sweeps),&
|
||||
& prolv(n_sweeps), restrv(n_sweeps),stat=info)
|
||||
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
ch_err='psb_realloc'
|
||||
call psb_errpush(info,name,a_err=ch_err)
|
||||
goto 9999
|
||||
|
||||
end if
|
||||
|
||||
|
||||
call acv(0)%mv_from(acsr)
|
||||
call ag%base_desc%clone(desc_acv(0),info)
|
||||
|
||||
else
|
||||
call a%cp_to(acsr)
|
||||
if (ag%do_clean_zeros) call acsr%clean_zeros(info)
|
||||
nr = acsr%get_nrows()
|
||||
if (psb_size(ag%w) < nr) call ag%bld_default_w(nr)
|
||||
isz = acsr%get_ncols()
|
||||
|
||||
call psb_realloc(isz,ixaggr,info)
|
||||
if (info == psb_success_) &
|
||||
& allocate(acv(0:n_sweeps), desc_acv(0:n_sweeps),&
|
||||
& prolv(n_sweeps), restrv(n_sweeps),stat=info)
|
||||
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
ch_err='psb_realloc'
|
||||
call psb_errpush(info,name,a_err=ch_err)
|
||||
goto 9999
|
||||
|
||||
end if
|
||||
|
||||
|
||||
call acv(0)%mv_from(acsr)
|
||||
call desc_a%clone(desc_acv(0),info)
|
||||
end if
|
||||
|
||||
nrac = desc_acv(0)%get_local_rows()
|
||||
ncac = desc_acv(0)%get_local_cols()
|
||||
if (debug) write(0,*) me,' On input to level: ',nrac, ncac
|
||||
if (allocated(ag%prol)) then
|
||||
call ag%prol%free()
|
||||
deallocate(ag%prol)
|
||||
end if
|
||||
if (allocated(ag%restr)) then
|
||||
call ag%restr%free()
|
||||
deallocate(ag%restr)
|
||||
end if
|
||||
|
||||
if (dump) then
|
||||
block
|
||||
type(psb_ldspmat_type) :: lac
|
||||
ivr = desc_acv(0)%get_global_indices(owned=.false.)
|
||||
prefix_ = "input_a"
|
||||
lname = len_trim(prefix_)
|
||||
fname = trim(prefix_)
|
||||
write(fname(lname+1:lname+9),'(a,i3.3,a)') '_p',me, '.mtx'
|
||||
call acv(0)%print(fname,head='Debug aggregates')
|
||||
call lac%cp_from(acv(0))
|
||||
write(fname(lname+1:lname+13),'(a,i3.3,a)') '_p',me, '-glb.mtx'
|
||||
call lac%print(fname,head='Debug aggregates',iv=ivr)
|
||||
call lac%free()
|
||||
end block
|
||||
end if
|
||||
|
||||
call psb_geall(tmpw,desc_acv(0),info)
|
||||
|
||||
tmpw(1:nr) = ag%w(1:nr)
|
||||
|
||||
call psb_geasb(tmpw,desc_acv(0),info)
|
||||
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me == 0) write(0,*) 'N_sweeps ',n_sweeps,nr,desc_acv(0)%is_ok(),max_csize
|
||||
end if
|
||||
|
||||
!
|
||||
! Prepare ag%ac, ag%desc_ac, ag%prol, ag%restr to enable
|
||||
! shortcuts in mat_bld and mat_asb
|
||||
! and ag%desc_ax which will be needed in backfix.
|
||||
!
|
||||
x_sweeps = -1
|
||||
sweeps_loop: do i=1, n_sweeps
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me==0) write(0,*) me,trim(name),' Start sweeps_loop iteration:',i,' of ',n_sweeps
|
||||
end if
|
||||
|
||||
!
|
||||
! Building prol and restr because this algorithm is not decoupled
|
||||
! On exit from matchbox_build_prol, prolv(i) is in global numbering
|
||||
!
|
||||
!
|
||||
if (debug) write(0,*) me,' Into matchbox_build_prol ',info
|
||||
if (do_timings) call psb_tic(idx_mboxp)
|
||||
call amg_ddecmatch_build_prol(tmpw,acv(i-1),desc_acv(i-1),ixaggr,nxaggr,tmp_prol,info,&
|
||||
& symmetrize=ag%need_symmetrize,reproducible=ag%reproducible_matching,&
|
||||
& parallel=ag%parallel_matching,matching=ag%matching_alg,lambda=ag%lambda)
|
||||
if (do_timings) call psb_toc(idx_mboxp)
|
||||
if (debug) write(0,*) me,' Out from matchbox_build_prol ',info
|
||||
if (psb_errstatus_fatal()) write(0,*)me,trim(name),'Error fatal on exit bld_tprol',info
|
||||
|
||||
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
!!$ write(0,*) name,' Call spmm_bld sweep:',i,n_sweeps
|
||||
if (me==0) write(0,*) me,trim(name),' Calling spmm_bld NSW>1:',i,&
|
||||
& desc_acv(i-1)%get_local_rows(),desc_acv(i-1)%get_local_cols(),&
|
||||
& desc_acv(i-1)%get_global_rows()
|
||||
end if
|
||||
if (i == n_sweeps) call tmp_prol%clone(tmp_pg,info)
|
||||
if (do_timings) call psb_tic(idx_spmmbld)
|
||||
!
|
||||
! On entry, prolv(i) is in global numbering,
|
||||
!
|
||||
call amg_d_newmatch_spmm_bld_ov(acv(i-1),desc_acv(i-1),ixaggr,nxaggr,parms,&
|
||||
& acv(i),desc_acv(i), prolv(i),restrv(1),tmp_prol,info)
|
||||
if (psb_errstatus_fatal()) write(0,*)me,trim(name),'Error fatal on exit from bld_ov(i)',info
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me==0) write(0,*) me,trim(name),' Done spmm_bld:',i
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_spmmbld)
|
||||
! Keep a copy of prolv(i) in global numbering for the time being, will
|
||||
! need it to build the final
|
||||
! if (i == n_sweeps) call prolv(i)%clone(tmp_prol,info)
|
||||
call ag%inner_mat_asb(parms,acv(i-1),desc_acv(i-1),&
|
||||
& acv(i),desc_acv(i),prolv(i),restrv(1),info)
|
||||
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me==0) write(0,*) me,trim(name),' Done mat_asb:',i,sum(nxaggr),max_csize,info
|
||||
csz = sum(nxaggr)
|
||||
call psb_bcast(ictxt,csz)
|
||||
if (csz /= sum(nxaggr)) write(0,*) me,trim(name),' Mismatch matasb',&
|
||||
& csz,sum(nxaggr),max_csize
|
||||
end if
|
||||
if (psb_errstatus_fatal()) write(0,*)me,trim(name),'Error fatal on entry to tmpwnxt 2'
|
||||
|
||||
|
||||
!
|
||||
! Fix wnxt
|
||||
!
|
||||
if (info == 0) call psb_geall(tmpwnxt,desc_acv(i),info)
|
||||
if (info == 0) call psb_geasb(tmpwnxt,desc_acv(i),info,scratch=.true.)
|
||||
if (info == 0) call psb_halo(tmpw,desc_acv(i-1),info)
|
||||
!!$ write(0,*) trestr%get_nrows(),size(tmpwnxt),trestr%get_ncols(),size(tmpw)
|
||||
|
||||
if (info == 0) call psb_csmm(done,restrv(1),tmpw,dzero,tmpwnxt,info)
|
||||
|
||||
if (info /= psb_success_) then
|
||||
write(0,*)me,trim(name),'Error from mat_asb/tmpw ',info
|
||||
info=psb_err_from_subroutine_
|
||||
call psb_errpush(info,name,a_err='mat_asb 2')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (i == 1) then
|
||||
nrac = desc_acv(1)%get_local_rows()
|
||||
!!$ write(0,*) 'Copying output w_nxt ',nrac
|
||||
call psb_realloc(nrac,ag%w_nxt,info)
|
||||
ag%w_nxt(1:nrac) = tmpwnxt(1:nrac)
|
||||
!
|
||||
! ILAGGR is fixed later on, but
|
||||
! get a copy in case of an early exit
|
||||
!
|
||||
call psb_safe_ab_cpy(ixaggr,ilaggr,info)
|
||||
end if
|
||||
call psb_safe_ab_cpy(nxaggr,nlaggr,info)
|
||||
call move_alloc(tmpwnxt,tmpw)
|
||||
if (debug) then
|
||||
if (csz /= sum(nlaggr)) write(0,*) me,trim(name),' Mismatch 2 matasb',&
|
||||
& csz,sum(nlaggr),max_csize, info
|
||||
end if
|
||||
call acv(i-1)%free()
|
||||
if ((sum(nlaggr) <= max_csize).or.(any(nlaggr==0))) then
|
||||
x_sweeps = i
|
||||
exit sweeps_loop
|
||||
end if
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me==0) write(0,*) me,trim(name),' Done sweeps_loop iteration:',i,' of ',n_sweeps
|
||||
end if
|
||||
|
||||
end do sweeps_loop
|
||||
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me==0) write(0,*) me,trim(name),' Done sweeps_loop:',x_sweeps
|
||||
end if
|
||||
if (x_sweeps<=0) x_sweeps = n_sweeps
|
||||
|
||||
if (do_timings) call psb_tic(idx_sweeps_mult)
|
||||
!
|
||||
! Ok, now we have all the prolongators, including the last one in global numbering.
|
||||
! Build the product of all prolongators. Need a tmp_desc_ax
|
||||
! which is correct but most of the time overdimensioned
|
||||
!
|
||||
if (.not.allocated(ag%desc_ax)) allocate(ag%desc_ax)
|
||||
!
|
||||
block
|
||||
integer(psb_ipk_) :: i, nnz
|
||||
integer(psb_lpk_) :: ncol, ncsave
|
||||
if (.not.allocated(ag%ac)) allocate(ag%ac)
|
||||
if (.not.allocated(ag%desc_ac)) allocate(ag%desc_ac)
|
||||
call desc_acv(x_sweeps)%clone(ag%desc_ac,info)
|
||||
call desc_acv(x_sweeps)%free(info)
|
||||
call acv(x_sweeps)%move_alloc(ag%ac,info)
|
||||
if (.not.allocated(ag%prol)) allocate(ag%prol)
|
||||
if (.not.allocated(ag%restr)) allocate(ag%restr)
|
||||
|
||||
call psb_cd_reinit(ag%desc_ac,info)
|
||||
ncsave = ag%desc_ac%get_global_rows()
|
||||
!
|
||||
! Note: prolv(i) is already in local numbering
|
||||
! because of the call to mat_asb in the loop above.
|
||||
!
|
||||
call prolv(x_sweeps)%mv_to(csr_prol)
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me == 0) write(0,*) 'Enter prolongator product loop ',x_sweeps
|
||||
end if
|
||||
|
||||
do i=x_sweeps-1, 1, -1
|
||||
call prolv(i)%mv_to(csr_pvi)
|
||||
if (psb_errstatus_fatal()) write(0,*) me,' Fatal error in prolongator loop 1'
|
||||
call psb_par_spspmm(csr_pvi,desc_acv(i),csr_prol,csr_prod_res,ag%desc_ac,info)
|
||||
if ((info /=0).or.psb_errstatus_fatal()) write(0,*) me,' Fatal error in prolongator loop 2',info
|
||||
call csr_pvi%free()
|
||||
call csr_prod_res%mv_to_fmt(csr_prol,info)
|
||||
if ((info /=0).or.psb_errstatus_fatal()) write(0,*) me,' Fatal error in prolongator loop 3',info
|
||||
call csr_prol%set_ncols(ag%desc_ac%get_local_cols())
|
||||
if ((info /=0).or.psb_errstatus_fatal()) write(0,*) me,' Fatal error in prolongator loop 4'
|
||||
end do
|
||||
call csr_prol%mv_to_lfmt(lcsr_prol,info)
|
||||
nnz = lcsr_prol%get_nzeros()
|
||||
call ag%desc_ac%l2gip(lcsr_prol%ja(1:nnz),info)
|
||||
call lcsr_prol%set_ncols(ncsave)
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me == 0) write(0,*) 'Done prolongator product loop ',x_sweeps
|
||||
end if
|
||||
!
|
||||
! Fix ILAGGR here by copying from CSR_PROL%JA
|
||||
!
|
||||
block
|
||||
integer(psb_ipk_) :: nr
|
||||
nr = lcsr_prol%get_nrows()
|
||||
if (nnz /= nr) then
|
||||
write(0,*) me,name,' Issue with prolongator? ',nr,nnz
|
||||
end if
|
||||
call psb_realloc(nr,ilaggr,info)
|
||||
ilaggr(1:nnz) = lcsr_prol%ja(1:nnz)
|
||||
end block
|
||||
call tmp_prol%mv_from(lcsr_prol)
|
||||
call psb_cdasb(ag%desc_ac,info)
|
||||
call ag%ac%set_ncols(ag%desc_ac%get_local_cols())
|
||||
end block
|
||||
|
||||
call tmp_prol%move_alloc(t_prol,info)
|
||||
call t_prol%set_ncols(ag%desc_ac%get_local_cols())
|
||||
call t_prol%set_nrows(desc_acv(0)%get_local_rows())
|
||||
|
||||
nrac = ag%desc_ac%get_local_rows()
|
||||
ncac = ag%desc_ac%get_local_cols()
|
||||
call psb_realloc(nrac,ag%w_nxt,info)
|
||||
ag%w_nxt(1:nrac) = tmpw(1:nrac)
|
||||
|
||||
|
||||
if (do_timings) call psb_toc(idx_sweeps_mult)
|
||||
|
||||
if (debug) then
|
||||
call psb_barrier(ictxt)
|
||||
if (me == 0) write(0,*) 'Out of build loop ',x_sweeps,': Output size:',sum(nlaggr)
|
||||
end if
|
||||
|
||||
|
||||
!call psb_set_debug_level(0)
|
||||
if (dump) then
|
||||
block
|
||||
ivr = desc_acv(x_sweeps)%get_global_indices(owned=.false.)
|
||||
prefix_ = "final_ac"
|
||||
lname = len_trim(prefix_)
|
||||
fname = trim(prefix_)
|
||||
write(fname(lname+1:lname+9),'(a,i3.3,a)') '_p',me, '.mtx'
|
||||
call acv(x_sweeps)%print(fname,head='Debug aggregates')
|
||||
write(fname(lname+1:lname+13),'(a,i3.3,a)') '_p',me, '-glb.mtx'
|
||||
call acv(x_sweeps)%print(fname,head='Debug aggregates',iv=ivr)
|
||||
prefix_ = "final_tp"
|
||||
lname = len_trim(prefix_)
|
||||
fname = trim(prefix_)
|
||||
write(fname(lname+1:lname+9),'(a,i3.3,a)') '_p',me, '.mtx'
|
||||
call t_prol%print(fname,head='Tentative prolongator')
|
||||
end block
|
||||
end if
|
||||
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='amg_bootCMatch_if')
|
||||
goto 9999
|
||||
end if
|
||||
#endif
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_d_newmatch_aggregator_build_tprol
|
||||
|
||||
@@ -0,0 +1,128 @@
|
||||
!
|
||||
!
|
||||
! File: amg_d_newmatch_map_to_tprol.f90
|
||||
!
|
||||
! Subroutine: amg_d_newmatch_map_to_tprol
|
||||
! Version: real
|
||||
!
|
||||
! This routine uses a mapping from the row indices of the fine-level matrix
|
||||
! to the row indices of the coarse-level matrix to build a tentative
|
||||
! prolongator, i.e. a piecewise constant operator.
|
||||
! This is later used to build the final operator; the code has been refactored here
|
||||
! to be shared among all the methods that provide the tentative prolongator
|
||||
! through a simple integer mapping.
|
||||
!
|
||||
! The aggregation algorithm is a parallel version of that described in
|
||||
! * M. Brezina and P. Vanek, A black-box iterative solver based on a
|
||||
! two-level Schwarz method, Computing, 63 (1999), 233-263.
|
||||
! * P. Vanek, J. Mandel and M. Brezina, Algebraic Multigrid by Smoothed
|
||||
! Aggregation for Second and Fourth Order Elliptic Problems, Computing, 56
|
||||
! (1996), 179-196.
|
||||
! For more details see
|
||||
! P. D'Ambra, D. di Serafino and S. Filippone, On the development of
|
||||
! PSBLAS-based parallel two-level Schwarz preconditioners, Appl. Num. Math.
|
||||
! 57 (2007), 1181-1196.
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! aggr_type - integer, input.
|
||||
! The scalar used to identify the aggregation algorithm.
|
||||
! theta - real, input.
|
||||
! The aggregation threshold used in the aggregation algorithm.
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
! desc_a - type(psb_desc_type), input.
|
||||
! The communication descriptor of the fine-level matrix.
|
||||
! ilaggr - integer, dimension(:), allocatable.
|
||||
! The mapping between the row indices of the coarse-level
|
||||
! matrix and the row indices of the fine-level matrix.
|
||||
! ilaggr(i)=j means that node i in the adjacency graph
|
||||
! of the fine-level matrix is mapped onto node j in the
|
||||
! adjacency graph of the coarse-level matrix. Note that on exit the indices
|
||||
! will be shifted so as to make sure the ranges on the various processes do not
|
||||
! overlap.
|
||||
! nlaggr - integer, dimension(:), allocatable.
|
||||
! nlaggr(i) contains the aggregates held by process i.
|
||||
! op_prol - type(psb_dspmat_type).
|
||||
! The tentative prolongator, based on ilaggr.
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_d_newmatch_map_to_tprol(desc_a,ilaggr,nlaggr,valaggr, op_prol,info)
|
||||
|
||||
use psb_base_mod
|
||||
use amg_d_inner_mod!, amg_protect_name => amg_d_newmatch_map_to_tprol
|
||||
use amg_d_newmatch_aggregator_mod, amg_protect_name => amg_d_newmatch_map_to_tprol
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
type(psb_desc_type), intent(in) :: desc_a
|
||||
integer(psb_lpk_), allocatable, intent(inout) :: ilaggr(:),nlaggr(:)
|
||||
real(psb_dpk_), allocatable, intent(inout) :: valaggr(:)
|
||||
type(psb_ldspmat_type), intent(out) :: op_prol
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_lpk_) :: icnt,nlp,k,n,ia,isz,nr, naggr,i,j,m,naggrm1, naggrp1, ntaggr
|
||||
type(psb_ld_coo_sparse_mat) :: tmpcoo
|
||||
integer(psb_ipk_) :: debug_level, debug_unit,err_act
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me
|
||||
integer(psb_lpk_) :: nrow, ncol, n_ne
|
||||
character(len=20) :: name, ch_err
|
||||
|
||||
if(psb_get_errstatus() /= 0) return
|
||||
info=psb_success_
|
||||
name = 'amg_d_newmatch_map_to_tprol'
|
||||
call psb_erractionsave(err_act)
|
||||
debug_unit = psb_get_debug_unit()
|
||||
debug_level = psb_get_debug_level()
|
||||
!
|
||||
ctxt=desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
naggr = nlaggr(me+1)
|
||||
ntaggr = sum(nlaggr)
|
||||
naggrm1 = sum(nlaggr(1:me))
|
||||
naggrp1 = sum(nlaggr(1:me+1))
|
||||
ilaggr(1:nrow) = ilaggr(1:nrow) + naggrm1
|
||||
call psb_halo(ilaggr,desc_a,info)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='psb_halo')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
call psb_halo(valaggr,desc_a,info)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='psb_halo')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
call tmpcoo%allocate(ncol,ntaggr,ncol)
|
||||
j = 0
|
||||
do i=1,ncol
|
||||
if (valaggr(i) /= dzero) then
|
||||
j = j + 1
|
||||
tmpcoo%val(j) = valaggr(i)
|
||||
tmpcoo%ia(j) = i
|
||||
tmpcoo%ja(j) = ilaggr(i)
|
||||
end if
|
||||
end do
|
||||
call tmpcoo%set_nzeros(j)
|
||||
call tmpcoo%set_dupl(psb_dupl_add_)
|
||||
call tmpcoo%set_sorted() ! At this point this is in row-major
|
||||
call op_prol%mv_from(tmpcoo)
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
|
||||
return
|
||||
|
||||
end subroutine amg_d_newmatch_map_to_tprol
|
||||
@@ -0,0 +1,218 @@
|
||||
!
|
||||
!
|
||||
! AMG4PSBLAS version 1.0
|
||||
! Algebraic Multigrid Package
|
||||
! based on PSBLAS (Parallel Sparse BLAS version 3.7)
|
||||
!
|
||||
! (C) Copyright 2021
|
||||
!
|
||||
! Salvatore Filippone
|
||||
! Pasqua D'Ambra
|
||||
! Fabio Durastante
|
||||
!
|
||||
! Redistribution and use in source and binary forms, with or without
|
||||
! modification, are permitted provided that the following conditions
|
||||
! are met:
|
||||
! 1. Redistributions of source code must retain the above copyright
|
||||
! notice, this list of conditions and the following disclaimer.
|
||||
! 2. Redistributions in binary form must reproduce the above copyright
|
||||
! notice, this list of conditions, and the following disclaimer in the
|
||||
! documentation and/or other materials provided with the distribution.
|
||||
! 3. The name of the AMG4PSBLAS group or the names of its contributors may
|
||||
! not be used to endorse or promote products derived from this
|
||||
! software without specific written permission.
|
||||
!
|
||||
! THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
! ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
|
||||
! TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
! PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AMG4PSBLAS GROUP OR ITS CONTRIBUTORS
|
||||
! BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
! CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
! SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
! INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
! CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
!
|
||||
! File: amg_daggrmat_nosmth_bld.F90
|
||||
!
|
||||
! Subroutine: amg_daggrmat_nosmth_bld
|
||||
! Version: real
|
||||
!
|
||||
! This routine builds a coarse-level matrix A_C from a fine-level matrix A
|
||||
! by using the Galerkin approach, i.e.
|
||||
!
|
||||
! A_C = P_C^T A P_C,
|
||||
!
|
||||
! where P_C is the piecewise constant interpolation operator corresponding
|
||||
! the fine-to-coarse level mapping built by amg_aggrmap_bld.
|
||||
!
|
||||
! The coarse-level matrix A_C is distributed among the parallel processes or
|
||||
! replicated on each of them, according to the value of p%parms%coarse_mat
|
||||
! specified by the user through amg_dprecinit and amg_zprecset.
|
||||
! On output from this routine the entries of AC, op_prol, op_restr
|
||||
! are still in "global numbering" mode; this is fixed in the calling routine
|
||||
!
|
||||
! For details see
|
||||
! P. D'Ambra, D. di Serafino and S. Filippone, On the development of
|
||||
! PSBLAS-based parallel two-level Schwarz preconditioners, Appl. Num. Math.,
|
||||
! 57 (2007), 1181-1196.
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
! desc_a - type(psb_desc_type), input.
|
||||
! The communication descriptor of the fine-level matrix.
|
||||
! p - type(amg_d_onelev_type), input/output.
|
||||
! The 'one-level' data structure that will contain the local
|
||||
! part of the matrix to be built as well as the information
|
||||
! concerning the prolongator and its transpose.
|
||||
! parms - type(amg_dml_parms), input
|
||||
! Parameters controlling the choice of algorithm
|
||||
! ac - type(psb_dspmat_type), output
|
||||
! The coarse matrix on output
|
||||
!
|
||||
! ilaggr - integer, dimension(:), input
|
||||
! The mapping between the row indices of the coarse-level
|
||||
! matrix and the row indices of the fine-level matrix.
|
||||
! ilaggr(i)=j means that node i in the adjacency graph
|
||||
! of the fine-level matrix is mapped onto node j in the
|
||||
! adjacency graph of the coarse-level matrix. Note that the indices
|
||||
! are assumed to be shifted so as to make sure the ranges on
|
||||
! the various processes do not overlap.
|
||||
! nlaggr - integer, dimension(:) input
|
||||
! nlaggr(i) contains the aggregates held by process i.
|
||||
! op_prol - type(psb_dspmat_type), input/output
|
||||
! The tentative prolongator on input, the computed prolongator on output
|
||||
!
|
||||
! op_restr - type(psb_dspmat_type), output
|
||||
! The restrictor operator; normally, it is the transpose of the prolongator.
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
!
|
||||
subroutine amg_d_newmatch_spmm_bld_inner(a_csr,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_d_inner_mod
|
||||
#if defined(SERIAL_MPI)
|
||||
use amg_d_newmatch_aggregator_mod
|
||||
#else
|
||||
use amg_d_newmatch_aggregator_mod, amg_protect_name => amg_d_newmatch_spmm_bld_inner
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
type(psb_d_csr_sparse_mat), intent(inout) :: a_csr
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_ldspmat_type), intent(inout) :: t_prol
|
||||
type(psb_dspmat_type), intent(inout) :: ac, op_prol, op_restr
|
||||
type(psb_desc_type), intent(out) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: err_act
|
||||
type(psb_ctxt_type) :: ictxt
|
||||
integer(psb_ipk_) :: np, me, ndx
|
||||
character(len=40) :: name
|
||||
type(psb_ld_coo_sparse_mat) :: tmpcoo
|
||||
type(psb_d_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
type(psb_d_csr_sparse_mat) :: ac_csr, csr_restr
|
||||
type(psb_desc_type), target :: tmp_desc
|
||||
type(psb_ldspmat_type) :: lac
|
||||
integer(psb_ipk_) :: debug_level, debug_unit, naggr
|
||||
integer(psb_lpk_) :: nrow, nglob, ncol, ntaggr, nrl, nzl, ip, &
|
||||
& nzt, naggrm1, naggrp1, i, k
|
||||
integer(psb_lpk_), allocatable :: ia(:),ja(:)
|
||||
!integer(psb_lpk_) :: nrsave, ncsave, nzsave, nza, nrpsave, ncpsave, nzpsave
|
||||
logical, parameter :: do_timings=.true., oldstyle=.false., debug=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_prolcnv=-1, idx_proltrans=-1, idx_asb=-1
|
||||
|
||||
name='amg_newmatch_spmm_bld_inner'
|
||||
if(psb_get_errstatus().ne.0) return
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
|
||||
|
||||
ictxt = desc_a%get_context()
|
||||
call psb_info(ictxt, me, np)
|
||||
debug_unit = psb_get_debug_unit()
|
||||
debug_level = psb_get_debug_level()
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("SPMM_BLD: spspmm ")
|
||||
if ((do_timings).and.(idx_prolcnv==-1)) &
|
||||
& idx_prolcnv = psb_get_timer_idx("SPMM_BLD: prolcnv ")
|
||||
if ((do_timings).and.(idx_proltrans==-1)) &
|
||||
& idx_proltrans = psb_get_timer_idx("SPMM_BLD: proltrans")
|
||||
if ((do_timings).and.(idx_asb==-1)) &
|
||||
& idx_asb = psb_get_timer_idx("SPMM_BLD: asb ")
|
||||
|
||||
if (do_timings) call psb_tic(idx_prolcnv)
|
||||
naggr = nlaggr(me+1)
|
||||
ntaggr = sum(nlaggr)
|
||||
naggrm1 = sum(nlaggr(1:me))
|
||||
naggrp1 = sum(nlaggr(1:me+1))
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
!
|
||||
! Here T_PROL should be arriving with GLOBAL indices on the cols
|
||||
! and LOCAL indices on the rows.
|
||||
!
|
||||
if (debug) write(0,*) me,' ',trim(name),' Size check on entry New: ',&
|
||||
& op_prol%get_fmt(),op_prol%get_nrows(),op_prol%get_ncols(),op_prol%get_nzeros(),&
|
||||
& nrow,ntaggr,naggr
|
||||
|
||||
call t_prol%cp_to(tmpcoo)
|
||||
|
||||
call psb_cdall(ictxt,desc_ac,info,nl=naggr)
|
||||
nzl = tmpcoo%get_nzeros()
|
||||
if (debug) write(0,*) me,' ',trim(name),' coo_prol: ',&
|
||||
& tmpcoo%ia(1:min(10,nzl)),' :',tmpcoo%ja(1:min(10,nzl))
|
||||
call desc_ac%indxmap%g2lip_ins(tmpcoo%ja(1:nzl),info)
|
||||
call tmpcoo%set_ncols(desc_ac%get_local_cols())
|
||||
call tmpcoo%cp_to_icoo(coo_prol,info)
|
||||
|
||||
call amg_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
& coo_prol,desc_ac,coo_restr,info)
|
||||
|
||||
nzl = coo_prol%get_nzeros()
|
||||
if (debug) write(0,*) me,' ',trim(name),' coo_prol: ',&
|
||||
& coo_prol%ia(1:min(10,nzl)),' :',coo_prol%ja(1:min(10,nzl))
|
||||
|
||||
call op_prol%mv_from(coo_prol)
|
||||
call op_restr%mv_from(coo_restr)
|
||||
|
||||
if (debug) then
|
||||
write(0,*) me,' ',trim(name),' Checkpoint at exit'
|
||||
call psb_barrier(ictxt)
|
||||
write(0,*) me,' ',trim(name),' Checkpoint through'
|
||||
end if
|
||||
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_internal_error_,name,a_err='Build ac = op_restr x a3')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
#endif
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
|
||||
return
|
||||
|
||||
end subroutine amg_d_newmatch_spmm_bld_inner
|
||||
@@ -0,0 +1,169 @@
|
||||
!
|
||||
!
|
||||
! AMG4PSBLAS version 1.0
|
||||
! Algebraic Multigrid Package
|
||||
! based on PSBLAS (Parallel Sparse BLAS version 3.7)
|
||||
!
|
||||
! (C) Copyright 2021
|
||||
!
|
||||
! Salvatore Filippone
|
||||
! Pasqua D'Ambra
|
||||
! Fabio Durastante
|
||||
!
|
||||
! Redistribution and use in source and binary forms, with or without
|
||||
! modification, are permitted provided that the following conditions
|
||||
! are met:
|
||||
! 1. Redistributions of source code must retain the above copyright
|
||||
! notice, this list of conditions and the following disclaimer.
|
||||
! 2. Redistributions in binary form must reproduce the above copyright
|
||||
! notice, this list of conditions, and the following disclaimer in the
|
||||
! documentation and/or other materials provided with the distribution.
|
||||
! 3. The name of the AMG4PSBLAS group or the names of its contributors may
|
||||
! not be used to endorse or promote products derived from this
|
||||
! software without specific written permission.
|
||||
!
|
||||
! THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
! ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
|
||||
! TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
! PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AMG4PSBLAS GROUP OR ITS CONTRIBUTORS
|
||||
! BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
! CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
! SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
! INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
! CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
!
|
||||
! File: amg_daggrmat_nosmth_bld_ov.F90
|
||||
!
|
||||
! Subroutine: amg_daggrmat_nosmth_bld_ov
|
||||
! Version: real
|
||||
!
|
||||
! This routine builds a coarse-level matrix A_C from a fine-level matrix A
|
||||
! by using the Galerkin approach, i.e.
|
||||
!
|
||||
! A_C = P_C^T A P_C,
|
||||
!
|
||||
! where P_C is the piecewise constant interpolation operator corresponding
|
||||
! the fine-to-coarse level mapping built by amg_aggrmap_bld_ov.
|
||||
!
|
||||
! The coarse-level matrix A_C is distributed among the parallel processes or
|
||||
! replicated on each of them, according to the value of p%parms%coarse_mat
|
||||
! specified by the user through amg_dprecinit and amg_zprecset.
|
||||
! On output from this routine the entries of AC, op_prol, op_restr
|
||||
! are still in "global numbering" mode; this is fixed in the calling routine
|
||||
!
|
||||
! For details see
|
||||
! P. D'Ambra, D. di Serafino and S. Filippone, On the development of
|
||||
! PSBLAS-based parallel two-level Schwarz preconditioners, Appl. Num. Math.,
|
||||
! 57 (2007), 1181-1196.
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
! desc_a - type(psb_desc_type), input.
|
||||
! The communication descriptor of the fine-level matrix.
|
||||
! p - type(amg_d_onelev_type), input/output.
|
||||
! The 'one-level' data structure that will contain the local
|
||||
! part of the matrix to be built as well as the information
|
||||
! concerning the prolongator and its transpose.
|
||||
! parms - type(amg_dml_parms), input
|
||||
! Parameters controlling the choice of algorithm
|
||||
! ac - type(psb_dspmat_type), output
|
||||
! The coarse matrix on output
|
||||
!
|
||||
! ilaggr - integer, dimension(:), input
|
||||
! The mapping between the row indices of the coarse-level
|
||||
! matrix and the row indices of the fine-level matrix.
|
||||
! ilaggr(i)=j means that node i in the adjacency graph
|
||||
! of the fine-level matrix is mapped onto node j in the
|
||||
! adjacency graph of the coarse-level matrix. Note that the indices
|
||||
! are assumed to be shifted so as to make sure the ranges on
|
||||
! the various processes do not overlap.
|
||||
! nlaggr - integer, dimension(:) input
|
||||
! nlaggr(i) contains the aggregates held by process i.
|
||||
! op_prol - type(psb_dspmat_type), input/output
|
||||
! The tentative prolongator on input, the computed prolongator on output
|
||||
!
|
||||
! op_restr - type(psb_dspmat_type), output
|
||||
! The restrictor operator; normally, it is the transpose of the prolongator.
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
!
|
||||
subroutine amg_d_newmatch_spmm_bld_ov(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_d_inner_mod
|
||||
#if defined(SERIAL_MPI)
|
||||
use amg_d_newmatch_aggregator_mod
|
||||
#else
|
||||
use amg_d_newmatch_aggregator_mod, amg_protect_name => amg_d_newmatch_spmm_bld_ov
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
type(psb_dspmat_type), intent(inout) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_ldspmat_type), intent(inout) :: t_prol
|
||||
type(psb_dspmat_type), intent(inout) :: ac, op_prol, op_restr
|
||||
type(psb_desc_type), intent(out) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: err_act
|
||||
|
||||
type(psb_ctxt_type) :: ictxt
|
||||
integer(psb_ipk_) :: np, me
|
||||
character(len=20) :: name
|
||||
type(psb_d_csr_sparse_mat) :: acsr
|
||||
type(psb_ld_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
integer(psb_lpk_) :: nrow, nglob, ncol, ntaggr, nzl, ip, &
|
||||
& naggr, nzt, naggrm1, naggrp1, i, k
|
||||
integer(psb_ipk_) :: inaggr, nzlp
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
logical, parameter :: debug=.false., new_version=.true.
|
||||
|
||||
name='amg_newmatch_spmm_bld_ov'
|
||||
if(psb_get_errstatus().ne.0) return
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
|
||||
|
||||
ictxt = desc_a%get_context()
|
||||
call psb_info(ictxt, me, np)
|
||||
debug_unit = psb_get_debug_unit()
|
||||
debug_level = psb_get_debug_level()
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
call a%mv_to(acsr)
|
||||
|
||||
call amg_d_newmatch_spmm_bld_inner(acsr,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
if (psb_errstatus_fatal()) write(0,*)me,trim(name),'Error fatal on exit from bld_inner',info
|
||||
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
call psb_errpush(info,name,a_err="SPMM_BLD_INNER")
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done spmm_bld '
|
||||
#endif
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
|
||||
return
|
||||
|
||||
end subroutine amg_d_newmatch_spmm_bld_ov
|
||||
@@ -140,6 +140,9 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
real(psb_dpk_) :: anorm, omega, tmp, dg, theta
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
name='amg_aggrmat_smth_bld'
|
||||
info=psb_success_
|
||||
@@ -153,6 +156,23 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
ctxt = desc_a%get_context()
|
||||
|
||||
call psb_info(ctxt, me, np)
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("DEC_SMTH_BLD: par_spspmm")
|
||||
if ((do_timings).and.(idx_phase1==-1)) &
|
||||
& idx_phase1 = psb_get_timer_idx("DEC_SMTH_BLD: phase1 ")
|
||||
if ((do_timings).and.(idx_phase2==-1)) &
|
||||
& idx_phase2 = psb_get_timer_idx("DEC_SMTH_BLD: phase2 ")
|
||||
if ((do_timings).and.(idx_phase3==-1)) &
|
||||
& idx_phase3 = psb_get_timer_idx("DEC_SMTH_BLD: phase3 ")
|
||||
if ((do_timings).and.(idx_gtrans==-1)) &
|
||||
& idx_gtrans = psb_get_timer_idx("DEC_SMTH_BLD: gtrans ")
|
||||
if ((do_timings).and.(idx_refine==-1)) &
|
||||
& idx_refine = psb_get_timer_idx("DEC_SMTH_BLD: refine ")
|
||||
if ((do_timings).and.(idx_cdasb==-1)) &
|
||||
& idx_cdasb = psb_get_timer_idx("DEC_SMTH_BLD: cdasb ")
|
||||
if ((do_timings).and.(idx_ptap==-1)) &
|
||||
& idx_ptap = psb_get_timer_idx("DEC_SMTH_BLD: ptap_bld ")
|
||||
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
@@ -171,6 +191,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
! naggr: number of local aggregates
|
||||
! nrow: local rows.
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_phase1)
|
||||
|
||||
! Get the diagonal D
|
||||
adiag = a%get_diag(info)
|
||||
@@ -196,7 +217,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!
|
||||
! Build the filtered matrix Af from A
|
||||
!
|
||||
|
||||
!$OMP parallel do private(i,j,tmp,jd) schedule(static)
|
||||
do i=1, nrow
|
||||
tmp = dzero
|
||||
jd = -1
|
||||
@@ -214,11 +235,13 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
enddo
|
||||
!$OMP end parallel do
|
||||
! Take out zeroed terms
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= dzero) then
|
||||
adiag(i) = done / adiag(i)
|
||||
@@ -226,7 +249,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
adiag(i) = done
|
||||
end if
|
||||
end do
|
||||
|
||||
!$OMP end parallel do
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
@@ -252,8 +275,9 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_omega_alg_')
|
||||
goto 9999
|
||||
end if
|
||||
if (do_timings) call psb_toc(idx_phase1)
|
||||
|
||||
|
||||
if (do_timings) call psb_tic(idx_phase2)
|
||||
call acsrf%scal(adiag,info)
|
||||
if (info /= psb_success_) goto 9999
|
||||
|
||||
@@ -267,6 +291,8 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
call psb_cdasb(desc_ac,info)
|
||||
call psb_cd_reinit(desc_ac,info)
|
||||
if (do_timings) call psb_toc(idx_phase2)
|
||||
if (do_timings) call psb_tic(idx_phase3)
|
||||
!
|
||||
! Build the smoothed prolongator using either A or Af
|
||||
! acsr1 = (I-w*D*A) Prol acsr1 = (I-w*D*Af) Prol
|
||||
@@ -279,8 +305,8 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spspmm 1')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (do_timings) call psb_toc(idx_phase3)
|
||||
if (do_timings) call psb_tic(idx_ptap)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done SPSPMM 1'
|
||||
@@ -292,7 +318,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
call op_prol%mv_from(coo_prol)
|
||||
call op_restr%mv_from(coo_restr)
|
||||
|
||||
if (do_timings) call psb_toc(idx_ptap)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
|
||||
@@ -97,6 +97,8 @@ subroutine amg_s_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
integer(psb_lpk_) :: ntaggr
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
logical :: clean_zeros
|
||||
integer(psb_ipk_), save :: idx_map_bld=-1, idx_map_tprol=-1
|
||||
logical, parameter :: do_timings=.false.
|
||||
|
||||
name='amg_s_dec_aggregator_tprol'
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -108,6 +110,10 @@ subroutine amg_s_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
if ((do_timings).and.(idx_map_bld==-1)) &
|
||||
& idx_map_bld = psb_get_timer_idx("DEC_TPROL: map_bld")
|
||||
if ((do_timings).and.(idx_map_tprol==-1)) &
|
||||
& idx_map_tprol = psb_get_timer_idx("DEC_TPROL: map_tprol")
|
||||
|
||||
call amg_check_def(parms%ml_cycle,'Multilevel cycle',&
|
||||
& amg_mult_ml_,is_legal_ml_cycle)
|
||||
@@ -121,10 +127,14 @@ subroutine amg_s_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
! The decoupled aggregator based on SOC measures ignores
|
||||
! ag_data except for clean_zeros; soc_map_bld is a procedure pointer.
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_map_bld)
|
||||
clean_zeros = ag%do_clean_zeros
|
||||
call ag%soc_map_bld(parms%aggr_ord,parms%aggr_thresh,clean_zeros,a,desc_a,nlaggr,ilaggr,info)
|
||||
if (do_timings) call psb_toc(idx_map_bld)
|
||||
if (do_timings) call psb_tic(idx_map_tprol)
|
||||
|
||||
if (info==psb_success_) call amg_map_to_tprol(desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
if (do_timings) call psb_toc(idx_map_tprol)
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
call psb_errpush(info,name,a_err='soc_map_bld/map_to_tprol')
|
||||
|
||||
@@ -140,6 +140,9 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
real(psb_spk_) :: anorm, omega, tmp, dg, theta
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
name='amg_aggrmat_smth_bld'
|
||||
info=psb_success_
|
||||
@@ -153,6 +156,23 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
ctxt = desc_a%get_context()
|
||||
|
||||
call psb_info(ctxt, me, np)
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("DEC_SMTH_BLD: par_spspmm")
|
||||
if ((do_timings).and.(idx_phase1==-1)) &
|
||||
& idx_phase1 = psb_get_timer_idx("DEC_SMTH_BLD: phase1 ")
|
||||
if ((do_timings).and.(idx_phase2==-1)) &
|
||||
& idx_phase2 = psb_get_timer_idx("DEC_SMTH_BLD: phase2 ")
|
||||
if ((do_timings).and.(idx_phase3==-1)) &
|
||||
& idx_phase3 = psb_get_timer_idx("DEC_SMTH_BLD: phase3 ")
|
||||
if ((do_timings).and.(idx_gtrans==-1)) &
|
||||
& idx_gtrans = psb_get_timer_idx("DEC_SMTH_BLD: gtrans ")
|
||||
if ((do_timings).and.(idx_refine==-1)) &
|
||||
& idx_refine = psb_get_timer_idx("DEC_SMTH_BLD: refine ")
|
||||
if ((do_timings).and.(idx_cdasb==-1)) &
|
||||
& idx_cdasb = psb_get_timer_idx("DEC_SMTH_BLD: cdasb ")
|
||||
if ((do_timings).and.(idx_ptap==-1)) &
|
||||
& idx_ptap = psb_get_timer_idx("DEC_SMTH_BLD: ptap_bld ")
|
||||
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
@@ -171,6 +191,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
! naggr: number of local aggregates
|
||||
! nrow: local rows.
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_phase1)
|
||||
|
||||
! Get the diagonal D
|
||||
adiag = a%get_diag(info)
|
||||
@@ -196,7 +217,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!
|
||||
! Build the filtered matrix Af from A
|
||||
!
|
||||
|
||||
!$OMP parallel do private(i,j,tmp,jd) schedule(static)
|
||||
do i=1, nrow
|
||||
tmp = szero
|
||||
jd = -1
|
||||
@@ -214,11 +235,13 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
enddo
|
||||
!$OMP end parallel do
|
||||
! Take out zeroed terms
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= szero) then
|
||||
adiag(i) = sone / adiag(i)
|
||||
@@ -226,7 +249,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
adiag(i) = sone
|
||||
end if
|
||||
end do
|
||||
|
||||
!$OMP end parallel do
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
@@ -252,8 +275,9 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_omega_alg_')
|
||||
goto 9999
|
||||
end if
|
||||
if (do_timings) call psb_toc(idx_phase1)
|
||||
|
||||
|
||||
if (do_timings) call psb_tic(idx_phase2)
|
||||
call acsrf%scal(adiag,info)
|
||||
if (info /= psb_success_) goto 9999
|
||||
|
||||
@@ -267,6 +291,8 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
call psb_cdasb(desc_ac,info)
|
||||
call psb_cd_reinit(desc_ac,info)
|
||||
if (do_timings) call psb_toc(idx_phase2)
|
||||
if (do_timings) call psb_tic(idx_phase3)
|
||||
!
|
||||
! Build the smoothed prolongator using either A or Af
|
||||
! acsr1 = (I-w*D*A) Prol acsr1 = (I-w*D*Af) Prol
|
||||
@@ -279,8 +305,8 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spspmm 1')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (do_timings) call psb_toc(idx_phase3)
|
||||
if (do_timings) call psb_tic(idx_ptap)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done SPSPMM 1'
|
||||
@@ -292,7 +318,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
call op_prol%mv_from(coo_prol)
|
||||
call op_restr%mv_from(coo_restr)
|
||||
|
||||
if (do_timings) call psb_toc(idx_ptap)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
|
||||
@@ -97,6 +97,8 @@ subroutine amg_z_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
integer(psb_lpk_) :: ntaggr
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
logical :: clean_zeros
|
||||
integer(psb_ipk_), save :: idx_map_bld=-1, idx_map_tprol=-1
|
||||
logical, parameter :: do_timings=.false.
|
||||
|
||||
name='amg_z_dec_aggregator_tprol'
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -108,6 +110,10 @@ subroutine amg_z_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
if ((do_timings).and.(idx_map_bld==-1)) &
|
||||
& idx_map_bld = psb_get_timer_idx("DEC_TPROL: map_bld")
|
||||
if ((do_timings).and.(idx_map_tprol==-1)) &
|
||||
& idx_map_tprol = psb_get_timer_idx("DEC_TPROL: map_tprol")
|
||||
|
||||
call amg_check_def(parms%ml_cycle,'Multilevel cycle',&
|
||||
& amg_mult_ml_,is_legal_ml_cycle)
|
||||
@@ -121,10 +127,14 @@ subroutine amg_z_dec_aggregator_build_tprol(ag,parms,ag_data,&
|
||||
! The decoupled aggregator based on SOC measures ignores
|
||||
! ag_data except for clean_zeros; soc_map_bld is a procedure pointer.
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_map_bld)
|
||||
clean_zeros = ag%do_clean_zeros
|
||||
call ag%soc_map_bld(parms%aggr_ord,parms%aggr_thresh,clean_zeros,a,desc_a,nlaggr,ilaggr,info)
|
||||
if (do_timings) call psb_toc(idx_map_bld)
|
||||
if (do_timings) call psb_tic(idx_map_tprol)
|
||||
|
||||
if (info==psb_success_) call amg_map_to_tprol(desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
if (do_timings) call psb_toc(idx_map_tprol)
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
call psb_errpush(info,name,a_err='soc_map_bld/map_to_tprol')
|
||||
|
||||
@@ -140,6 +140,9 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
real(psb_dpk_) :: anorm, omega, tmp, dg, theta
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
name='amg_aggrmat_smth_bld'
|
||||
info=psb_success_
|
||||
@@ -153,6 +156,23 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
ctxt = desc_a%get_context()
|
||||
|
||||
call psb_info(ctxt, me, np)
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("DEC_SMTH_BLD: par_spspmm")
|
||||
if ((do_timings).and.(idx_phase1==-1)) &
|
||||
& idx_phase1 = psb_get_timer_idx("DEC_SMTH_BLD: phase1 ")
|
||||
if ((do_timings).and.(idx_phase2==-1)) &
|
||||
& idx_phase2 = psb_get_timer_idx("DEC_SMTH_BLD: phase2 ")
|
||||
if ((do_timings).and.(idx_phase3==-1)) &
|
||||
& idx_phase3 = psb_get_timer_idx("DEC_SMTH_BLD: phase3 ")
|
||||
if ((do_timings).and.(idx_gtrans==-1)) &
|
||||
& idx_gtrans = psb_get_timer_idx("DEC_SMTH_BLD: gtrans ")
|
||||
if ((do_timings).and.(idx_refine==-1)) &
|
||||
& idx_refine = psb_get_timer_idx("DEC_SMTH_BLD: refine ")
|
||||
if ((do_timings).and.(idx_cdasb==-1)) &
|
||||
& idx_cdasb = psb_get_timer_idx("DEC_SMTH_BLD: cdasb ")
|
||||
if ((do_timings).and.(idx_ptap==-1)) &
|
||||
& idx_ptap = psb_get_timer_idx("DEC_SMTH_BLD: ptap_bld ")
|
||||
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
@@ -171,6 +191,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
! naggr: number of local aggregates
|
||||
! nrow: local rows.
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_phase1)
|
||||
|
||||
! Get the diagonal D
|
||||
adiag = a%get_diag(info)
|
||||
@@ -196,7 +217,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!
|
||||
! Build the filtered matrix Af from A
|
||||
!
|
||||
|
||||
!$OMP parallel do private(i,j,tmp,jd) schedule(static)
|
||||
do i=1, nrow
|
||||
tmp = zzero
|
||||
jd = -1
|
||||
@@ -214,11 +235,13 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
enddo
|
||||
!$OMP end parallel do
|
||||
! Take out zeroed terms
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= zzero) then
|
||||
adiag(i) = zone / adiag(i)
|
||||
@@ -226,7 +249,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
adiag(i) = zone
|
||||
end if
|
||||
end do
|
||||
|
||||
!$OMP end parallel do
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
@@ -252,8 +275,9 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_omega_alg_')
|
||||
goto 9999
|
||||
end if
|
||||
if (do_timings) call psb_toc(idx_phase1)
|
||||
|
||||
|
||||
if (do_timings) call psb_tic(idx_phase2)
|
||||
call acsrf%scal(adiag,info)
|
||||
if (info /= psb_success_) goto 9999
|
||||
|
||||
@@ -267,6 +291,8 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
call psb_cdasb(desc_ac,info)
|
||||
call psb_cd_reinit(desc_ac,info)
|
||||
if (do_timings) call psb_toc(idx_phase2)
|
||||
if (do_timings) call psb_tic(idx_phase3)
|
||||
!
|
||||
! Build the smoothed prolongator using either A or Af
|
||||
! acsr1 = (I-w*D*A) Prol acsr1 = (I-w*D*Af) Prol
|
||||
@@ -279,8 +305,8 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spspmm 1')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (do_timings) call psb_toc(idx_phase3)
|
||||
if (do_timings) call psb_tic(idx_ptap)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done SPSPMM 1'
|
||||
@@ -292,7 +318,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
call op_prol%mv_from(coo_prol)
|
||||
call op_restr%mv_from(coo_restr)
|
||||
|
||||
if (do_timings) call psb_toc(idx_ptap)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
// TODO comment
|
||||
|
||||
void clean(MilanLongInt NLVer,
|
||||
MilanInt myRank,
|
||||
MilanLongInt MessageIndex,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus,
|
||||
MilanInt BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
MilanLongInt msgActual,
|
||||
MilanLongInt *msgActualSent,
|
||||
MilanLongInt msgInd,
|
||||
MilanLongInt *msgIndSent,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanReal *msgPercent)
|
||||
{
|
||||
// Cleanup Phase
|
||||
|
||||
#pragma omp parallel
|
||||
{
|
||||
#pragma omp master
|
||||
{
|
||||
#pragma omp task
|
||||
{
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ") Waitall= " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << "\n(" << myRank << ") Waitall " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
//return;
|
||||
|
||||
MPI_Waitall(MessageIndex, &SRequest[0], &SStatus[0]);
|
||||
|
||||
// MPI_Buffer_attach(&Buffer, BufferSize); //Attach the Buffer
|
||||
if (BufferSize > 0)
|
||||
{
|
||||
MPI_Buffer_detach(&Buffer, &BufferSize); // Detach the Buffer
|
||||
free(Buffer); // Free the memory that was allocated
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")End of function to compute matching: " << endl;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")myCardinality: " << myCard << endl;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")Matching took " << finishTime - startTime << "seconds" << endl;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")** Getting out of the matching function **" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ") Number of Ghost edges = " << numGhostEdges;
|
||||
cout << "\n(" << myRank << ") Total number of potential message X 2 = " << numGhostEdges * 2;
|
||||
cout << "\n(" << myRank << ") Number messages bundled = " << NumMessagesBundled;
|
||||
cout << "\n(" << myRank << ") Total Individual Messages sent = " << msgInd;
|
||||
if (msgInd > 0)
|
||||
{
|
||||
cout << "\n(" << myRank << ") Percentage of messages bundled = " << ((double)NumMessagesBundled / (double)(msgInd)) * 100.0 << "% \n";
|
||||
}
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#pragma omp task
|
||||
{
|
||||
*msgActualSent = msgActual;
|
||||
*msgIndSent = msgInd;
|
||||
if (msgInd > 0)
|
||||
{
|
||||
*msgPercent = ((double)NumMessagesBundled / (double)(msgInd)) * 100.0;
|
||||
}
|
||||
else
|
||||
{
|
||||
*msgPercent = 0;
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0)
|
||||
cout << "\n(" << myRank << ") Done" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
/**
|
||||
* Execute the research fr the Candidate Mate without controlling if the vertices are already matched.
|
||||
* Returns the vertices with the highest weight
|
||||
* @param adj1
|
||||
* @param adj2
|
||||
* @param verLocInd
|
||||
* @param edgeLocWeight
|
||||
* @return
|
||||
*/
|
||||
MilanLongInt firstComputeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight)
|
||||
{
|
||||
MilanInt w = -1;
|
||||
MilanReal heaviestEdgeWt = MilanRealMin; // Assign the smallest Value possible first LDBL_MIN
|
||||
int finalK;
|
||||
for (int k = adj1; k < adj2; k++) {
|
||||
if ((edgeLocWeight[k] > heaviestEdgeWt) ||
|
||||
((edgeLocWeight[k] == heaviestEdgeWt) && (w < verLocInd[k]))) {
|
||||
heaviestEdgeWt = edgeLocWeight[k];
|
||||
w = verLocInd[k];
|
||||
finalK = k;
|
||||
}
|
||||
} // End of for loop
|
||||
return finalK;
|
||||
}
|
||||
|
||||
/**
|
||||
* //TODO documentation
|
||||
* @param adj1
|
||||
* @param adj2
|
||||
* @param edgeLocWeight
|
||||
* @param k
|
||||
* @param verLocInd
|
||||
* @param StartIndex
|
||||
* @param EndIndex
|
||||
* @param GMate
|
||||
* @param Mate
|
||||
* @param Ghost2LocalMap
|
||||
* @return
|
||||
*/
|
||||
MilanLongInt computeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap)
|
||||
{
|
||||
// Start: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
|
||||
MilanInt w = -1;
|
||||
MilanReal heaviestEdgeWt = MilanRealMin; // Assign the smallest Value possible first LDBL_MIN
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap))
|
||||
continue;
|
||||
|
||||
if ((edgeLocWeight[k] > heaviestEdgeWt) ||
|
||||
((edgeLocWeight[k] == heaviestEdgeWt) && (w < verLocInd[k]))) {
|
||||
heaviestEdgeWt = edgeLocWeight[k];
|
||||
w = verLocInd[k];
|
||||
}
|
||||
} // End of for loop
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
|
||||
return w;
|
||||
}
|
||||
@@ -80,9 +80,11 @@ class staticQueue
|
||||
MilanLongInt squeueTail;
|
||||
MilanLongInt NumNodes;
|
||||
|
||||
//FIXME I had to comment this piece of code in order to make everything work.
|
||||
// why?
|
||||
//Prevent Assignment and Pass by Value:
|
||||
staticQueue(const staticQueue& src);
|
||||
staticQueue& operator=(const staticQueue& rhs);
|
||||
//staticQueue(const staticQueue& src);
|
||||
//staticQueue& operator=(const staticQueue& rhs);
|
||||
|
||||
public:
|
||||
//Constructors and Destructors
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void extractUChunk(
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU)
|
||||
{
|
||||
|
||||
UChunkBeingProcessed.clear();
|
||||
#pragma omp critical(U)
|
||||
{
|
||||
|
||||
if (U.empty() && !privateU.empty()) // If U is empty but there are nodes in private U
|
||||
{
|
||||
while (!privateU.empty())
|
||||
UChunkBeingProcessed.push_back(privateU.back());
|
||||
privateU.pop_back();
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int i = 0; i < UCHUNK; i++)
|
||||
{ // Pop the new nodes
|
||||
if (U.empty())
|
||||
break;
|
||||
UChunkBeingProcessed.push_back(U.back());
|
||||
U.pop_back();
|
||||
}
|
||||
}
|
||||
|
||||
} // End of critical U // End of critical U
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
/// Find the owner of a ghost node:
|
||||
MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs)
|
||||
{
|
||||
|
||||
MilanLongInt mStartInd = mVerDistance[myRank];
|
||||
MilanInt Start = 0;
|
||||
MilanInt End = numProcs;
|
||||
MilanInt Current = 0;
|
||||
|
||||
while (Start <= End)
|
||||
{
|
||||
Current = (End + Start) / 2;
|
||||
// CASE-1:
|
||||
if (mVerDistance[Current] == vtxIndex) return Current;
|
||||
else // CASE 2:
|
||||
if (mVerDistance[Current] > vtxIndex)
|
||||
End = Current - 1;
|
||||
else // CASE 3:
|
||||
Start = Current + 1;
|
||||
} // End of While()
|
||||
|
||||
if (mVerDistance[Current] > vtxIndex)
|
||||
return (Current - 1);
|
||||
|
||||
return Current;
|
||||
} // End of findOwnerOfGhost()
|
||||
@@ -0,0 +1,304 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt StartIndex, MilanLongInt EndIndex,
|
||||
MilanLongInt *numGhostEdges,
|
||||
MilanLongInt *numGhostVertices,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &Counter,
|
||||
vector<MilanLongInt> &verGhostPtr,
|
||||
vector<MilanLongInt> &verGhostInd,
|
||||
vector<MilanLongInt> &tempCounter,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Message,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
MilanLongInt *&candidateMate,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
MilanLongInt insertMe = 0;
|
||||
MilanLongInt adj1, adj2;
|
||||
int i, v, k, w;
|
||||
// index that starts with zero to |Vg| - 1
|
||||
map<MilanLongInt, MilanLongInt>::iterator storedAlready;
|
||||
|
||||
#pragma omp parallel private(insertMe, k, w, v, adj1, adj2) firstprivate(StartIndex, EndIndex) default(shared) num_threads(NUM_THREAD)
|
||||
{
|
||||
|
||||
#pragma omp single
|
||||
{
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
double Ghost2LocalInitialization = MPI_Wtime();
|
||||
#endif
|
||||
|
||||
/*
|
||||
* OMP Ghost2LocalInitialization
|
||||
* This loop analyzes all the edges and when finds a ghost edge
|
||||
* puts it in the Ghost2LocalMap.
|
||||
* A critical region is needed when inserting data in the map.
|
||||
*
|
||||
* Despite the critical region it is still productive to
|
||||
* parallelize this cycle because the critical region is exeuted
|
||||
* only when a ghost edge is found and ghost edges are a minority,
|
||||
* circa 3.5% during the tests.
|
||||
*/
|
||||
#pragma omp task depend(out \
|
||||
: *numGhostEdges, Counter, Ghost2LocalMap, insertMe, storedAlready, *numGhostVertices)
|
||||
{
|
||||
#pragma omp taskloop num_tasks(NUM_THREAD) reduction(+ \
|
||||
: numGhostEdges[:1])
|
||||
for (i = 0; i < NLEdge; i++)
|
||||
{ // O(m) - Each edge stored twice
|
||||
insertMe = verLocInd[i];
|
||||
if ((insertMe < StartIndex) || (insertMe > EndIndex))
|
||||
{ // Find a ghost
|
||||
(*numGhostEdges)++;
|
||||
#pragma omp critical
|
||||
{
|
||||
storedAlready = Ghost2LocalMap.find(insertMe);
|
||||
if (storedAlready != Ghost2LocalMap.end())
|
||||
{ // Has already been added
|
||||
Counter[storedAlready->second]++; // Increment the counter
|
||||
}
|
||||
else
|
||||
{ // Insert an entry for the ghost:
|
||||
Ghost2LocalMap[insertMe] = *numGhostVertices; // Add a map entry
|
||||
Counter.push_back(1); // Initialize the counter
|
||||
(*numGhostVertices)++; // Increment the number of ghost vertices
|
||||
} // End of else()
|
||||
}
|
||||
} // End of if ( (insertMe < StartIndex) || (insertMe > EndIndex) )
|
||||
} // End of for(ghost vertices)
|
||||
} // end of task depend
|
||||
|
||||
// *numGhostEdges = atomicNumGhostEdges;
|
||||
#ifdef TIME_TRACKER
|
||||
Ghost2LocalInitialization = MPI_Wtime() - Ghost2LocalInitialization;
|
||||
fprintf(stderr, "Ghost2LocalInitialization time: %f\n", Ghost2LocalInitialization);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")NGhosts:" << *numGhostVertices << " GhostEdges: " << *numGhostEdges;
|
||||
if (!Ghost2LocalMap.empty())
|
||||
{
|
||||
cout << "\n(" << myRank << ")Final Map : on process ";
|
||||
cout << "\n(" << myRank << ")Key \t Value \t Counter \n";
|
||||
fflush(stdout);
|
||||
storedAlready = Ghost2LocalMap.begin();
|
||||
do
|
||||
{
|
||||
cout << storedAlready->second << " - " << storedAlready->first << " : " << Counter[storedAlready->second] << endl;
|
||||
fflush(stdout);
|
||||
storedAlready++;
|
||||
} while (storedAlready != Ghost2LocalMap.end());
|
||||
}
|
||||
#endif
|
||||
|
||||
#pragma omp task depend(out \
|
||||
: verGhostPtr, tempCounter, verGhostInd, GMate) depend(in \
|
||||
: *numGhostVertices, *numGhostEdges)
|
||||
{
|
||||
|
||||
// Initialize adjacency Lists for Ghost Vertices:
|
||||
try
|
||||
{
|
||||
verGhostPtr.reserve(*numGhostVertices + 1); // Pointer Vector
|
||||
tempCounter.reserve(*numGhostVertices); // Pointer Vector
|
||||
verGhostInd.reserve(*numGhostEdges); // Index Vector
|
||||
GMate.reserve(*numGhostVertices); // Ghost Mate Vector
|
||||
}
|
||||
catch (length_error)
|
||||
{
|
||||
cout << "Error in function algoDistEdgeApproxDominatingEdgesLinearSearch: \n";
|
||||
cout << "Not enough memory to allocate the internal variables \n";
|
||||
exit(1);
|
||||
}
|
||||
// Initialize the Vectors:
|
||||
verGhostPtr.resize(*numGhostVertices + 1, 0); // Pointer Vector
|
||||
tempCounter.resize(*numGhostVertices, 0); // Temporary Counter
|
||||
verGhostInd.resize(*numGhostEdges, -1); // Index Vector
|
||||
GMate.resize(*numGhostVertices, -1); // Temporary Counter
|
||||
verGhostPtr[0] = 0; // The first value
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Ghost Vertex Pointer: ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
} // End of task
|
||||
|
||||
#pragma omp task depend(out \
|
||||
: verGhostPtr) depend(in \
|
||||
: Counter, *numGhostVertices)
|
||||
{
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
double verGhostPtrInitialization = MPI_Wtime();
|
||||
#endif
|
||||
for (i = 0; i < *numGhostVertices; i++)
|
||||
{ // O(|Ghost Vertices|)
|
||||
verGhostPtr[i + 1] = verGhostPtr[i] + Counter[i];
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << verGhostPtr[i] << "\t";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
verGhostPtrInitialization = MPI_Wtime() - verGhostPtrInitialization;
|
||||
fprintf(stderr, "verGhostPtrInitialization time: %f\n", verGhostPtrInitialization);
|
||||
#endif
|
||||
} // End of task
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
if (*numGhostVertices > 0)
|
||||
cout << verGhostPtr[*numGhostVertices] << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
double verGhostIndInitialization = MPI_Wtime();
|
||||
#endif
|
||||
|
||||
/*
|
||||
* OMP verGhostIndInitialization
|
||||
*
|
||||
* In this cycle the verGhostInd is initialized
|
||||
* with the datas related to ghost edges.
|
||||
* The check to see if a node is a ghost node is
|
||||
* executed in paralle and when a ghost node
|
||||
* is found a critical region is started.
|
||||
*
|
||||
* Despite the critical region it's still useful to
|
||||
* parallelize the for cause the ghost nodes
|
||||
* are a minority hence the critical region is executed
|
||||
* few times, circa 3.5% of the times in the tests.
|
||||
*/
|
||||
#pragma omp task depend(in \
|
||||
: insertMe, Ghost2LocalMap, tempCounter, verGhostPtr) depend(out \
|
||||
: verGhostInd)
|
||||
{
|
||||
#pragma omp taskloop num_tasks(NUM_THREAD)
|
||||
for (v = 0; v < NLVer; v++)
|
||||
{
|
||||
adj1 = verLocPtr[v]; // Vertex Pointer
|
||||
adj2 = verLocPtr[v + 1];
|
||||
for (k = adj1; k < adj2; k++)
|
||||
{
|
||||
w = verLocInd[k]; // Get the adjacent vertex
|
||||
if ((w < StartIndex) || (w > EndIndex))
|
||||
{ // Find a ghost
|
||||
#pragma omp critical
|
||||
{
|
||||
insertMe = verGhostPtr[Ghost2LocalMap[w]] + tempCounter[Ghost2LocalMap[w]]; // Where to insert
|
||||
tempCounter[Ghost2LocalMap[w]]++; // Increment the counter
|
||||
}
|
||||
verGhostInd[insertMe] = v + StartIndex; // Add the adjacency
|
||||
} // End of if((w < StartIndex) || (w > EndIndex))
|
||||
} // End of for(k)
|
||||
} // End of for (v)
|
||||
} // end of tasklopp
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
verGhostIndInitialization = MPI_Wtime() - verGhostIndInitialization;
|
||||
fprintf(stderr, "verGhostIndInitialization time: %f\n", verGhostIndInitialization);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Ghost Vertex Index: ";
|
||||
for (v = 0; v < *numGhostEdges; v++)
|
||||
cout << verGhostInd[v] << "\t";
|
||||
cout << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#pragma omp task depend(in \
|
||||
: *numGhostEdges) depend(out \
|
||||
: QLocalVtx, QGhostVtx, QMsgType, QOwner)
|
||||
{
|
||||
try
|
||||
{
|
||||
QLocalVtx.reserve(*numGhostEdges); // Local Vertex
|
||||
QGhostVtx.reserve(*numGhostEdges); // Ghost Vertex
|
||||
QMsgType.reserve(*numGhostEdges); // Message Type (Request/Failure)
|
||||
QOwner.reserve(*numGhostEdges); // Owner of the ghost: COmpute once and use later
|
||||
}
|
||||
catch (length_error)
|
||||
{
|
||||
cout << "Error in function algoDistEdgeApproxDominatingEdgesMessageBundling: \n";
|
||||
cout << "Not enough memory to allocate the internal variables \n";
|
||||
exit(1);
|
||||
}
|
||||
} // end of task
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Allocating CandidateMate.. ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ") Setup Time :" << *ph0_time << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0)
|
||||
cout << "\n(" << myRank << ") Setup Time :" << *ph0_time << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#pragma omp task depend(in \
|
||||
: *numGhostVertices) depend(out \
|
||||
: candidateMate, S, U, privateU, privateQLocalVtx, privateQGhostVtx, privateQMsgType, privateQOwner)
|
||||
{
|
||||
|
||||
// Allocate Data Structures:
|
||||
/*
|
||||
* candidateMate was a vector and has been replaced with an array
|
||||
* there is no point in using the vector (or maybe there is (???))
|
||||
* so I replaced it with an array wich is slightly faster
|
||||
*/
|
||||
candidateMate = new MilanLongInt[NLVer + (*numGhostVertices)];
|
||||
|
||||
*S = (*numGhostVertices); // Initialize S with number of Ghost Vertices
|
||||
|
||||
/*
|
||||
* Create the Queue Data Structure for the Dominating Set
|
||||
*
|
||||
* I had to declare the staticuQueue U before the parallel region
|
||||
* to have it in the correct scope. Since we can't change the dimension
|
||||
* of a staticQueue I had to destroy the previous object and instantiate
|
||||
* a new one of the correct size.
|
||||
*/
|
||||
//new (&U) staticQueue(NLVer + (*numGhostVertices));
|
||||
U.reserve(NLVer + (*numGhostVertices));
|
||||
|
||||
// Initialize the private vectors
|
||||
privateQLocalVtx.reserve(*numGhostVertices);
|
||||
privateQGhostVtx.reserve(*numGhostVertices);
|
||||
privateQMsgType.reserve(*numGhostVertices);
|
||||
privateQOwner.reserve(*numGhostVertices);
|
||||
privateU.reserve(*numGhostVertices);
|
||||
} // end of task
|
||||
|
||||
} // End of single region
|
||||
} // End of parallel region
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
/**
|
||||
* //TODO documentation
|
||||
* @param k
|
||||
* @param verLocInd
|
||||
* @param StartIndex
|
||||
* @param EndIndex
|
||||
* @param GMate
|
||||
* @param Mate
|
||||
* @param Ghost2LocalMap
|
||||
* @return
|
||||
*/
|
||||
bool isAlreadyMatched(MilanLongInt node,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap)
|
||||
{
|
||||
|
||||
/*
|
||||
#pragma omp critical(Mate)
|
||||
{
|
||||
if ((node < StartIndex) || (node > EndIndex)) { //Is it a ghost vertex?
|
||||
result = GMate[Ghost2LocalMap[node]] >= 0;// Already matched
|
||||
} else { //A local vertex
|
||||
result = (Mate[node - StartIndex] >= 0); // Already matched
|
||||
}
|
||||
|
||||
}
|
||||
*/
|
||||
MilanLongInt val;
|
||||
if ((node < StartIndex) || (node > EndIndex)) // if ghost vertex
|
||||
{
|
||||
#pragma omp atomic read
|
||||
val = GMate[Ghost2LocalMap[node]];
|
||||
return val >= 0; // Already matched
|
||||
}
|
||||
|
||||
// If not ghost vertex
|
||||
#pragma omp atomic read
|
||||
val = Mate[node - StartIndex];
|
||||
|
||||
return val >= 0; // Already matched
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
#include <string.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <math.h>
|
||||
#include "psb_base_cbind.h"
|
||||
#include "MatchingAlgorithms.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
psb_i_t dnew_Match_If(psb_i_t ipar, psb_i_t matching, psb_d_t lambda,
|
||||
psb_i_t nr, psb_i_t irp[], psb_i_t ja[],
|
||||
psb_d_t val[], psb_d_t diag[],
|
||||
psb_d_t w[], psb_i_t mate[]);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
psb_i_t dnew_Match_If(psb_i_t ipar, psb_i_t matching, psb_d_t lambda,
|
||||
psb_i_t nr, psb_i_t irp[], psb_i_t ja[],
|
||||
psb_d_t val[], psb_d_t diag[], psb_d_t w[],
|
||||
psb_i_t mate[])
|
||||
{
|
||||
psb_i_t info;
|
||||
psb_i_t i,j,k;
|
||||
psb_i_t ftcoarse=1;
|
||||
psb_i_t cr_it=0, cr_relax_type=0;
|
||||
psb_d_t cr_relax_weight=0.0;
|
||||
|
||||
vector<NODE_T> s;
|
||||
vector<NODE_T> t;
|
||||
vector<VAL_T> weights;
|
||||
vector<NODE_T> mateNode;
|
||||
NODE_T u,v;
|
||||
VAL_T weight;
|
||||
psb_i_t preprocess = matching; // 0 no greedy 1 greedy
|
||||
psb_i_t romaInput = ipar; // 1 sequential 2 parallel
|
||||
// VAL_T lambda = 2; // positive real value
|
||||
psb_d_t aii, ajj, aij, wii, wjj, tmp1, tmp2, minabs, edgnrm;
|
||||
psb_i_t nt; // number of threads, got with 1 for testing purposes.
|
||||
psb_d_t timeDiff;
|
||||
MatchStat pstat;
|
||||
double eps=1e-16;
|
||||
double minweight,maxweight;
|
||||
char *numthreadsenv;
|
||||
|
||||
numthreadsenv=getenv("OMP_NUM_THREADS");
|
||||
if (numthreadsenv) {
|
||||
sscanf(numthreadsenv,"%d",&nt);
|
||||
} else {
|
||||
nt = 1;
|
||||
}
|
||||
|
||||
minabs = 1e300;
|
||||
// fprintf(stderr,"Sanity check: %d %d \n",nr,nc);
|
||||
k=0;
|
||||
for (i=1; i<nr; i++) {
|
||||
for (j=irp[i-1]; j<irp[i]; j++) {
|
||||
v = i-1; // I
|
||||
u = ja[j-1] - 1; // J
|
||||
if (v>u) {
|
||||
// Define Ahat entry
|
||||
aij = val[j-1];
|
||||
aii = diag[v];
|
||||
ajj = diag[u];
|
||||
wii = w[v];
|
||||
wjj = w[u];
|
||||
edgnrm = aii*(wii*wii) + ajj*(wjj*wjj);
|
||||
if (edgnrm > eps) {
|
||||
weight = abs(1.0 - (2*1.0*aij*wii*wjj)/(aii*(wii*wii) + ajj*(wjj*wjj)));
|
||||
} else {
|
||||
weight = eps;
|
||||
}
|
||||
//
|
||||
s.push_back(u);
|
||||
t.push_back(v);
|
||||
weights.push_back(weight);
|
||||
k = k + 1 ;
|
||||
if (weight<minabs) minabs=weight;
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
maxweight = eps;
|
||||
minweight = 1e300;
|
||||
//fprintf(stderr,"minabs %g\n",minabs);
|
||||
for (i=0; i<k; i++) {
|
||||
weights[i] = log(weights[i]/(0.999*minabs));
|
||||
if (weights[i]>maxweight) maxweight=weights[i];
|
||||
if (weights[i]<minweight) minweight=weights[i];
|
||||
}
|
||||
|
||||
if (lambda<0.0){
|
||||
lambda = maxweight-2.0*minweight+eps;
|
||||
if (lambda<0.0) lambda=eps;
|
||||
} else if (lambda >= 0 && lambda <= 1.0){
|
||||
lambda = lambda*eps + (1.0-lambda)*(fmax(maxweight-2.0*minweight,0.0) );
|
||||
}
|
||||
//fprintf(stderr,"Calling matching: pre %d nt %d lambda %g %g %g\n",
|
||||
// preprocess,nt,lambda,maxweight,minweight);
|
||||
|
||||
runRomaWrapper(s,t,weights, nr, mateNode,preprocess,romaInput,lambda ,nt, pstat, timeDiff);
|
||||
/* loop here only makes sense when nr==nz */
|
||||
for (i=0; i< nr; i++) {
|
||||
//fprintf(stderr,"From runRomaWrapper: %d %d\n",i,mateNode[i]);
|
||||
if (mateNode[i]>=0) {
|
||||
mate[i] = mateNode[i]+1;
|
||||
} else {
|
||||
mate[i] = mateNode[i];
|
||||
//fprintf(stderr,"From runRomaWrapper: %d %d\n",i,mateNode[i]);
|
||||
}
|
||||
}
|
||||
return(0);
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_B(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *candidateMate)
|
||||
{
|
||||
|
||||
MilanLongInt v = -1;
|
||||
|
||||
#pragma omp parallel private(v) default(shared) num_threads(NUM_THREAD)
|
||||
{
|
||||
|
||||
#pragma omp for schedule(static)
|
||||
for (v = 0; v < NLVer; v++) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing: " << v + StartIndex << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Start: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
candidateMate[v] = firstComputeCandidateMate(verLocPtr[v], verLocPtr[v + 1], verLocInd, edgeLocWeight);
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
MilanLongInt *S)
|
||||
{
|
||||
// Start: PARALLEL_PROCESS_CROSS_EDGE_B
|
||||
MilanLongInt captureCounter;
|
||||
|
||||
#pragma omp atomic capture
|
||||
captureCounter = --(*edge); // Decrement
|
||||
|
||||
//assert(captureCounter >= 0);
|
||||
|
||||
if (captureCounter == 0)
|
||||
#pragma omp atomic
|
||||
(*S)--; // Decrement S
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Decrementing S: Ghost vertex " << edge << " has received all its messages";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// End: PARALLEL_PROCESS_CROSS_EDGE_B
|
||||
}
|
||||
@@ -0,0 +1,195 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
MilanLongInt v = -1, k = -1, w = -1, adj11 = 0, adj12 = 0, k1 = 0;
|
||||
MilanInt ghostOwner = 0, option, igw;
|
||||
|
||||
#pragma omp parallel private(option, k, w, v, k1, adj11, adj12, ghostOwner) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx, privateQMsgType, privateQOwner) \
|
||||
default(shared) num_threads(NUM_THREAD)
|
||||
|
||||
{
|
||||
#pragma omp for reduction(+ \
|
||||
: PCounter[:numProcs], myCard \
|
||||
[:1], msgInd \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1]) \
|
||||
schedule(static)
|
||||
for (v = 0; v < NLVer; v++) {
|
||||
option = -1;
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
k = candidateMate[v];
|
||||
candidateMate[v] = verLocInd[k];
|
||||
w = candidateMate[v];
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing: " << v + StartIndex << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v + StartIndex << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0)
|
||||
{
|
||||
|
||||
#pragma omp critical(processExposed)
|
||||
{
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap)) {
|
||||
w = computeCandidateMate(verLocPtr[v],
|
||||
verLocPtr[v + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
candidateMate[v] = w;
|
||||
}
|
||||
|
||||
if (w >= 0) {
|
||||
(*myCard)++;
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // w is a ghost vertex
|
||||
option = 2;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v + StartIndex) {
|
||||
option = 1;
|
||||
Mate[v] = w;
|
||||
GMate[Ghost2LocalMap[w]] = v + StartIndex; // w is a Ghost
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
|
||||
if (candidateMate[w - StartIndex] == (v + StartIndex)) {
|
||||
option = 3;
|
||||
Mate[v] = w; // v is local
|
||||
Mate[w - StartIndex] = v + StartIndex; // w is local
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
} // End of if ( candidateMate[w-StartIndex] == (v+StartIndex) )
|
||||
} // End of Else
|
||||
|
||||
} // End of second if
|
||||
|
||||
} // End critical processExposed
|
||||
|
||||
} // End of if(w >=0)
|
||||
else {
|
||||
// This piece of code is executed a really small amount of times
|
||||
adj11 = verLocPtr[v];
|
||||
adj12 = verLocPtr[v + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
}
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
break;
|
||||
case 1:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ")";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
case 2:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message (291):";
|
||||
cout << "\n(" << myRank << ")Local is: " << v + StartIndex << " Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
default:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
break;
|
||||
}
|
||||
|
||||
} // End of for ( v=0; v < NLVer; v++ )
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
} // End of parallel region
|
||||
}
|
||||
@@ -0,0 +1,294 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void processMatchedVertices(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
MilanLongInt adj1, adj2, adj11, adj12, k, k1, v = -1, w = -1, ghostOwner;
|
||||
int option;
|
||||
MilanLongInt mateVal;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
MilanLongInt localVertices = 0;
|
||||
#endif
|
||||
//#pragma omp parallel private(k, w, v, k1, adj1, adj2, adj11, adj12, ghostOwner, option) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx, \
|
||||
privateQMsgType, privateQOwner, UChunkBeingProcessed) \
|
||||
default(shared) num_threads(NUM_THREAD) \
|
||||
reduction(+ \
|
||||
: msgInd[:1], PCounter \
|
||||
[:numProcs], myCard \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1])
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
|
||||
extractUChunk(UChunkBeingProcessed, U, privateU);
|
||||
|
||||
for (MilanLongInt u : UChunkBeingProcessed) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")u: " << u;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((u >= StartIndex) && (u <= EndIndex)) { // Process Only the Local Vertices
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
option = -1;
|
||||
v = verLocInd[k];
|
||||
|
||||
if ((v >= StartIndex) && (v <= EndIndex)) { // If Local Vertex:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")v: " << v << " c(v)= " << candidateMate[v - StartIndex] << " Mate[v]: " << Mate[v];
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
|
||||
candidateMate[v - StartIndex] = w;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
else { // Neighbor is a ghost vertex
|
||||
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[v]] == u)
|
||||
candidateMate[NLVer + Ghost2LocalMap[v]] = -1;
|
||||
if (v != Mate[u - StartIndex])
|
||||
option = 5; // u is local
|
||||
} // End of critical
|
||||
} // End of Else //A Ghost Vertex
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
// No things to do
|
||||
break;
|
||||
case 1:
|
||||
// Found a dominating edge, it is a ghost and candidateMate[NLVer + Ghost2LocalMap[w]] == v
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], SPtr);
|
||||
case 2:
|
||||
|
||||
// Found a dominating edge, it is a ghost
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
|
||||
(*myCard)++;
|
||||
break;
|
||||
case 4:
|
||||
// Could not find a dominating vertex
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
break;
|
||||
case 5:
|
||||
default:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a success message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << v << " Owner is: " << findOwnerOfGhost(v, verDistance, myRank, numProcs) << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
|
||||
(*NumMessagesBundled)++;
|
||||
PCounter[ghostOwner]++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
#pragma omp critical(U)
|
||||
{
|
||||
U.insert(U.end(), privateU.begin(), privateU.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
|
||||
#pragma omp critical(sendMessageTransfer)
|
||||
{
|
||||
|
||||
QLocalVtx.insert(QLocalVtx.end(), privateQLocalVtx.begin(), privateQLocalVtx.end());
|
||||
QGhostVtx.insert(QGhostVtx.end(), privateQGhostVtx.begin(), privateQGhostVtx.end());
|
||||
QMsgType.insert(QMsgType.end(), privateQMsgType.begin(), privateQMsgType.end());
|
||||
QOwner.insert(QOwner.end(), privateQOwner.begin(), privateQOwner.end());
|
||||
}
|
||||
|
||||
privateQLocalVtx.clear();
|
||||
privateQGhostVtx.clear();
|
||||
privateQMsgType.clear();
|
||||
privateQOwner.clear();
|
||||
|
||||
} // End of while ( !U.empty() )
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices,
|
||||
omp_get_thread_num(),
|
||||
myRank);
|
||||
|
||||
#endif
|
||||
} // End of parallel region
|
||||
}
|
||||
@@ -0,0 +1,308 @@
|
||||
#include "MatchBoxPC.h"
|
||||
//#define DEBUG_HANG_
|
||||
void processMatchedVerticesAndSendMessages(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message)
|
||||
{
|
||||
|
||||
MilanLongInt initialSize = QLocalVtx.size();
|
||||
MilanLongInt adj1, adj2, adj11, adj12, k, k1, v = -1, w = -1, ghostOwner;
|
||||
int option;
|
||||
MilanLongInt mateVal;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
MilanLongInt localVertices = 0;
|
||||
#endif
|
||||
//#pragma omp parallel private(k, w, v, k1, adj1, adj2, adj11, adj12, ghostOwner, option) \
|
||||
firstprivate(Message, privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx,\
|
||||
privateQMsgType, privateQOwner, UChunkBeingProcessed) default(shared) \
|
||||
num_threads(NUM_THREAD) \
|
||||
reduction(+ \
|
||||
: msgInd[:1], PCounter \
|
||||
[:numProcs], myCard \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1], msgActual \
|
||||
[:1])
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
|
||||
extractUChunk(UChunkBeingProcessed, U, privateU);
|
||||
|
||||
for (MilanLongInt u : UChunkBeingProcessed) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")u: " << u;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((u >= StartIndex) && (u <= EndIndex)) { // Process Only the Local Vertices
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
option = -1;
|
||||
v = verLocInd[k];
|
||||
|
||||
if ((v >= StartIndex) && (v <= EndIndex)) { // If Local Vertex:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")v: " << v << " c(v)= " << candidateMate[v - StartIndex] << " Mate[v]: " << Mate[v];
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
|
||||
candidateMate[v - StartIndex] = w;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
else { // Neighbor is a ghost vertex
|
||||
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[v]] == u)
|
||||
candidateMate[NLVer + Ghost2LocalMap[v]] = -1;
|
||||
if (v != Mate[u - StartIndex])
|
||||
option = 5; // u is local
|
||||
} // End of critical
|
||||
} // End of Else //A Ghost Vertex
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
// No things to do
|
||||
break;
|
||||
case 1:
|
||||
// Found a dominating edge, it is a ghost and candidateMate[NLVer + Ghost2LocalMap[w]] == v
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], SPtr);
|
||||
case 2:
|
||||
|
||||
// Found a dominating edge, it is a ghost
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
// Message[2] = REQUEST; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
(*myCard)++;
|
||||
break;
|
||||
case 4:
|
||||
// Could not find a dominating vertex
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
// Message[2] = FAILURE; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
break;
|
||||
case 5:
|
||||
default:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a success message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << v << " Owner is: " << findOwnerOfGhost(v, verDistance, myRank, numProcs) << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = u; // LOCAL
|
||||
// Message[1] = v; // GHOST
|
||||
// Message[2] = SUCCESS; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
} // End of while ( !U.empty() )
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices,
|
||||
omp_get_thread_num(),
|
||||
myRank);
|
||||
|
||||
#endif
|
||||
} // End of parallel region
|
||||
|
||||
// Send the messages
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank<<" Sending: "<<QOwner.size()-initialSize<<" messages" <<endl;
|
||||
#endif
|
||||
for (int i = initialSize; i < QOwner.size(); i++) {
|
||||
|
||||
Message[0] = QLocalVtx[i];
|
||||
Message[1] = QGhostVtx[i];
|
||||
Message[2] = QMsgType[i];
|
||||
ghostOwner = QOwner[i];
|
||||
|
||||
//MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
}
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank<<" Done sending messages"<<endl;
|
||||
#endif
|
||||
}
|
||||
@@ -0,0 +1,315 @@
|
||||
#include "MatchBoxPC.h"
|
||||
//#define DEBUG_HANG_
|
||||
|
||||
void processMessages(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *msgActual,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *S,
|
||||
vector<MilanLongInt> &U)
|
||||
{
|
||||
|
||||
//#define PRINT_DEBUG_INFO_
|
||||
|
||||
MilanInt Sender;
|
||||
MPI_Status computeStatus;
|
||||
MilanLongInt bundleSize, w;
|
||||
MilanLongInt adj11, adj12, k1;
|
||||
MilanLongInt ghostOwner;
|
||||
int error_codeC;
|
||||
error_codeC = MPI_Comm_set_errhandler(MPI_COMM_WORLD, MPI_ERRORS_RETURN);
|
||||
char error_message[MPI_MAX_ERROR_STRING];
|
||||
int message_length;
|
||||
MilanLongInt message_type = 0;
|
||||
|
||||
// Buffer to receive bundled messages
|
||||
// Maximum messages that can be received from any processor is
|
||||
// twice the edge cut: REQUEST; REQUEST+(FAILURE/SUCCESS)
|
||||
vector<MilanLongInt> ReceiveBuffer;
|
||||
try
|
||||
{
|
||||
ReceiveBuffer.reserve(numGhostEdges * 2 * 3); // Three integers per cross edge
|
||||
}
|
||||
catch (length_error)
|
||||
{
|
||||
cout << "Error in function algoDistEdgeApproxDominatingEdgesMessageBundling: \n";
|
||||
cout << "Not enough memory to allocate the internal variables \n";
|
||||
exit(1);
|
||||
}
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout
|
||||
<< "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")About to begin Message processing phase ... *S=" << *S << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// BLOCKING RECEIVE:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << " Waiting for blocking receive..." << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
//cout << myRank<<" Receiving ...";
|
||||
error_codeC = MPI_Recv(&Message[0], 3, TypeMap<MilanLongInt>(), MPI_ANY_SOURCE, ComputeTag, comm, &computeStatus);
|
||||
if (error_codeC != MPI_SUCCESS)
|
||||
{
|
||||
MPI_Error_string(error_codeC, error_message, &message_length);
|
||||
cout << "\n*Error in call to MPI_Receive on Slave: " << error_message << "\n";
|
||||
fflush(stdout);
|
||||
}
|
||||
Sender = computeStatus.MPI_SOURCE;
|
||||
//cout << " ...from "<<Sender << endl;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received message from Process " << Sender << " Type= " << Message[2] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
if (Message[2] == SIZEINFO) {
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received bundled message from Process " << Sender << " Size= " << Message[0] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
bundleSize = Message[0]; //#of integers in the message
|
||||
// Build the Message Buffer:
|
||||
if (!ReceiveBuffer.empty())
|
||||
ReceiveBuffer.clear(); // Empty it out first
|
||||
ReceiveBuffer.resize(bundleSize, -1); // Initialize
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message Bundle Before: " << endl;
|
||||
for (int i = 0; i < bundleSize; i++)
|
||||
cout << ReceiveBuffer[i] << ",";
|
||||
cout << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Receive the message
|
||||
//cout << myRank<<" Receiving from "<<Sender<<endl;
|
||||
error_codeC = MPI_Recv(&ReceiveBuffer[0], bundleSize, TypeMap<MilanLongInt>(), Sender, BundleTag, comm, &computeStatus);
|
||||
if (error_codeC != MPI_SUCCESS) {
|
||||
MPI_Error_string(error_codeC, error_message, &message_length);
|
||||
cout << "\n*Error in call to MPI_Receive on processor " << myRank << " Error: " << error_message << "\n";
|
||||
fflush(stdout);
|
||||
}
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message Bundle After: " << endl;
|
||||
for (int i = 0; i < bundleSize; i++)
|
||||
cout << ReceiveBuffer[i] << ",";
|
||||
cout << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} else { // Just a single message:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received regular message from Process " << Sender << " u= " << Message[0] << " v= " << Message[1] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Add the current message to Queue:
|
||||
bundleSize = 3; //#of integers in the message
|
||||
// Build the Message Buffer:
|
||||
if (!ReceiveBuffer.empty())
|
||||
ReceiveBuffer.clear(); // Empty it out first
|
||||
ReceiveBuffer.resize(bundleSize, -1); // Initialize
|
||||
|
||||
ReceiveBuffer[0] = Message[0]; // u
|
||||
ReceiveBuffer[1] = Message[1]; // v
|
||||
ReceiveBuffer[2] = Message[2]; // message_type
|
||||
}
|
||||
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < StartIndex) || (v > EndIndex)) {
|
||||
cout << "\n(" << myRank << ") From ReceiveBuffer: This should not happen: u= " << u << " v= " << v << " Type= " << message_type << " StartIndex " << StartIndex << " EndIndex " << EndIndex << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing message: u= " << u << " v= " << v << " Type= " << message_type << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Most of the time bundleSize == 3, thus, it's not worth parallelizing thi loop
|
||||
for (MilanLongInt bundleCounter = 3; bundleCounter < bundleSize + 3; bundleCounter += 3) {
|
||||
u = ReceiveBuffer[bundleCounter - 3]; // GHOST
|
||||
v = ReceiveBuffer[bundleCounter - 2]; // LOCAL
|
||||
message_type = ReceiveBuffer[bundleCounter - 1]; // TYPE
|
||||
|
||||
// CASE I: REQUEST
|
||||
if (message_type == REQUEST) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is REQUEST" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < 0) || (v < StartIndex) || ((v - StartIndex) > NLVer)) {
|
||||
cout << "\n(" << myRank << ") case 1 Bad address " << v << " " << StartIndex << " " << v - StartIndex << " " << NLVer << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
if (Mate[v - StartIndex] == -1) {
|
||||
// Process only if not already matched (v is local)
|
||||
candidateMate[NLVer + Ghost2LocalMap[u]] = v; // Set CandidateMate for the ghost
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
GMate[Ghost2LocalMap[u]] = v; // u is ghost
|
||||
Mate[v - StartIndex] = u; // v is local
|
||||
U.push_back(v);
|
||||
U.push_back(u);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << u << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S);
|
||||
} // End of if ( candidateMate[v-StartIndex] == u )e
|
||||
} // End of if ( Mate[v] == -1 )
|
||||
} // End of REQUEST
|
||||
else { // CASE II: SUCCESS
|
||||
if (message_type == SUCCESS) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is SUCCESS" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
GMate[Ghost2LocalMap[u]] = EndIndex + 1; // Set a Dummy Mate to make sure that we do not (u is a ghost) process it again
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S);
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < 0) || (v < StartIndex) || ((v - StartIndex) > NLVer)) {
|
||||
cout << "\n(" << myRank << ") case 2 Bad address " << v << " " << StartIndex << " " << v - StartIndex << " " << NLVer << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
#endif
|
||||
if (Mate[v - StartIndex] == -1) {
|
||||
// Process only if not already matched ( v is local)
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex], verLocPtr[v - StartIndex + 1], edgeLocWeight, k,
|
||||
verLocInd, StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) {
|
||||
// w is a ghost
|
||||
// Build the Message Packet:
|
||||
Message[0] = v; // LOCAL
|
||||
Message[1] = w; // GHOST
|
||||
Message[2] = REQUEST; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
//assert(ghostOwner != -1);
|
||||
//assert(ghostOwner != myRank);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
(*msgInd)++;
|
||||
(*msgActual)++;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
Mate[v - StartIndex] = w; // v is local
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is ghost
|
||||
U.push_back(v);
|
||||
U.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
Mate[v - StartIndex] = w; // v is local
|
||||
Mate[w - StartIndex] = v; // w is local
|
||||
// Q.push_back(u);
|
||||
U.push_back(v);
|
||||
U.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else { // No dominant edge found
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) {
|
||||
// A ghost
|
||||
// Build the Message Packet:
|
||||
Message[0] = v; // LOCAL
|
||||
Message[1] = w; // GHOST
|
||||
Message[2] = FAILURE; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
//assert(ghostOwner != -1);
|
||||
//assert(ghostOwner != myRank);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
(*msgInd)++;
|
||||
(*msgActual)++;
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
} // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of if ( candidateMate[v-StartIndex] == u )
|
||||
} // End of if ( Mate[v] == -1 )
|
||||
} // End of if ( message_type == SUCCESS )
|
||||
else {
|
||||
// CASE III: FAILURE
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is FAILURE" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
GMate[Ghost2LocalMap[u]] = EndIndex + 1; // Set a Dummy Mate to make sure that we do not (u is a ghost) process this anymore
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S); // Decrease the counter
|
||||
} // End of else: CASE III
|
||||
} // End of else: CASE I
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void queuesTransfer(vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
#pragma omp critical(U)
|
||||
{
|
||||
U.insert(U.end(), privateU.begin(), privateU.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
|
||||
#pragma omp critical(sendMessageTransfer)
|
||||
{
|
||||
|
||||
QLocalVtx.insert(QLocalVtx.end(), privateQLocalVtx.begin(), privateQLocalVtx.end());
|
||||
QGhostVtx.insert(QGhostVtx.end(), privateQGhostVtx.begin(), privateQGhostVtx.end());
|
||||
QMsgType.insert(QMsgType.end(), privateQMsgType.begin(), privateQMsgType.end());
|
||||
QOwner.insert(QOwner.end(), privateQOwner.begin(), privateQOwner.end());
|
||||
}
|
||||
|
||||
privateQLocalVtx.clear();
|
||||
privateQGhostVtx.clear();
|
||||
privateQMsgType.clear();
|
||||
privateQOwner.clear();
|
||||
|
||||
}
|
||||
@@ -0,0 +1,209 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void sendBundledMessages(MilanLongInt *numGhostEdges,
|
||||
MilanInt *BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActual,
|
||||
MilanLongInt *msgInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus)
|
||||
{
|
||||
|
||||
MilanLongInt myIndex = 0, numMessagesToSend;
|
||||
MilanInt i = 0, OneMessageSize = 0;
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0)
|
||||
cout << "\n(" << myRank << ") Send Bundles" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#pragma omp parallel private(i) default(shared) num_threads(NUM_THREAD)
|
||||
{
|
||||
#pragma omp master
|
||||
{
|
||||
// Data structures for Bundled Messages:
|
||||
#pragma omp task depend(inout \
|
||||
: PCumulative, PMessageBundle, PSizeInfoMessages) depend(in \
|
||||
: NumMessagesBundled, numProcs)
|
||||
{
|
||||
try {
|
||||
PMessageBundle.reserve(NumMessagesBundled * 3); // Three integers per message
|
||||
PCumulative.reserve(numProcs + 1); // Similar to Row Pointer vector in CSR data structure
|
||||
PSizeInfoMessages.reserve(numProcs * 3); // Buffer to hold the Size info message packets
|
||||
}
|
||||
catch (length_error)
|
||||
{
|
||||
cout << "Error in function algoDistEdgeApproxDominatingEdgesMessageBundling: \n";
|
||||
cout << "Not enough memory to allocate the internal variables \n";
|
||||
exit(1);
|
||||
}
|
||||
PMessageBundle.resize(NumMessagesBundled * 3, -1); // Initialize
|
||||
PCumulative.resize(numProcs + 1, 0); // Only initialize the counter variable
|
||||
PSizeInfoMessages.resize(numProcs * 3, 0);
|
||||
}
|
||||
|
||||
#pragma omp task depend(inout \
|
||||
: PCumulative) depend(in \
|
||||
: PCounter)
|
||||
{
|
||||
for (i = 0; i < numProcs; i++)
|
||||
PCumulative[i + 1] = PCumulative[i] + PCounter[i];
|
||||
}
|
||||
|
||||
#pragma omp task depend(inout \
|
||||
: PCounter)
|
||||
{
|
||||
// Reuse PCounter to keep track of how many messages were inserted:
|
||||
for (MilanInt i = 0; i < numProcs; i++) // Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
PCounter[i] = 0;
|
||||
}
|
||||
|
||||
// Build the Message Bundle packet:
|
||||
#pragma omp task depend(in \
|
||||
: PCounter, QLocalVtx, QGhostVtx, QMsgType, QOwner, PMessageBundle, PCumulative) depend(out \
|
||||
: myIndex, PMessageBundle, PCounter)
|
||||
{
|
||||
for (i = 0; i < NumMessagesBundled; i++) {
|
||||
myIndex = (PCumulative[QOwner[i]] + PCounter[QOwner[i]]) * 3;
|
||||
PMessageBundle[myIndex + 0] = QLocalVtx[i];
|
||||
PMessageBundle[myIndex + 1] = QGhostVtx[i];
|
||||
PMessageBundle[myIndex + 2] = QMsgType[i];
|
||||
PCounter[QOwner[i]]++;
|
||||
}
|
||||
}
|
||||
|
||||
// Send the Bundled Messages: Use ISend
|
||||
#pragma omp task depend(out \
|
||||
: SRequest, SStatus)
|
||||
{
|
||||
try
|
||||
{
|
||||
SRequest.reserve(numProcs * 2); // At most two messages per processor
|
||||
SStatus.reserve(numProcs * 2); // At most two messages per processor
|
||||
}
|
||||
catch (length_error)
|
||||
{
|
||||
cout << "Error in function algoDistEdgeApproxDominatingEdgesLinearSearchImmediateSend: \n";
|
||||
cout << "Not enough memory to allocate the internal variables \n";
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
// Send the Messages
|
||||
#pragma omp task depend(inout \
|
||||
: SRequest, PSizeInfoMessages, PCumulative) depend(out \
|
||||
: *msgActual, *msgInd)
|
||||
{
|
||||
for (i = 0; i < numProcs; i++) { // Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
if (i == myRank) // Do not send anything to yourself
|
||||
continue;
|
||||
// Send the Message with information about the size of next message:
|
||||
// Build the Message Packet:
|
||||
PSizeInfoMessages[i * 3 + 0] = (PCumulative[i + 1] - PCumulative[i]) * 3; // # of integers in the next message
|
||||
PSizeInfoMessages[i * 3 + 1] = -1; // Dummy packet
|
||||
PSizeInfoMessages[i * 3 + 2] = SIZEINFO; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending bundled message to process " << i << " size: " << PSizeInfoMessages[i * 3 + 0] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if (PSizeInfoMessages[i * 3 + 0] > 0)
|
||||
{ // Send only if it is a nonempty packet
|
||||
MPI_Isend(&PSizeInfoMessages[i * 3 + 0], 3, TypeMap<MilanLongInt>(), i, ComputeTag, comm,
|
||||
&SRequest[(*msgInd)]);
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
// Now Send the message with the data packet:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")SendiFFng Bundle to : " << i << endl;
|
||||
for (k = (PCumulative[i] * 3); k < (PCumulative[i] * 3 + PSizeInfoMessages[i * 3 + 0]); k++)
|
||||
cout << PMessageBundle[k] << ",";
|
||||
cout << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
MPI_Isend(&PMessageBundle[PCumulative[i] * 3], PSizeInfoMessages[i * 3 + 0],
|
||||
TypeMap<MilanLongInt>(), i, BundleTag, comm, &SRequest[(*msgInd)]);
|
||||
(*msgInd)++;
|
||||
} // End of if size > 0
|
||||
}
|
||||
}
|
||||
|
||||
#pragma omp task depend(inout \
|
||||
: PCumulative, QLocalVtx, QGhostVtx, QMsgType, QOwner)
|
||||
{
|
||||
|
||||
// Free up temporary memory:
|
||||
PCumulative.clear();
|
||||
QLocalVtx.clear();
|
||||
QGhostVtx.clear();
|
||||
QMsgType.clear();
|
||||
QOwner.clear();
|
||||
}
|
||||
|
||||
#pragma omp task depend(inout : OneMessageSize, *BufferSize) depend(out : numMessagesToSend) depend(in : *numGhostEdges)
|
||||
{
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Number of Ghost edges = " << *numGhostEdges;
|
||||
cout << "\n(" << myRank << ")Total number of potential message X 2 = " << *numGhostEdges * 2;
|
||||
cout << "\n(" << myRank << ")Number messages already sent in bundles = " << NumMessagesBundled;
|
||||
if (*numGhostEdges > 0)
|
||||
{
|
||||
cout << "\n(" << myRank << ")Percentage of total = " << ((double)NumMessagesBundled / (double)(*numGhostEdges * 2)) * 100.0 << "% \n";
|
||||
}
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Allocate memory for MPI Send messages:
|
||||
/* WILL COME BACK HERE - NO NEED TO STORE ALL THIS MEMORY !! */
|
||||
OneMessageSize = 0;
|
||||
MPI_Pack_size(3, TypeMap<MilanLongInt>(), comm, &OneMessageSize); // Size of one message packet
|
||||
// How many messages to send?
|
||||
// Potentially three kinds of messages will be sent/received:
|
||||
// Request, Success, Failure.
|
||||
// But only two will be sent from a given processor.
|
||||
// Substract the number of messages that have already been sent as bundled messages:
|
||||
numMessagesToSend = (*numGhostEdges) * 2 - NumMessagesBundled;
|
||||
*BufferSize = (OneMessageSize + MPI_BSEND_OVERHEAD) * numMessagesToSend;
|
||||
}
|
||||
|
||||
#pragma omp task depend(out : Buffer) depend(in : *BufferSize)
|
||||
{
|
||||
Buffer = 0;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Size of One Message from PACK= " << OneMessageSize;
|
||||
cout << "\n(" << myRank << ")Size of Message overhead = " << MPI_BSEND_OVERHEAD;
|
||||
cout << "\n(" << myRank << ")Number of Ghost edges = " << *numGhostEdges;
|
||||
cout << "\n(" << myRank << ")Number of remaining message = " << numMessagesToSend;
|
||||
cout << "\n(" << myRank << ")BufferSize = " << (*BufferSize);
|
||||
cout << "\n(" << myRank << ")Attaching Buffer on.. ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((*BufferSize) > 0)
|
||||
{
|
||||
Buffer = (MilanLongInt *)malloc((*BufferSize)); // Allocate memory
|
||||
if (Buffer == 0)
|
||||
{
|
||||
cout << "Error in function algoDistEdgeApproxDominatingEdgesLinearSearch: \n";
|
||||
cout << "Not enough memory to allocate for send buffer on process " << myRank << "\n";
|
||||
exit(1);
|
||||
}
|
||||
MPI_Buffer_attach(Buffer, *BufferSize); // Attach the Buffer
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -109,6 +109,8 @@ subroutine amg_c_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
type(psb_cspmat_type) :: ac, op_restr, op_prol
|
||||
integer(psb_ipk_) :: nzl, inl
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
integer(psb_ipk_), save :: idx_matbld=-1, idx_matasb=-1, idx_mapbld=-1
|
||||
logical, parameter :: do_timings=.false.
|
||||
|
||||
name='amg_c_onelev_mat_asb'
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -120,6 +122,12 @@ subroutine amg_c_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
if ((do_timings).and.(idx_matbld==-1)) &
|
||||
& idx_matbld = psb_get_timer_idx("LEV_MASB: mat_bld")
|
||||
if ((do_timings).and.(idx_matasb==-1)) &
|
||||
& idx_matasb = psb_get_timer_idx("LEV_MASB: mat_asb")
|
||||
if ((do_timings).and.(idx_mapbld==-1)) &
|
||||
& idx_mapbld = psb_get_timer_idx("LEV_MASB: map_bld")
|
||||
|
||||
call amg_check_def(lv%parms%aggr_prol,'Smoother',&
|
||||
& amg_smooth_prol_,is_legal_ml_aggr_prol)
|
||||
@@ -139,9 +147,10 @@ subroutine amg_c_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
! the mapping defined by amg_aggrmap_bld and applying the aggregation
|
||||
! algorithm specified by lv%iprcparm(amg_aggr_prol_)
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_matbld)
|
||||
call lv%aggr%mat_bld(lv%parms,a,desc_a,ilaggr,nlaggr,&
|
||||
& lv%ac,lv%desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_matbld)
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='amg_aggrmat_asb')
|
||||
goto 9999
|
||||
@@ -151,14 +160,17 @@ subroutine amg_c_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
! Now build its descriptor and convert global indices for
|
||||
! ac, op_restr and op_prol
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_matasb)
|
||||
if (info == psb_success_) &
|
||||
& call lv%aggr%mat_asb(lv%parms,a,desc_a,&
|
||||
& lv%ac,lv%desc_ac,op_prol,op_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_matasb)
|
||||
if (do_timings) call psb_tic(idx_mapbld)
|
||||
if (info == psb_success_) call lv%ac%cscnv(info,type='csr',dupl=psb_dupl_add_)
|
||||
|
||||
if (info == psb_success_) call lv%aggr%bld_map(desc_a, lv%desc_ac,&
|
||||
& ilaggr,nlaggr,op_restr,op_prol,lv%linmap,info)
|
||||
if (do_timings) call psb_toc(idx_mapbld)
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='mat_asb/map_bld')
|
||||
goto 9999
|
||||
|
||||
@@ -43,6 +43,7 @@ subroutine amg_d_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
use amg_d_dec_aggregator_mod
|
||||
use amg_d_symdec_aggregator_mod
|
||||
use amg_d_parmatch_aggregator_mod
|
||||
use amg_d_newmatch_aggregator_mod
|
||||
use amg_d_jac_smoother
|
||||
use amg_d_as_smoother
|
||||
use amg_d_diag_solver
|
||||
@@ -252,8 +253,6 @@ subroutine amg_d_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
lv%parms%ml_cycle = amg_stringval(val)
|
||||
|
||||
case ('PAR_AGGR_ALG')
|
||||
ival = amg_stringval(val)
|
||||
lv%parms%par_aggr_alg = ival
|
||||
if (allocated(lv%aggr)) then
|
||||
call lv%aggr%free(info)
|
||||
if (info == 0) deallocate(lv%aggr,stat=info)
|
||||
@@ -263,6 +262,9 @@ subroutine amg_d_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
end if
|
||||
end if
|
||||
|
||||
ival = amg_stringval(val)
|
||||
lv%parms%par_aggr_alg = ival
|
||||
|
||||
select case(val)
|
||||
case('DEC')
|
||||
allocate(amg_d_dec_aggregator_type :: lv%aggr, stat=info)
|
||||
@@ -270,9 +272,12 @@ subroutine amg_d_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
allocate(amg_d_symdec_aggregator_type :: lv%aggr, stat=info)
|
||||
case('COUP','COUPLED')
|
||||
allocate(amg_d_parmatch_aggregator_type :: lv%aggr, stat=info)
|
||||
case('NEWMTC')
|
||||
allocate(amg_d_newmatch_aggregator_type :: lv%aggr, stat=info)
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
end select
|
||||
|
||||
if (info == psb_success_) call lv%aggr%default()
|
||||
|
||||
case ('AGGR_ORD')
|
||||
|
||||
@@ -109,6 +109,8 @@ subroutine amg_d_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
type(psb_dspmat_type) :: ac, op_restr, op_prol
|
||||
integer(psb_ipk_) :: nzl, inl
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
integer(psb_ipk_), save :: idx_matbld=-1, idx_matasb=-1, idx_mapbld=-1
|
||||
logical, parameter :: do_timings=.false.
|
||||
|
||||
name='amg_d_onelev_mat_asb'
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -120,6 +122,12 @@ subroutine amg_d_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
if ((do_timings).and.(idx_matbld==-1)) &
|
||||
& idx_matbld = psb_get_timer_idx("LEV_MASB: mat_bld")
|
||||
if ((do_timings).and.(idx_matasb==-1)) &
|
||||
& idx_matasb = psb_get_timer_idx("LEV_MASB: mat_asb")
|
||||
if ((do_timings).and.(idx_mapbld==-1)) &
|
||||
& idx_mapbld = psb_get_timer_idx("LEV_MASB: map_bld")
|
||||
|
||||
call amg_check_def(lv%parms%aggr_prol,'Smoother',&
|
||||
& amg_smooth_prol_,is_legal_ml_aggr_prol)
|
||||
@@ -139,9 +147,10 @@ subroutine amg_d_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
! the mapping defined by amg_aggrmap_bld and applying the aggregation
|
||||
! algorithm specified by lv%iprcparm(amg_aggr_prol_)
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_matbld)
|
||||
call lv%aggr%mat_bld(lv%parms,a,desc_a,ilaggr,nlaggr,&
|
||||
& lv%ac,lv%desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_matbld)
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='amg_aggrmat_asb')
|
||||
goto 9999
|
||||
@@ -151,14 +160,17 @@ subroutine amg_d_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
! Now build its descriptor and convert global indices for
|
||||
! ac, op_restr and op_prol
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_matasb)
|
||||
if (info == psb_success_) &
|
||||
& call lv%aggr%mat_asb(lv%parms,a,desc_a,&
|
||||
& lv%ac,lv%desc_ac,op_prol,op_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_matasb)
|
||||
if (do_timings) call psb_tic(idx_mapbld)
|
||||
if (info == psb_success_) call lv%ac%cscnv(info,type='csr',dupl=psb_dupl_add_)
|
||||
|
||||
if (info == psb_success_) call lv%aggr%bld_map(desc_a, lv%desc_ac,&
|
||||
& ilaggr,nlaggr,op_restr,op_prol,lv%linmap,info)
|
||||
if (do_timings) call psb_toc(idx_mapbld)
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='mat_asb/map_bld')
|
||||
goto 9999
|
||||
|
||||
@@ -109,6 +109,8 @@ subroutine amg_s_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
type(psb_sspmat_type) :: ac, op_restr, op_prol
|
||||
integer(psb_ipk_) :: nzl, inl
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
integer(psb_ipk_), save :: idx_matbld=-1, idx_matasb=-1, idx_mapbld=-1
|
||||
logical, parameter :: do_timings=.false.
|
||||
|
||||
name='amg_s_onelev_mat_asb'
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -120,6 +122,12 @@ subroutine amg_s_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
if ((do_timings).and.(idx_matbld==-1)) &
|
||||
& idx_matbld = psb_get_timer_idx("LEV_MASB: mat_bld")
|
||||
if ((do_timings).and.(idx_matasb==-1)) &
|
||||
& idx_matasb = psb_get_timer_idx("LEV_MASB: mat_asb")
|
||||
if ((do_timings).and.(idx_mapbld==-1)) &
|
||||
& idx_mapbld = psb_get_timer_idx("LEV_MASB: map_bld")
|
||||
|
||||
call amg_check_def(lv%parms%aggr_prol,'Smoother',&
|
||||
& amg_smooth_prol_,is_legal_ml_aggr_prol)
|
||||
@@ -139,9 +147,10 @@ subroutine amg_s_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
! the mapping defined by amg_aggrmap_bld and applying the aggregation
|
||||
! algorithm specified by lv%iprcparm(amg_aggr_prol_)
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_matbld)
|
||||
call lv%aggr%mat_bld(lv%parms,a,desc_a,ilaggr,nlaggr,&
|
||||
& lv%ac,lv%desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_matbld)
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='amg_aggrmat_asb')
|
||||
goto 9999
|
||||
@@ -151,14 +160,17 @@ subroutine amg_s_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
! Now build its descriptor and convert global indices for
|
||||
! ac, op_restr and op_prol
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_matasb)
|
||||
if (info == psb_success_) &
|
||||
& call lv%aggr%mat_asb(lv%parms,a,desc_a,&
|
||||
& lv%ac,lv%desc_ac,op_prol,op_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_matasb)
|
||||
if (do_timings) call psb_tic(idx_mapbld)
|
||||
if (info == psb_success_) call lv%ac%cscnv(info,type='csr',dupl=psb_dupl_add_)
|
||||
|
||||
if (info == psb_success_) call lv%aggr%bld_map(desc_a, lv%desc_ac,&
|
||||
& ilaggr,nlaggr,op_restr,op_prol,lv%linmap,info)
|
||||
if (do_timings) call psb_toc(idx_mapbld)
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='mat_asb/map_bld')
|
||||
goto 9999
|
||||
|
||||
@@ -109,6 +109,8 @@ subroutine amg_z_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
type(psb_zspmat_type) :: ac, op_restr, op_prol
|
||||
integer(psb_ipk_) :: nzl, inl
|
||||
integer(psb_ipk_) :: debug_level, debug_unit
|
||||
integer(psb_ipk_), save :: idx_matbld=-1, idx_matasb=-1, idx_mapbld=-1
|
||||
logical, parameter :: do_timings=.false.
|
||||
|
||||
name='amg_z_onelev_mat_asb'
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -120,6 +122,12 @@ subroutine amg_z_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
info = psb_success_
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt,me,np)
|
||||
if ((do_timings).and.(idx_matbld==-1)) &
|
||||
& idx_matbld = psb_get_timer_idx("LEV_MASB: mat_bld")
|
||||
if ((do_timings).and.(idx_matasb==-1)) &
|
||||
& idx_matasb = psb_get_timer_idx("LEV_MASB: mat_asb")
|
||||
if ((do_timings).and.(idx_mapbld==-1)) &
|
||||
& idx_mapbld = psb_get_timer_idx("LEV_MASB: map_bld")
|
||||
|
||||
call amg_check_def(lv%parms%aggr_prol,'Smoother',&
|
||||
& amg_smooth_prol_,is_legal_ml_aggr_prol)
|
||||
@@ -139,9 +147,10 @@ subroutine amg_z_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
! the mapping defined by amg_aggrmap_bld and applying the aggregation
|
||||
! algorithm specified by lv%iprcparm(amg_aggr_prol_)
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_matbld)
|
||||
call lv%aggr%mat_bld(lv%parms,a,desc_a,ilaggr,nlaggr,&
|
||||
& lv%ac,lv%desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_matbld)
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='amg_aggrmat_asb')
|
||||
goto 9999
|
||||
@@ -151,14 +160,17 @@ subroutine amg_z_base_onelev_mat_asb(lv,a,desc_a,ilaggr,nlaggr,t_prol,info)
|
||||
! Now build its descriptor and convert global indices for
|
||||
! ac, op_restr and op_prol
|
||||
!
|
||||
if (do_timings) call psb_tic(idx_matasb)
|
||||
if (info == psb_success_) &
|
||||
& call lv%aggr%mat_asb(lv%parms,a,desc_a,&
|
||||
& lv%ac,lv%desc_ac,op_prol,op_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_matasb)
|
||||
if (do_timings) call psb_tic(idx_mapbld)
|
||||
if (info == psb_success_) call lv%ac%cscnv(info,type='csr',dupl=psb_dupl_add_)
|
||||
|
||||
if (info == psb_success_) call lv%aggr%bld_map(desc_a, lv%desc_ac,&
|
||||
& ilaggr,nlaggr,op_restr,op_prol,lv%linmap,info)
|
||||
if (do_timings) call psb_toc(idx_mapbld)
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='mat_asb/map_bld')
|
||||
goto 9999
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
cd amgprec/impl/aggregator/
|
||||
rm MatchBoxPC.o
|
||||
rm sendBundledMessages.o
|
||||
rm initialize.o
|
||||
rm extractUChunk.o
|
||||
rm isAlreadyMatched.o
|
||||
rm findOwnerOfGhost.o
|
||||
rm computeCandidateMate.o
|
||||
rm parallelComputeCandidateMateB.o
|
||||
rm processMatchedVertices.o
|
||||
rm processCrossEdge.o
|
||||
rm queueTransfer.o
|
||||
rm processMessages.o
|
||||
rm processExposedVertex.o
|
||||
rm algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC.o
|
||||
rm algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP.o
|
||||
cd ../../../
|
||||
make all
|
||||
cd samples/advanced/pdegen
|
||||
make amg_d_pde3d
|
||||
cd runs
|
||||
mpirun -np 4 amg_d_pde3d amg_pde3d.inp
|
||||
|
||||
|
||||
|
||||
@@ -3,24 +3,25 @@ AMGINCDIR=$(AMGDIR)/include
|
||||
include $(AMGINCDIR)/Make.inc.amg4psblas
|
||||
AMGMODDIR=$(AMGDIR)/modules
|
||||
AMGLIBDIR=$(AMGDIR)/lib
|
||||
AMG_LIBS=-L$(AMGLIBDIR) -lpsb_krylov -lamg_prec -lpsb_prec
|
||||
AMG_LIBS=-L$(AMGLIBDIR) -lpsb_krylov -lamg_prec -lpsb_prec
|
||||
FINCLUDES=$(FMFLAG). $(FMFLAG)$(AMGMODDIR) $(FMFLAG)$(AMGINCDIR) $(PSBLAS_INCLUDES) $(FIFLAG).
|
||||
|
||||
LINKOPT=
|
||||
XTRALINK=-lstdc++ -lroma -fopenmp
|
||||
EXEDIR=./runs
|
||||
|
||||
all: amg_s_pde3d amg_d_pde3d amg_s_pde2d amg_d_pde2d
|
||||
|
||||
amg_d_pde3d: amg_d_pde3d.o amg_d_genpde_mod.o amg_d_pde3d_base_mod.o amg_d_pde3d_exp_mod.o amg_d_pde3d_gauss_mod.o data_input.o
|
||||
$(FLINK) $(LINKOPT) amg_d_pde3d.o amg_d_genpde_mod.o amg_d_pde3d_base_mod.o amg_d_pde3d_exp_mod.o amg_d_pde3d_gauss_mod.o data_input.o -o amg_d_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
/bin/mv amg_d_pde3d $(EXEDIR)
|
||||
$(FLINK) $(LINKOPT) amg_d_pde3d.o amg_d_genpde_mod.o amg_d_pde3d_base_mod.o amg_d_pde3d_exp_mod.o amg_d_pde3d_gauss_mod.o data_input.o -o amg_d_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS) $(XTRALINK)
|
||||
/bin/mv amg_d_pde3d $(EXEDIR)
|
||||
|
||||
amg_s_pde3d: amg_s_pde3d.o amg_s_genpde_mod.o amg_s_pde3d_base_mod.o amg_s_pde3d_exp_mod.o amg_s_pde3d_gauss_mod.o data_input.o
|
||||
$(FLINK) $(LINKOPT) amg_s_pde3d.o amg_s_genpde_mod.o amg_s_pde3d_base_mod.o amg_s_pde3d_exp_mod.o amg_s_pde3d_gauss_mod.o data_input.o -o amg_s_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
/bin/mv amg_s_pde3d $(EXEDIR)
|
||||
|
||||
amg_d_pde2d: amg_d_pde2d.o amg_d_genpde_mod.o amg_d_pde2d_base_mod.o amg_d_pde2d_exp_mod.o amg_d_pde2d_box_mod.o data_input.o
|
||||
$(FLINK) $(LINKOPT) amg_d_pde2d.o amg_d_genpde_mod.o amg_d_pde2d_base_mod.o amg_d_pde2d_exp_mod.o amg_d_pde2d_box_mod.o data_input.o -o amg_d_pde2d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
$(FLINK) $(LINKOPT) amg_d_pde2d.o amg_d_genpde_mod.o amg_d_pde2d_base_mod.o amg_d_pde2d_exp_mod.o amg_d_pde2d_box_mod.o data_input.o -o amg_d_pde2d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS) $(XTRALINK)
|
||||
/bin/mv amg_d_pde2d $(EXEDIR)
|
||||
|
||||
amg_s_pde2d: amg_s_pde2d.o amg_s_genpde_mod.o amg_s_pde2d_base_mod.o amg_s_pde2d_exp_mod.o amg_s_pde2d_box_mod.o data_input.o
|
||||
|
||||
@@ -93,6 +93,9 @@ contains
|
||||
& a1,a2,a3,b1,b2,b3,c,g,info,f,amold,vmold,partition, nrl,iv)
|
||||
use psb_base_mod
|
||||
use psb_util_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
!
|
||||
! Discretizes the partial differential equation
|
||||
!
|
||||
@@ -128,7 +131,6 @@ contains
|
||||
type(psb_d_csc_sparse_mat) :: acsc
|
||||
type(psb_d_coo_sparse_mat) :: acoo
|
||||
type(psb_d_csr_sparse_mat) :: acsr
|
||||
real(psb_dpk_) :: zt(nb),x,y,z,xph,xmh,yph,ymh,zph,zmh
|
||||
integer(psb_ipk_) :: nnz,nr,nlr,i,j,ii,ib,k, partition_
|
||||
integer(psb_lpk_) :: m,n,glob_row,nt
|
||||
integer(psb_ipk_) :: ix,iy,iz,ia,indx_owner
|
||||
@@ -141,8 +143,7 @@ contains
|
||||
! Process grid
|
||||
integer(psb_ipk_) :: np, iam
|
||||
integer(psb_ipk_) :: icoeff
|
||||
integer(psb_lpk_), allocatable :: irow(:),icol(:),myidx(:)
|
||||
real(psb_dpk_), allocatable :: val(:)
|
||||
integer(psb_lpk_), allocatable :: myidx(:)
|
||||
! deltah dimension of each grid cell
|
||||
! deltat discretization time
|
||||
real(psb_dpk_) :: deltah, sqdeltah, deltah2
|
||||
@@ -368,119 +369,128 @@ contains
|
||||
call psb_barrier(ctxt)
|
||||
talc = psb_wtime()-t0
|
||||
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
ch_err='allocation rout.'
|
||||
call psb_errpush(info,name,a_err=ch_err)
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
! we build an auxiliary matrix consisting of one row at a
|
||||
! time; just a small matrix. might be extended to generate
|
||||
! a bunch of rows per call.
|
||||
!
|
||||
allocate(val(20*nb),irow(20*nb),&
|
||||
&icol(20*nb),stat=info)
|
||||
if (info /= psb_success_ ) then
|
||||
info=psb_err_alloc_dealloc_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
endif
|
||||
|
||||
|
||||
! loop over rows belonging to current process in a block
|
||||
! distribution.
|
||||
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
do ii=1, nlr,nb
|
||||
ib = min(nb,nlr-ii+1)
|
||||
icoeff = 1
|
||||
do k=1,ib
|
||||
i=ii+k-1
|
||||
! local matrix pointer
|
||||
glob_row=myidx(i)
|
||||
! compute gridpoint coordinates
|
||||
call idx2ijk(ix,iy,iz,glob_row,idim,idim,idim)
|
||||
! x, y, z coordinates
|
||||
x = (ix-1)*deltah
|
||||
y = (iy-1)*deltah
|
||||
z = (iz-1)*deltah
|
||||
zt(k) = f_(x,y,z)
|
||||
! internal point: build discretization
|
||||
!
|
||||
! term depending on (x-1,y,z)
|
||||
!
|
||||
val(icoeff) = -a1(x,y,z)/sqdeltah-b1(x,y,z)/deltah2
|
||||
if (ix == 1) then
|
||||
zt(k) = g(dzero,y,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix-1,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y-1,z)
|
||||
val(icoeff) = -a2(x,y,z)/sqdeltah-b2(x,y,z)/deltah2
|
||||
if (iy == 1) then
|
||||
zt(k) = g(x,dzero,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy-1,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y,z-1)
|
||||
val(icoeff)=-a3(x,y,z)/sqdeltah-b3(x,y,z)/deltah2
|
||||
if (iz == 1) then
|
||||
zt(k) = g(x,y,dzero)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz-1,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
!$omp parallel shared(deltah,myidx,a,desc_a)
|
||||
!
|
||||
block
|
||||
integer(psb_ipk_) :: i,j,k,ii,ib,icoeff, ix,iy,iz, ith,nth
|
||||
integer(psb_lpk_) :: glob_row
|
||||
integer(psb_lpk_), allocatable :: irow(:),icol(:)
|
||||
real(psb_dpk_), allocatable :: val(:)
|
||||
real(psb_dpk_) :: x,y,z, zt(nb)
|
||||
#if defined(OPENMP)
|
||||
nth = omp_get_num_threads()
|
||||
ith = omp_get_thread_num()
|
||||
#else
|
||||
nth = 1
|
||||
ith = 0
|
||||
#endif
|
||||
allocate(val(20*nb),irow(20*nb),&
|
||||
&icol(20*nb),stat=info)
|
||||
if (info /= psb_success_ ) then
|
||||
info=psb_err_alloc_dealloc_
|
||||
call psb_errpush(info,name)
|
||||
!goto 9999
|
||||
endif
|
||||
|
||||
! term depending on (x,y,z)
|
||||
val(icoeff)=(2*done)*(a1(x,y,z)+a2(x,y,z)+a3(x,y,z))/sqdeltah &
|
||||
& + c(x,y,z)
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
! term depending on (x,y,z+1)
|
||||
val(icoeff)=-a3(x,y,z)/sqdeltah+b3(x,y,z)/deltah2
|
||||
if (iz == idim) then
|
||||
zt(k) = g(x,y,done)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz+1,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y+1,z)
|
||||
val(icoeff)=-a2(x,y,z)/sqdeltah+b2(x,y,z)/deltah2
|
||||
if (iy == idim) then
|
||||
zt(k) = g(x,done,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy+1,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x+1,y,z)
|
||||
val(icoeff)=-a1(x,y,z)/sqdeltah+b1(x,y,z)/deltah2
|
||||
if (ix==idim) then
|
||||
zt(k) = g(done,y,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix+1,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
!$omp do schedule(dynamic)
|
||||
!
|
||||
do ii=1, nlr, nb
|
||||
if (info /= psb_success_) cycle
|
||||
ib = min(nb,nlr-ii+1)
|
||||
icoeff = 1
|
||||
do k=1,ib
|
||||
i=ii+k-1
|
||||
! local matrix pointer
|
||||
glob_row=myidx(i)
|
||||
! compute gridpoint coordinates
|
||||
call idx2ijk(ix,iy,iz,glob_row,idim,idim,idim)
|
||||
! x, y, z coordinates
|
||||
x = (ix-1)*deltah
|
||||
y = (iy-1)*deltah
|
||||
z = (iz-1)*deltah
|
||||
zt(k) = f_(x,y,z)
|
||||
! internal point: build discretization
|
||||
!
|
||||
! term depending on (x-1,y,z)
|
||||
!
|
||||
val(icoeff) = -a1(x,y,z)/sqdeltah-b1(x,y,z)/deltah2
|
||||
if (ix == 1) then
|
||||
zt(k) = g(dzero,y,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix-1,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y-1,z)
|
||||
val(icoeff) = -a2(x,y,z)/sqdeltah-b2(x,y,z)/deltah2
|
||||
if (iy == 1) then
|
||||
zt(k) = g(x,dzero,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy-1,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y,z-1)
|
||||
val(icoeff)=-a3(x,y,z)/sqdeltah-b3(x,y,z)/deltah2
|
||||
if (iz == 1) then
|
||||
zt(k) = g(x,y,dzero)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz-1,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
|
||||
! term depending on (x,y,z)
|
||||
val(icoeff)=(2*done)*(a1(x,y,z)+a2(x,y,z)+a3(x,y,z))/sqdeltah &
|
||||
& + c(x,y,z)
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
! term depending on (x,y,z+1)
|
||||
val(icoeff)=-a3(x,y,z)/sqdeltah+b3(x,y,z)/deltah2
|
||||
if (iz == idim) then
|
||||
zt(k) = g(x,y,done)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz+1,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y+1,z)
|
||||
val(icoeff)=-a2(x,y,z)/sqdeltah+b2(x,y,z)/deltah2
|
||||
if (iy == idim) then
|
||||
zt(k) = g(x,done,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy+1,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x+1,y,z)
|
||||
val(icoeff)=-a1(x,y,z)/sqdeltah+b1(x,y,z)/deltah2
|
||||
if (ix==idim) then
|
||||
zt(k) = g(done,y,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix+1,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
|
||||
end do
|
||||
!write(0,*) ' Outer in_parallel ',omp_in_parallel()
|
||||
call psb_spins(icoeff-1,irow,icol,val,a,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),bv,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
zt(:)=dzero
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),xv,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
end do
|
||||
call psb_spins(icoeff-1,irow,icol,val,a,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),bv,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
zt(:)=dzero
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),xv,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
deallocate(val,irow,icol)
|
||||
end block
|
||||
!$omp end parallel
|
||||
|
||||
tgen = psb_wtime()-t1
|
||||
if(info /= psb_success_) then
|
||||
@@ -490,7 +500,6 @@ contains
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
deallocate(val,irow,icol)
|
||||
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
@@ -557,6 +566,9 @@ contains
|
||||
& a1,a2,b1,b2,c,g,info,f,amold,vmold,partition, nrl,iv)
|
||||
use psb_base_mod
|
||||
use psb_util_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
!
|
||||
! Discretizes the partial differential equation
|
||||
!
|
||||
@@ -591,7 +603,6 @@ contains
|
||||
type(psb_d_csc_sparse_mat) :: acsc
|
||||
type(psb_d_coo_sparse_mat) :: acoo
|
||||
type(psb_d_csr_sparse_mat) :: acsr
|
||||
real(psb_dpk_) :: zt(nb),x,y,z,xph,xmh,yph,ymh,zph,zmh
|
||||
integer(psb_ipk_) :: nnz,nr,nlr,i,j,ii,ib,k, partition_
|
||||
integer(psb_lpk_) :: m,n,glob_row,nt
|
||||
integer(psb_ipk_) :: ix,iy,iz,ia,indx_owner
|
||||
@@ -604,8 +615,7 @@ contains
|
||||
! Process grid
|
||||
integer(psb_ipk_) :: np, iam
|
||||
integer(psb_ipk_) :: icoeff
|
||||
integer(psb_lpk_), allocatable :: irow(:),icol(:),myidx(:)
|
||||
real(psb_dpk_), allocatable :: val(:)
|
||||
integer(psb_lpk_), allocatable :: myidx(:)
|
||||
! deltah dimension of each grid cell
|
||||
! deltat discretization time
|
||||
real(psb_dpk_) :: deltah, sqdeltah, deltah2, dd
|
||||
@@ -791,7 +801,7 @@ contains
|
||||
!write(0,*) iam,' Check on neighbours: ',desc_a%get_p_adjcncy()
|
||||
end if
|
||||
end block
|
||||
|
||||
|
||||
case default
|
||||
write(psb_err_unit,*) iam, 'Initialization error: should not get here'
|
||||
info = -1
|
||||
@@ -816,93 +826,109 @@ contains
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
! we build an auxiliary matrix consisting of one row at a
|
||||
! time; just a small matrix. might be extended to generate
|
||||
! a bunch of rows per call.
|
||||
!
|
||||
allocate(val(20*nb),irow(20*nb),&
|
||||
&icol(20*nb),stat=info)
|
||||
if (info /= psb_success_ ) then
|
||||
info=psb_err_alloc_dealloc_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
endif
|
||||
|
||||
|
||||
! loop over rows belonging to current process in a block
|
||||
! distribution.
|
||||
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
do ii=1, nlr,nb
|
||||
ib = min(nb,nlr-ii+1)
|
||||
icoeff = 1
|
||||
do k=1,ib
|
||||
i=ii+k-1
|
||||
! local matrix pointer
|
||||
glob_row=myidx(i)
|
||||
! compute gridpoint coordinates
|
||||
call idx2ijk(ix,iy,glob_row,idim,idim)
|
||||
! x, y coordinates
|
||||
x = (ix-1)*deltah
|
||||
y = (iy-1)*deltah
|
||||
!$omp parallel shared(deltah,myidx,a,desc_a)
|
||||
!
|
||||
block
|
||||
integer(psb_ipk_) :: i,j,k,ii,ib,icoeff, ix,iy,iz, ith,nth
|
||||
integer(psb_lpk_) :: glob_row
|
||||
integer(psb_lpk_), allocatable :: irow(:),icol(:)
|
||||
real(psb_dpk_), allocatable :: val(:)
|
||||
real(psb_dpk_) :: x,y,z, zt(nb)
|
||||
#if defined(OPENMP)
|
||||
nth = omp_get_num_threads()
|
||||
ith = omp_get_thread_num()
|
||||
#else
|
||||
nth = 1
|
||||
ith = 0
|
||||
#endif
|
||||
allocate(val(20*nb),irow(20*nb),&
|
||||
&icol(20*nb),stat=info)
|
||||
if (info /= psb_success_ ) then
|
||||
info=psb_err_alloc_dealloc_
|
||||
call psb_errpush(info,name)
|
||||
!goto 9999
|
||||
endif
|
||||
|
||||
zt(k) = f_(x,y)
|
||||
! internal point: build discretization
|
||||
!
|
||||
! term depending on (x-1,y)
|
||||
!
|
||||
val(icoeff) = -a1(x,y)/sqdeltah-b1(x,y)/deltah2
|
||||
if (ix == 1) then
|
||||
zt(k) = g(dzero,y)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix-1,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y-1)
|
||||
val(icoeff) = -a2(x,y)/sqdeltah-b2(x,y)/deltah2
|
||||
if (iy == 1) then
|
||||
zt(k) = g(x,dzero)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy-1,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! loop over rows belonging to current process in a block
|
||||
! distribution.
|
||||
!$omp do schedule(dynamic)
|
||||
!
|
||||
do ii=1, nlr,nb
|
||||
ib = min(nb,nlr-ii+1)
|
||||
icoeff = 1
|
||||
do k=1,ib
|
||||
i=ii+k-1
|
||||
! local matrix pointer
|
||||
glob_row=myidx(i)
|
||||
! compute gridpoint coordinates
|
||||
call idx2ijk(ix,iy,glob_row,idim,idim)
|
||||
! x, y coordinates
|
||||
x = (ix-1)*deltah
|
||||
y = (iy-1)*deltah
|
||||
|
||||
! term depending on (x,y)
|
||||
val(icoeff)=(2*done)*(a1(x,y) + a2(x,y))/sqdeltah + c(x,y)
|
||||
call ijk2idx(icol(icoeff),ix,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
! term depending on (x,y+1)
|
||||
val(icoeff)=-a2(x,y)/sqdeltah+b2(x,y)/deltah2
|
||||
if (iy == idim) then
|
||||
zt(k) = g(x,done)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy+1,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x+1,y)
|
||||
val(icoeff)=-a1(x,y)/sqdeltah+b1(x,y)/deltah2
|
||||
if (ix==idim) then
|
||||
zt(k) = g(done,y)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix+1,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
zt(k) = f_(x,y)
|
||||
! internal point: build discretization
|
||||
!
|
||||
! term depending on (x-1,y)
|
||||
!
|
||||
val(icoeff) = -a1(x,y)/sqdeltah-b1(x,y)/deltah2
|
||||
if (ix == 1) then
|
||||
zt(k) = g(dzero,y)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix-1,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y-1)
|
||||
val(icoeff) = -a2(x,y)/sqdeltah-b2(x,y)/deltah2
|
||||
if (iy == 1) then
|
||||
zt(k) = g(x,dzero)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy-1,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
|
||||
! term depending on (x,y)
|
||||
val(icoeff)=(2*done)*(a1(x,y) + a2(x,y))/sqdeltah + c(x,y)
|
||||
call ijk2idx(icol(icoeff),ix,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
! term depending on (x,y+1)
|
||||
val(icoeff)=-a2(x,y)/sqdeltah+b2(x,y)/deltah2
|
||||
if (iy == idim) then
|
||||
zt(k) = g(x,done)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy+1,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x+1,y)
|
||||
val(icoeff)=-a1(x,y)/sqdeltah+b1(x,y)/deltah2
|
||||
if (ix==idim) then
|
||||
zt(k) = g(done,y)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix+1,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
|
||||
end do
|
||||
call psb_spins(icoeff-1,irow,icol,val,a,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),bv,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
zt(:)=dzero
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),xv,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
end do
|
||||
call psb_spins(icoeff-1,irow,icol,val,a,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),bv,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
zt(:)=dzero
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),xv,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
deallocate(val,irow,icol)
|
||||
end block
|
||||
!$omp end parallel
|
||||
|
||||
tgen = psb_wtime()-t1
|
||||
if(info /= psb_success_) then
|
||||
@@ -912,8 +938,6 @@ contains
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
deallocate(val,irow,icol)
|
||||
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
call psb_cdasb(desc_a,info)
|
||||
|
||||
@@ -74,6 +74,9 @@ program amg_d_pde3d
|
||||
use amg_d_pde3d_exp_mod
|
||||
use amg_d_pde3d_gauss_mod
|
||||
use amg_d_genpde_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! input parameters
|
||||
@@ -94,7 +97,7 @@ program amg_d_pde3d
|
||||
type(psb_d_vect_type) :: x,b,r
|
||||
! parallel environment
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: iam, np
|
||||
integer(psb_ipk_) :: iam, np, nth
|
||||
|
||||
! solver parameters
|
||||
integer(psb_ipk_) :: iter, itmax,itrace, istopc, irst, nlv
|
||||
@@ -133,6 +136,9 @@ program amg_d_pde3d
|
||||
character(len=16) :: aggr_ord ! ordering for aggregation: NATURAL, DEGREE
|
||||
character(len=16) :: aggr_filter ! filtering: FILTER, NO_FILTER
|
||||
real(psb_dpk_) :: mncrratio ! minimum aggregation ratio
|
||||
integer(psb_ipk_) :: matching_alg ! For NEW matching 1 2 3 variant
|
||||
real(psb_dpk_) :: lambda ! matching LAMBDA
|
||||
|
||||
real(psb_dpk_), allocatable :: athresv(:) ! smoothed aggregation threshold vector
|
||||
integer(psb_ipk_) :: thrvsz ! size of threshold vector
|
||||
real(psb_dpk_) :: athres ! smoothed aggregation threshold
|
||||
@@ -198,6 +204,15 @@ program amg_d_pde3d
|
||||
|
||||
call psb_init(ctxt)
|
||||
call psb_info(ctxt,iam,np)
|
||||
#if defined(OPENMP)
|
||||
!$OMP parallel shared(nth)
|
||||
!$OMP master
|
||||
nth = omp_get_num_threads()
|
||||
!$OMP end master
|
||||
!$OMP end parallel
|
||||
#else
|
||||
nth = 1
|
||||
#endif
|
||||
|
||||
if (iam < 0) then
|
||||
! This should not happen, but just in case
|
||||
@@ -309,6 +324,11 @@ program amg_d_pde3d
|
||||
call prec%set('par_aggr_alg', p_choice%par_aggr_alg, info)
|
||||
call prec%set('aggr_type', p_choice%aggr_type, info)
|
||||
call prec%set('aggr_size', p_choice%aggr_size, info)
|
||||
write(0,*) 'match variant ',p_choice%matching_alg
|
||||
if (p_choice%matching_alg>0)&
|
||||
& call prec%set('nwm_matching_alg', p_choice%matching_alg, info)
|
||||
if (p_choice%lambda>0)&
|
||||
& call prec%set('nwm_lambda', p_choice%lambda, info)
|
||||
|
||||
call prec%set('aggr_ord', p_choice%aggr_ord, info)
|
||||
call prec%set('aggr_filter', p_choice%aggr_filter,info)
|
||||
@@ -456,6 +476,8 @@ program amg_d_pde3d
|
||||
call prec%descr(info,iout=psb_out_unit)
|
||||
if (iam == psb_root_) then
|
||||
write(psb_out_unit,'("Computed solution on ",i8," processors")') np
|
||||
write(psb_out_unit,'("Number of threads : ",i12)') nth
|
||||
write(psb_out_unit,'("Total number of tasks : ",i12)') nth*np
|
||||
write(psb_out_unit,'("Linear system size : ",i12)') system_size
|
||||
write(psb_out_unit,'("PDE Coefficients : ",a)') trim(pdecoeff)
|
||||
write(psb_out_unit,'("Krylov method : ",a)') trim(s_choice%kmethd)
|
||||
@@ -585,9 +607,10 @@ contains
|
||||
call read_data(prec%aggr_type,inp_unit) ! type of aggregation
|
||||
call read_data(prec%aggr_size,inp_unit) ! Requested size of the aggregates for MATCHBOXP
|
||||
call read_data(prec%aggr_ord,inp_unit) ! ordering for aggregation
|
||||
call read_data(prec%mncrratio,inp_unit) ! minimum aggregation ratio
|
||||
call read_data(prec%aggr_filter,inp_unit) ! filtering
|
||||
call read_data(prec%athres,inp_unit) ! smoothed aggr thresh
|
||||
call read_data(prec%mncrratio,inp_unit) ! minimum aggregation ratio
|
||||
call read_data(prec%matching_alg,inp_unit) ! matching variant
|
||||
call read_data(prec%lambda,inp_unit) ! lambda
|
||||
call read_data(prec%thrvsz,inp_unit) ! size of aggr thresh vector
|
||||
if (prec%thrvsz > 0) then
|
||||
call psb_realloc(prec%thrvsz,prec%athresv,info)
|
||||
@@ -595,6 +618,7 @@ contains
|
||||
else
|
||||
read(inp_unit,*) ! dummy read to skip a record
|
||||
end if
|
||||
call read_data(prec%athres,inp_unit) ! smoothed aggr thresh
|
||||
! coasest-level solver
|
||||
call read_data(prec%csolve,inp_unit) ! coarsest-lev solver
|
||||
call read_data(prec%csbsolve,inp_unit) ! coarsest-lev subsolver
|
||||
@@ -668,6 +692,10 @@ contains
|
||||
call psb_bcast(ctxt,prec%aggr_ord)
|
||||
call psb_bcast(ctxt,prec%aggr_filter)
|
||||
call psb_bcast(ctxt,prec%mncrratio)
|
||||
call psb_bcast(ctxt,prec%matching_alg)
|
||||
call psb_bcast(ctxt,prec%lambda)
|
||||
|
||||
|
||||
call psb_bcast(ctxt,prec%thrvsz)
|
||||
if (prec%thrvsz > 0) then
|
||||
if (iam /= psb_root_) call psb_realloc(prec%thrvsz,prec%athresv,info)
|
||||
@@ -93,6 +93,9 @@ contains
|
||||
& a1,a2,a3,b1,b2,b3,c,g,info,f,amold,vmold,partition, nrl,iv)
|
||||
use psb_base_mod
|
||||
use psb_util_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
!
|
||||
! Discretizes the partial differential equation
|
||||
!
|
||||
@@ -128,7 +131,6 @@ contains
|
||||
type(psb_s_csc_sparse_mat) :: acsc
|
||||
type(psb_s_coo_sparse_mat) :: acoo
|
||||
type(psb_s_csr_sparse_mat) :: acsr
|
||||
real(psb_spk_) :: zt(nb),x,y,z,xph,xmh,yph,ymh,zph,zmh
|
||||
integer(psb_ipk_) :: nnz,nr,nlr,i,j,ii,ib,k, partition_
|
||||
integer(psb_lpk_) :: m,n,glob_row,nt
|
||||
integer(psb_ipk_) :: ix,iy,iz,ia,indx_owner
|
||||
@@ -141,8 +143,7 @@ contains
|
||||
! Process grid
|
||||
integer(psb_ipk_) :: np, iam
|
||||
integer(psb_ipk_) :: icoeff
|
||||
integer(psb_lpk_), allocatable :: irow(:),icol(:),myidx(:)
|
||||
real(psb_spk_), allocatable :: val(:)
|
||||
integer(psb_lpk_), allocatable :: myidx(:)
|
||||
! deltah dimension of each grid cell
|
||||
! deltat discretization time
|
||||
real(psb_spk_) :: deltah, sqdeltah, deltah2
|
||||
@@ -368,119 +369,128 @@ contains
|
||||
call psb_barrier(ctxt)
|
||||
talc = psb_wtime()-t0
|
||||
|
||||
if (info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
ch_err='allocation rout.'
|
||||
call psb_errpush(info,name,a_err=ch_err)
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
! we build an auxiliary matrix consisting of one row at a
|
||||
! time; just a small matrix. might be extended to generate
|
||||
! a bunch of rows per call.
|
||||
!
|
||||
allocate(val(20*nb),irow(20*nb),&
|
||||
&icol(20*nb),stat=info)
|
||||
if (info /= psb_success_ ) then
|
||||
info=psb_err_alloc_dealloc_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
endif
|
||||
|
||||
|
||||
! loop over rows belonging to current process in a block
|
||||
! distribution.
|
||||
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
do ii=1, nlr,nb
|
||||
ib = min(nb,nlr-ii+1)
|
||||
icoeff = 1
|
||||
do k=1,ib
|
||||
i=ii+k-1
|
||||
! local matrix pointer
|
||||
glob_row=myidx(i)
|
||||
! compute gridpoint coordinates
|
||||
call idx2ijk(ix,iy,iz,glob_row,idim,idim,idim)
|
||||
! x, y, z coordinates
|
||||
x = (ix-1)*deltah
|
||||
y = (iy-1)*deltah
|
||||
z = (iz-1)*deltah
|
||||
zt(k) = f_(x,y,z)
|
||||
! internal point: build discretization
|
||||
!
|
||||
! term depending on (x-1,y,z)
|
||||
!
|
||||
val(icoeff) = -a1(x,y,z)/sqdeltah-b1(x,y,z)/deltah2
|
||||
if (ix == 1) then
|
||||
zt(k) = g(szero,y,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix-1,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y-1,z)
|
||||
val(icoeff) = -a2(x,y,z)/sqdeltah-b2(x,y,z)/deltah2
|
||||
if (iy == 1) then
|
||||
zt(k) = g(x,szero,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy-1,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y,z-1)
|
||||
val(icoeff)=-a3(x,y,z)/sqdeltah-b3(x,y,z)/deltah2
|
||||
if (iz == 1) then
|
||||
zt(k) = g(x,y,szero)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz-1,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
!$omp parallel shared(deltah,myidx,a,desc_a)
|
||||
!
|
||||
block
|
||||
integer(psb_ipk_) :: i,j,k,ii,ib,icoeff, ix,iy,iz, ith,nth
|
||||
integer(psb_lpk_) :: glob_row
|
||||
integer(psb_lpk_), allocatable :: irow(:),icol(:)
|
||||
real(psb_spk_), allocatable :: val(:)
|
||||
real(psb_spk_) :: x,y,z, zt(nb)
|
||||
#if defined(OPENMP)
|
||||
nth = omp_get_num_threads()
|
||||
ith = omp_get_thread_num()
|
||||
#else
|
||||
nth = 1
|
||||
ith = 0
|
||||
#endif
|
||||
allocate(val(20*nb),irow(20*nb),&
|
||||
&icol(20*nb),stat=info)
|
||||
if (info /= psb_success_ ) then
|
||||
info=psb_err_alloc_dealloc_
|
||||
call psb_errpush(info,name)
|
||||
!goto 9999
|
||||
endif
|
||||
|
||||
! term depending on (x,y,z)
|
||||
val(icoeff)=(2*sone)*(a1(x,y,z)+a2(x,y,z)+a3(x,y,z))/sqdeltah &
|
||||
& + c(x,y,z)
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
! term depending on (x,y,z+1)
|
||||
val(icoeff)=-a3(x,y,z)/sqdeltah+b3(x,y,z)/deltah2
|
||||
if (iz == idim) then
|
||||
zt(k) = g(x,y,sone)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz+1,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y+1,z)
|
||||
val(icoeff)=-a2(x,y,z)/sqdeltah+b2(x,y,z)/deltah2
|
||||
if (iy == idim) then
|
||||
zt(k) = g(x,sone,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy+1,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x+1,y,z)
|
||||
val(icoeff)=-a1(x,y,z)/sqdeltah+b1(x,y,z)/deltah2
|
||||
if (ix==idim) then
|
||||
zt(k) = g(sone,y,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix+1,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
!$omp do schedule(dynamic)
|
||||
!
|
||||
do ii=1, nlr, nb
|
||||
if (info /= psb_success_) cycle
|
||||
ib = min(nb,nlr-ii+1)
|
||||
icoeff = 1
|
||||
do k=1,ib
|
||||
i=ii+k-1
|
||||
! local matrix pointer
|
||||
glob_row=myidx(i)
|
||||
! compute gridpoint coordinates
|
||||
call idx2ijk(ix,iy,iz,glob_row,idim,idim,idim)
|
||||
! x, y, z coordinates
|
||||
x = (ix-1)*deltah
|
||||
y = (iy-1)*deltah
|
||||
z = (iz-1)*deltah
|
||||
zt(k) = f_(x,y,z)
|
||||
! internal point: build discretization
|
||||
!
|
||||
! term depending on (x-1,y,z)
|
||||
!
|
||||
val(icoeff) = -a1(x,y,z)/sqdeltah-b1(x,y,z)/deltah2
|
||||
if (ix == 1) then
|
||||
zt(k) = g(szero,y,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix-1,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y-1,z)
|
||||
val(icoeff) = -a2(x,y,z)/sqdeltah-b2(x,y,z)/deltah2
|
||||
if (iy == 1) then
|
||||
zt(k) = g(x,szero,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy-1,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y,z-1)
|
||||
val(icoeff)=-a3(x,y,z)/sqdeltah-b3(x,y,z)/deltah2
|
||||
if (iz == 1) then
|
||||
zt(k) = g(x,y,szero)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz-1,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
|
||||
! term depending on (x,y,z)
|
||||
val(icoeff)=(2*sone)*(a1(x,y,z)+a2(x,y,z)+a3(x,y,z))/sqdeltah &
|
||||
& + c(x,y,z)
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
! term depending on (x,y,z+1)
|
||||
val(icoeff)=-a3(x,y,z)/sqdeltah+b3(x,y,z)/deltah2
|
||||
if (iz == idim) then
|
||||
zt(k) = g(x,y,sone)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy,iz+1,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y+1,z)
|
||||
val(icoeff)=-a2(x,y,z)/sqdeltah+b2(x,y,z)/deltah2
|
||||
if (iy == idim) then
|
||||
zt(k) = g(x,sone,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy+1,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x+1,y,z)
|
||||
val(icoeff)=-a1(x,y,z)/sqdeltah+b1(x,y,z)/deltah2
|
||||
if (ix==idim) then
|
||||
zt(k) = g(sone,y,z)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix+1,iy,iz,idim,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
|
||||
end do
|
||||
!write(0,*) ' Outer in_parallel ',omp_in_parallel()
|
||||
call psb_spins(icoeff-1,irow,icol,val,a,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),bv,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
zt(:)=szero
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),xv,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
end do
|
||||
call psb_spins(icoeff-1,irow,icol,val,a,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),bv,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
zt(:)=szero
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),xv,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
deallocate(val,irow,icol)
|
||||
end block
|
||||
!$omp end parallel
|
||||
|
||||
tgen = psb_wtime()-t1
|
||||
if(info /= psb_success_) then
|
||||
@@ -490,7 +500,6 @@ contains
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
deallocate(val,irow,icol)
|
||||
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
@@ -557,6 +566,9 @@ contains
|
||||
& a1,a2,b1,b2,c,g,info,f,amold,vmold,partition, nrl,iv)
|
||||
use psb_base_mod
|
||||
use psb_util_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
!
|
||||
! Discretizes the partial differential equation
|
||||
!
|
||||
@@ -591,7 +603,6 @@ contains
|
||||
type(psb_s_csc_sparse_mat) :: acsc
|
||||
type(psb_s_coo_sparse_mat) :: acoo
|
||||
type(psb_s_csr_sparse_mat) :: acsr
|
||||
real(psb_spk_) :: zt(nb),x,y,z,xph,xmh,yph,ymh,zph,zmh
|
||||
integer(psb_ipk_) :: nnz,nr,nlr,i,j,ii,ib,k, partition_
|
||||
integer(psb_lpk_) :: m,n,glob_row,nt
|
||||
integer(psb_ipk_) :: ix,iy,iz,ia,indx_owner
|
||||
@@ -604,8 +615,7 @@ contains
|
||||
! Process grid
|
||||
integer(psb_ipk_) :: np, iam
|
||||
integer(psb_ipk_) :: icoeff
|
||||
integer(psb_lpk_), allocatable :: irow(:),icol(:),myidx(:)
|
||||
real(psb_spk_), allocatable :: val(:)
|
||||
integer(psb_lpk_), allocatable :: myidx(:)
|
||||
! deltah dimension of each grid cell
|
||||
! deltat discretization time
|
||||
real(psb_spk_) :: deltah, sqdeltah, deltah2, dd
|
||||
@@ -791,7 +801,7 @@ contains
|
||||
!write(0,*) iam,' Check on neighbours: ',desc_a%get_p_adjcncy()
|
||||
end if
|
||||
end block
|
||||
|
||||
|
||||
case default
|
||||
write(psb_err_unit,*) iam, 'Initialization error: should not get here'
|
||||
info = -1
|
||||
@@ -816,93 +826,109 @@ contains
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
! we build an auxiliary matrix consisting of one row at a
|
||||
! time; just a small matrix. might be extended to generate
|
||||
! a bunch of rows per call.
|
||||
!
|
||||
allocate(val(20*nb),irow(20*nb),&
|
||||
&icol(20*nb),stat=info)
|
||||
if (info /= psb_success_ ) then
|
||||
info=psb_err_alloc_dealloc_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
endif
|
||||
|
||||
|
||||
! loop over rows belonging to current process in a block
|
||||
! distribution.
|
||||
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
do ii=1, nlr,nb
|
||||
ib = min(nb,nlr-ii+1)
|
||||
icoeff = 1
|
||||
do k=1,ib
|
||||
i=ii+k-1
|
||||
! local matrix pointer
|
||||
glob_row=myidx(i)
|
||||
! compute gridpoint coordinates
|
||||
call idx2ijk(ix,iy,glob_row,idim,idim)
|
||||
! x, y coordinates
|
||||
x = (ix-1)*deltah
|
||||
y = (iy-1)*deltah
|
||||
!$omp parallel shared(deltah,myidx,a,desc_a)
|
||||
!
|
||||
block
|
||||
integer(psb_ipk_) :: i,j,k,ii,ib,icoeff, ix,iy,iz, ith,nth
|
||||
integer(psb_lpk_) :: glob_row
|
||||
integer(psb_lpk_), allocatable :: irow(:),icol(:)
|
||||
real(psb_spk_), allocatable :: val(:)
|
||||
real(psb_spk_) :: x,y,z, zt(nb)
|
||||
#if defined(OPENMP)
|
||||
nth = omp_get_num_threads()
|
||||
ith = omp_get_thread_num()
|
||||
#else
|
||||
nth = 1
|
||||
ith = 0
|
||||
#endif
|
||||
allocate(val(20*nb),irow(20*nb),&
|
||||
&icol(20*nb),stat=info)
|
||||
if (info /= psb_success_ ) then
|
||||
info=psb_err_alloc_dealloc_
|
||||
call psb_errpush(info,name)
|
||||
!goto 9999
|
||||
endif
|
||||
|
||||
zt(k) = f_(x,y)
|
||||
! internal point: build discretization
|
||||
!
|
||||
! term depending on (x-1,y)
|
||||
!
|
||||
val(icoeff) = -a1(x,y)/sqdeltah-b1(x,y)/deltah2
|
||||
if (ix == 1) then
|
||||
zt(k) = g(szero,y)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix-1,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y-1)
|
||||
val(icoeff) = -a2(x,y)/sqdeltah-b2(x,y)/deltah2
|
||||
if (iy == 1) then
|
||||
zt(k) = g(x,szero)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy-1,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! loop over rows belonging to current process in a block
|
||||
! distribution.
|
||||
!$omp do schedule(dynamic)
|
||||
!
|
||||
do ii=1, nlr,nb
|
||||
ib = min(nb,nlr-ii+1)
|
||||
icoeff = 1
|
||||
do k=1,ib
|
||||
i=ii+k-1
|
||||
! local matrix pointer
|
||||
glob_row=myidx(i)
|
||||
! compute gridpoint coordinates
|
||||
call idx2ijk(ix,iy,glob_row,idim,idim)
|
||||
! x, y coordinates
|
||||
x = (ix-1)*deltah
|
||||
y = (iy-1)*deltah
|
||||
|
||||
! term depending on (x,y)
|
||||
val(icoeff)=(2*sone)*(a1(x,y) + a2(x,y))/sqdeltah + c(x,y)
|
||||
call ijk2idx(icol(icoeff),ix,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
! term depending on (x,y+1)
|
||||
val(icoeff)=-a2(x,y)/sqdeltah+b2(x,y)/deltah2
|
||||
if (iy == idim) then
|
||||
zt(k) = g(x,sone)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy+1,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x+1,y)
|
||||
val(icoeff)=-a1(x,y)/sqdeltah+b1(x,y)/deltah2
|
||||
if (ix==idim) then
|
||||
zt(k) = g(sone,y)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix+1,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
zt(k) = f_(x,y)
|
||||
! internal point: build discretization
|
||||
!
|
||||
! term depending on (x-1,y)
|
||||
!
|
||||
val(icoeff) = -a1(x,y)/sqdeltah-b1(x,y)/deltah2
|
||||
if (ix == 1) then
|
||||
zt(k) = g(szero,y)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix-1,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x,y-1)
|
||||
val(icoeff) = -a2(x,y)/sqdeltah-b2(x,y)/deltah2
|
||||
if (iy == 1) then
|
||||
zt(k) = g(x,szero)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy-1,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
|
||||
! term depending on (x,y)
|
||||
val(icoeff)=(2*sone)*(a1(x,y) + a2(x,y))/sqdeltah + c(x,y)
|
||||
call ijk2idx(icol(icoeff),ix,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
! term depending on (x,y+1)
|
||||
val(icoeff)=-a2(x,y)/sqdeltah+b2(x,y)/deltah2
|
||||
if (iy == idim) then
|
||||
zt(k) = g(x,sone)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix,iy+1,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
! term depending on (x+1,y)
|
||||
val(icoeff)=-a1(x,y)/sqdeltah+b1(x,y)/deltah2
|
||||
if (ix==idim) then
|
||||
zt(k) = g(sone,y)*(-val(icoeff)) + zt(k)
|
||||
else
|
||||
call ijk2idx(icol(icoeff),ix+1,iy,idim,idim)
|
||||
irow(icoeff) = glob_row
|
||||
icoeff = icoeff+1
|
||||
endif
|
||||
|
||||
end do
|
||||
call psb_spins(icoeff-1,irow,icol,val,a,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),bv,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
zt(:)=szero
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),xv,desc_a,info)
|
||||
if(info /= psb_success_) cycle
|
||||
end do
|
||||
call psb_spins(icoeff-1,irow,icol,val,a,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),bv,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
zt(:)=szero
|
||||
call psb_geins(ib,myidx(ii:ii+ib-1),zt(1:ib),xv,desc_a,info)
|
||||
if(info /= psb_success_) exit
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
deallocate(val,irow,icol)
|
||||
end block
|
||||
!$omp end parallel
|
||||
|
||||
tgen = psb_wtime()-t1
|
||||
if(info /= psb_success_) then
|
||||
@@ -912,8 +938,6 @@ contains
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
deallocate(val,irow,icol)
|
||||
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
call psb_cdasb(desc_a,info)
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
%%%%%%%%%%% General arguments % Lines starting with % are ignored.
|
||||
CSR ! Storage format CSR COO JAD
|
||||
0080 ! IDIM; domain size. Linear system size is IDIM**3
|
||||
0200 ! IDIM; domain size. Linear system size is IDIM**3
|
||||
CONST ! PDECOEFF: CONST, EXP, GAUSS Coefficients of the PDE
|
||||
BICGSTAB ! Iterative method: BiCGSTAB BiCGSTABL BiCG CG CGS FCG GCR RGMRES
|
||||
2 ! ISTOPC
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
%%%%%%%%%%% General arguments % Lines starting with % are ignored.
|
||||
CSR ! Storage format CSR COO JAD
|
||||
0140 ! IDIM; domain size. Linear system size is IDIM**3
|
||||
CONST ! PDECOEFF: CONST, EXP, GAUSS Coefficients of the PDE
|
||||
BICGSTAB ! Iterative method: BiCGSTAB BiCGSTABL BiCG CG CGS FCG GCR RGMRES
|
||||
2 ! ISTOPC
|
||||
00500 ! ITMAX
|
||||
1 ! ITRACE
|
||||
30 ! IRST (restart for RGMRES and BiCGSTABL)
|
||||
1.d-6 ! EPS
|
||||
%%%%%%%%%%% Main preconditioner choices %%%%%%%%%%%%%%%%
|
||||
ML-VCYCLE-BJAC-D-BJAC ! Longer descriptive name for preconditioner (up to 20 chars)
|
||||
ML ! Preconditioner type: NONE JACOBI GS FBGS BJAC AS ML
|
||||
%%%%%%%%%%% First smoother (for all levels but coarsest) %%%%%%%%%%%%%%%%
|
||||
FBGS ! Smoother type JACOBI FBGS GS BWGS BJAC AS. For 1-level, repeats previous.
|
||||
1 ! Number of sweeps for smoother
|
||||
0 ! Number of overlap layers for AS preconditioner
|
||||
HALO ! AS restriction operator: NONE HALO
|
||||
NONE ! AS prolongation operator: NONE SUM AVG
|
||||
INVK ! Subdomain solver for BJAC/AS: JACOBI GS BGS ILU ILUT MILU MUMPS SLU UMF
|
||||
LLK ! AINV variant
|
||||
0 ! Fill level P for ILU(P) and ILU(T,P)
|
||||
1 ! Inverse Fill level P for INVK
|
||||
1.d-4 ! Threshold T for ILU(T,P)
|
||||
%%%%%%%%%%% Second smoother, always ignored for non-ML %%%%%%%%%%%%%%%%
|
||||
NONE ! Second (post) smoother, ignored if NONE
|
||||
1 ! Number of sweeps for (post) smoother
|
||||
0 ! Number of overlap layers for AS preconditioner
|
||||
HALO ! AS restriction operator: NONE HALO
|
||||
NONE ! AS prolongation operator: NONE SUM AVG
|
||||
ILU ! Subdomain solver for BJAC/AS: JACOBI GS BGS ILU ILUT MILU MUMPS SLU UMF
|
||||
LLK ! AINV variant
|
||||
0 ! Fill level P for ILU(P) and ILU(T,P)
|
||||
8 ! Inverse Fill level P for INVK
|
||||
1.d-4 ! Threshold T for ILU(T,P)
|
||||
%%%%%%%%%%% Multilevel parameters %%%%%%%%%%%%%%%%
|
||||
VCYCLE ! Type of multilevel CYCLE: VCYCLE WCYCLE KCYCLE MULT ADD
|
||||
1 ! Number of outer sweeps for ML
|
||||
-3 ! Max Number of levels in a multilevel preconditioner; if <0, lib default
|
||||
-3 ! Target coarse matrix size per process; if <0, lib default
|
||||
SMOOTHED ! Type of aggregation: SMOOTHED UNSMOOTHED
|
||||
NEWMTC ! Parallel aggregation: DEC, SYMDEC, COUPLED NEWMTC
|
||||
NEWMTC ! aggregation measure SOC1, MATCHBOXP NEWMTC
|
||||
8 ! Requested size of the aggregates for MATCHBOXP
|
||||
NATURAL ! Ordering of aggregation NATURAL DEGREE
|
||||
NOFILTER ! Filtering of matrix: FILTER NOFILTER
|
||||
-1.5 ! Coarsening ratio, if < 0 use library default
|
||||
2 ! MATCHING variant
|
||||
8.0 ! LAMBDA
|
||||
-2 ! Number of thresholds in vector, next line ignored if <= 0
|
||||
0.05 0.025 ! Thresholds
|
||||
-0.0100d0 ! Smoothed aggregation threshold, ignored if < 0
|
||||
%%%%%%%%%%% Coarse level solver %%%%%%%%%%%%%%%%
|
||||
BJAC ! Coarsest-level solver: MUMPS UMF SLU SLUDIST JACOBI GS BJAC
|
||||
ILU ! Coarsest-level subsolver for BJAC: ILU ILUT MILU UMF MUMPS SLU
|
||||
DIST ! Coarsest-level matrix distribution: DIST REPL
|
||||
1 ! Coarsest-level fillin P for ILU(P) and ILU(T,P)
|
||||
1.d-4 ! Coarsest-level threshold T for ILU(T,P)
|
||||
1 ! Number of sweeps for JACOBI/GS/BJAC coarsest-level solver
|
||||
%%%%%%%%%%% Dump parms %%%%%%%%%%%%%%%%%%%%%%%%%%
|
||||
F ! Dump preconditioner on file
|
||||
1 ! Min level
|
||||
20 ! Max level
|
||||
T ! Dump AC
|
||||
T ! Dump RP
|
||||
F ! Dump TPROL
|
||||
F ! Dump SMOOTHER
|
||||
F ! Dump SOLVER
|
||||
F ! Global numering ?
|
||||
Reference in New Issue
Block a user