mirror of
https://github.com/sfilippone/amg4psblas.git
synced 2026-10-07 15:15:07 +00:00
Compare commits
13
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
24c85c7114 | ||
|
|
53998a1da9 | ||
|
|
11421f53a2 | ||
|
|
d33bcfe107 | ||
|
|
5bcd36f394 | ||
|
|
73495edf09 | ||
|
|
9e82d2e311 | ||
|
|
c1ecb4ebec | ||
|
|
e78449d0f5 | ||
|
|
e3de565b6d | ||
|
|
7b9c722a1a | ||
|
|
2fd718be6f | ||
|
|
3a5e73e4c8 |
@@ -189,6 +189,7 @@ module amg_c_onelev_mod
|
||||
procedure, pass(lv) :: descr => amg_c_base_onelev_descr
|
||||
procedure, pass(lv) :: default => c_base_onelev_default
|
||||
procedure, pass(lv) :: free => amg_c_base_onelev_free
|
||||
procedure, pass(lv) :: free_smoothers => amg_c_base_onelev_free_smoothers
|
||||
procedure, pass(lv) :: nullify => c_base_onelev_nullify
|
||||
procedure, pass(lv) :: check => amg_c_base_onelev_check
|
||||
procedure, pass(lv) :: dump => amg_c_base_onelev_dump
|
||||
@@ -285,7 +286,7 @@ module amg_c_onelev_mod
|
||||
end subroutine amg_c_base_onelev_cnv
|
||||
end interface
|
||||
|
||||
interface
|
||||
interface
|
||||
subroutine amg_c_base_onelev_free(lv,info)
|
||||
import :: psb_cspmat_type, psb_c_vect_type, psb_c_base_vect_type, &
|
||||
& psb_clinmap_type, psb_spk_, amg_c_onelev_type, &
|
||||
@@ -297,6 +298,18 @@ interface
|
||||
end subroutine amg_c_base_onelev_free
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_c_base_onelev_free_smoothers(lv,info)
|
||||
import :: psb_cspmat_type, psb_c_vect_type, psb_c_base_vect_type, &
|
||||
& psb_clinmap_type, psb_spk_, amg_c_onelev_type, &
|
||||
& psb_ipk_, psb_epk_, psb_desc_type
|
||||
implicit none
|
||||
|
||||
class(amg_c_onelev_type), intent(inout) :: lv
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_c_base_onelev_free_smoothers
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_c_base_onelev_check(lv,info)
|
||||
import :: psb_cspmat_type, psb_c_vect_type, psb_c_base_vect_type, &
|
||||
|
||||
@@ -135,7 +135,9 @@ module amg_c_prec_type
|
||||
procedure, pass(prec) :: build => amg_cprecbld
|
||||
procedure, pass(prec) :: hierarchy_build => amg_c_hierarchy_bld
|
||||
procedure, pass(prec) :: hierarchy_rebuild => amg_c_hierarchy_rebld
|
||||
procedure, pass(prec) :: hierarchy_free => amg_c_hierarchy_free
|
||||
procedure, pass(prec) :: smoothers_build => amg_c_smoothers_bld
|
||||
procedure, pass(prec) :: smoothers_free => amg_c_smoothers_free
|
||||
procedure, pass(prec) :: descr => amg_cfile_prec_descr
|
||||
end type amg_cprec_type
|
||||
|
||||
@@ -345,6 +347,14 @@ module amg_c_prec_type
|
||||
end subroutine amg_c_smoothers_bld
|
||||
end interface amg_smoothers_bld
|
||||
|
||||
interface amg_smoothers_free
|
||||
module procedure amg_c_smoothers_free
|
||||
end interface amg_smoothers_free
|
||||
|
||||
interface amg_hierarchy_free
|
||||
module procedure amg_c_hierarchy_free
|
||||
end interface amg_hierarchy_free
|
||||
|
||||
contains
|
||||
!
|
||||
! Function returning a pointer to the smoother
|
||||
@@ -618,6 +628,68 @@ contains
|
||||
|
||||
end subroutine amg_c_prec_free
|
||||
|
||||
subroutine amg_c_smoothers_free(prec,info)
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
class(amg_cprec_type), intent(inout) :: prec
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: me,err_act,i
|
||||
character(len=20) :: name
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_c_smoothers_free'
|
||||
call psb_erractionsave(err_act)
|
||||
if (psb_errstatus_fatal()) then
|
||||
info = psb_err_internal_error_; goto 9999
|
||||
end if
|
||||
|
||||
if (allocated(prec%precv)) then
|
||||
do i=1,size(prec%precv)
|
||||
call prec%precv(i)%free_smoothers(info)
|
||||
end do
|
||||
end if
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_c_smoothers_free
|
||||
|
||||
subroutine amg_c_hierarchy_free(prec,info)
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
class(amg_cprec_type), intent(inout) :: prec
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: me,err_act,i
|
||||
character(len=20) :: name
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_c_hierarchy_free'
|
||||
call psb_erractionsave(err_act)
|
||||
if (psb_errstatus_fatal()) then
|
||||
info = psb_err_internal_error_; goto 9999
|
||||
end if
|
||||
|
||||
me=-1
|
||||
write(0,*) 'Missing implementation '
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_c_hierarchy_free
|
||||
|
||||
|
||||
!
|
||||
|
||||
@@ -190,6 +190,7 @@ module amg_d_onelev_mod
|
||||
procedure, pass(lv) :: descr => amg_d_base_onelev_descr
|
||||
procedure, pass(lv) :: default => d_base_onelev_default
|
||||
procedure, pass(lv) :: free => amg_d_base_onelev_free
|
||||
procedure, pass(lv) :: free_smoothers => amg_d_base_onelev_free_smoothers
|
||||
procedure, pass(lv) :: nullify => d_base_onelev_nullify
|
||||
procedure, pass(lv) :: check => amg_d_base_onelev_check
|
||||
procedure, pass(lv) :: dump => amg_d_base_onelev_dump
|
||||
@@ -286,7 +287,7 @@ module amg_d_onelev_mod
|
||||
end subroutine amg_d_base_onelev_cnv
|
||||
end interface
|
||||
|
||||
interface
|
||||
interface
|
||||
subroutine amg_d_base_onelev_free(lv,info)
|
||||
import :: psb_dspmat_type, psb_d_vect_type, psb_d_base_vect_type, &
|
||||
& psb_dlinmap_type, psb_dpk_, amg_d_onelev_type, &
|
||||
@@ -298,6 +299,18 @@ interface
|
||||
end subroutine amg_d_base_onelev_free
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_base_onelev_free_smoothers(lv,info)
|
||||
import :: psb_dspmat_type, psb_d_vect_type, psb_d_base_vect_type, &
|
||||
& psb_dlinmap_type, psb_dpk_, amg_d_onelev_type, &
|
||||
& psb_ipk_, psb_epk_, psb_desc_type
|
||||
implicit none
|
||||
|
||||
class(amg_d_onelev_type), intent(inout) :: lv
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_base_onelev_free_smoothers
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_base_onelev_check(lv,info)
|
||||
import :: psb_dspmat_type, psb_d_vect_type, psb_d_base_vect_type, &
|
||||
|
||||
@@ -135,7 +135,9 @@ module amg_d_prec_type
|
||||
procedure, pass(prec) :: build => amg_dprecbld
|
||||
procedure, pass(prec) :: hierarchy_build => amg_d_hierarchy_bld
|
||||
procedure, pass(prec) :: hierarchy_rebuild => amg_d_hierarchy_rebld
|
||||
procedure, pass(prec) :: hierarchy_free => amg_d_hierarchy_free
|
||||
procedure, pass(prec) :: smoothers_build => amg_d_smoothers_bld
|
||||
procedure, pass(prec) :: smoothers_free => amg_d_smoothers_free
|
||||
procedure, pass(prec) :: descr => amg_dfile_prec_descr
|
||||
end type amg_dprec_type
|
||||
|
||||
@@ -345,6 +347,14 @@ module amg_d_prec_type
|
||||
end subroutine amg_d_smoothers_bld
|
||||
end interface amg_smoothers_bld
|
||||
|
||||
interface amg_smoothers_free
|
||||
module procedure amg_d_smoothers_free
|
||||
end interface amg_smoothers_free
|
||||
|
||||
interface amg_hierarchy_free
|
||||
module procedure amg_d_hierarchy_free
|
||||
end interface amg_hierarchy_free
|
||||
|
||||
contains
|
||||
!
|
||||
! Function returning a pointer to the smoother
|
||||
@@ -618,6 +628,68 @@ contains
|
||||
|
||||
end subroutine amg_d_prec_free
|
||||
|
||||
subroutine amg_d_smoothers_free(prec,info)
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
class(amg_dprec_type), intent(inout) :: prec
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: me,err_act,i
|
||||
character(len=20) :: name
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_d_smoothers_free'
|
||||
call psb_erractionsave(err_act)
|
||||
if (psb_errstatus_fatal()) then
|
||||
info = psb_err_internal_error_; goto 9999
|
||||
end if
|
||||
|
||||
if (allocated(prec%precv)) then
|
||||
do i=1,size(prec%precv)
|
||||
call prec%precv(i)%free_smoothers(info)
|
||||
end do
|
||||
end if
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_d_smoothers_free
|
||||
|
||||
subroutine amg_d_hierarchy_free(prec,info)
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
class(amg_dprec_type), intent(inout) :: prec
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: me,err_act,i
|
||||
character(len=20) :: name
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_d_hierarchy_free'
|
||||
call psb_erractionsave(err_act)
|
||||
if (psb_errstatus_fatal()) then
|
||||
info = psb_err_internal_error_; goto 9999
|
||||
end if
|
||||
|
||||
me=-1
|
||||
write(0,*) 'Missing implementation '
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_d_hierarchy_free
|
||||
|
||||
|
||||
!
|
||||
|
||||
@@ -190,6 +190,7 @@ module amg_s_onelev_mod
|
||||
procedure, pass(lv) :: descr => amg_s_base_onelev_descr
|
||||
procedure, pass(lv) :: default => s_base_onelev_default
|
||||
procedure, pass(lv) :: free => amg_s_base_onelev_free
|
||||
procedure, pass(lv) :: free_smoothers => amg_s_base_onelev_free_smoothers
|
||||
procedure, pass(lv) :: nullify => s_base_onelev_nullify
|
||||
procedure, pass(lv) :: check => amg_s_base_onelev_check
|
||||
procedure, pass(lv) :: dump => amg_s_base_onelev_dump
|
||||
@@ -286,7 +287,7 @@ module amg_s_onelev_mod
|
||||
end subroutine amg_s_base_onelev_cnv
|
||||
end interface
|
||||
|
||||
interface
|
||||
interface
|
||||
subroutine amg_s_base_onelev_free(lv,info)
|
||||
import :: psb_sspmat_type, psb_s_vect_type, psb_s_base_vect_type, &
|
||||
& psb_slinmap_type, psb_spk_, amg_s_onelev_type, &
|
||||
@@ -298,6 +299,18 @@ interface
|
||||
end subroutine amg_s_base_onelev_free
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_s_base_onelev_free_smoothers(lv,info)
|
||||
import :: psb_sspmat_type, psb_s_vect_type, psb_s_base_vect_type, &
|
||||
& psb_slinmap_type, psb_spk_, amg_s_onelev_type, &
|
||||
& psb_ipk_, psb_epk_, psb_desc_type
|
||||
implicit none
|
||||
|
||||
class(amg_s_onelev_type), intent(inout) :: lv
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_s_base_onelev_free_smoothers
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_s_base_onelev_check(lv,info)
|
||||
import :: psb_sspmat_type, psb_s_vect_type, psb_s_base_vect_type, &
|
||||
|
||||
@@ -135,7 +135,9 @@ module amg_s_prec_type
|
||||
procedure, pass(prec) :: build => amg_sprecbld
|
||||
procedure, pass(prec) :: hierarchy_build => amg_s_hierarchy_bld
|
||||
procedure, pass(prec) :: hierarchy_rebuild => amg_s_hierarchy_rebld
|
||||
procedure, pass(prec) :: hierarchy_free => amg_s_hierarchy_free
|
||||
procedure, pass(prec) :: smoothers_build => amg_s_smoothers_bld
|
||||
procedure, pass(prec) :: smoothers_free => amg_s_smoothers_free
|
||||
procedure, pass(prec) :: descr => amg_sfile_prec_descr
|
||||
end type amg_sprec_type
|
||||
|
||||
@@ -345,6 +347,14 @@ module amg_s_prec_type
|
||||
end subroutine amg_s_smoothers_bld
|
||||
end interface amg_smoothers_bld
|
||||
|
||||
interface amg_smoothers_free
|
||||
module procedure amg_s_smoothers_free
|
||||
end interface amg_smoothers_free
|
||||
|
||||
interface amg_hierarchy_free
|
||||
module procedure amg_s_hierarchy_free
|
||||
end interface amg_hierarchy_free
|
||||
|
||||
contains
|
||||
!
|
||||
! Function returning a pointer to the smoother
|
||||
@@ -618,6 +628,68 @@ contains
|
||||
|
||||
end subroutine amg_s_prec_free
|
||||
|
||||
subroutine amg_s_smoothers_free(prec,info)
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
class(amg_sprec_type), intent(inout) :: prec
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: me,err_act,i
|
||||
character(len=20) :: name
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_s_smoothers_free'
|
||||
call psb_erractionsave(err_act)
|
||||
if (psb_errstatus_fatal()) then
|
||||
info = psb_err_internal_error_; goto 9999
|
||||
end if
|
||||
|
||||
if (allocated(prec%precv)) then
|
||||
do i=1,size(prec%precv)
|
||||
call prec%precv(i)%free_smoothers(info)
|
||||
end do
|
||||
end if
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_s_smoothers_free
|
||||
|
||||
subroutine amg_s_hierarchy_free(prec,info)
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
class(amg_sprec_type), intent(inout) :: prec
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: me,err_act,i
|
||||
character(len=20) :: name
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_s_hierarchy_free'
|
||||
call psb_erractionsave(err_act)
|
||||
if (psb_errstatus_fatal()) then
|
||||
info = psb_err_internal_error_; goto 9999
|
||||
end if
|
||||
|
||||
me=-1
|
||||
write(0,*) 'Missing implementation '
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_s_hierarchy_free
|
||||
|
||||
|
||||
!
|
||||
|
||||
@@ -189,6 +189,7 @@ module amg_z_onelev_mod
|
||||
procedure, pass(lv) :: descr => amg_z_base_onelev_descr
|
||||
procedure, pass(lv) :: default => z_base_onelev_default
|
||||
procedure, pass(lv) :: free => amg_z_base_onelev_free
|
||||
procedure, pass(lv) :: free_smoothers => amg_z_base_onelev_free_smoothers
|
||||
procedure, pass(lv) :: nullify => z_base_onelev_nullify
|
||||
procedure, pass(lv) :: check => amg_z_base_onelev_check
|
||||
procedure, pass(lv) :: dump => amg_z_base_onelev_dump
|
||||
@@ -285,7 +286,7 @@ module amg_z_onelev_mod
|
||||
end subroutine amg_z_base_onelev_cnv
|
||||
end interface
|
||||
|
||||
interface
|
||||
interface
|
||||
subroutine amg_z_base_onelev_free(lv,info)
|
||||
import :: psb_zspmat_type, psb_z_vect_type, psb_z_base_vect_type, &
|
||||
& psb_zlinmap_type, psb_dpk_, amg_z_onelev_type, &
|
||||
@@ -297,6 +298,18 @@ interface
|
||||
end subroutine amg_z_base_onelev_free
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_z_base_onelev_free_smoothers(lv,info)
|
||||
import :: psb_zspmat_type, psb_z_vect_type, psb_z_base_vect_type, &
|
||||
& psb_zlinmap_type, psb_dpk_, amg_z_onelev_type, &
|
||||
& psb_ipk_, psb_epk_, psb_desc_type
|
||||
implicit none
|
||||
|
||||
class(amg_z_onelev_type), intent(inout) :: lv
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_z_base_onelev_free_smoothers
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_z_base_onelev_check(lv,info)
|
||||
import :: psb_zspmat_type, psb_z_vect_type, psb_z_base_vect_type, &
|
||||
|
||||
@@ -135,7 +135,9 @@ module amg_z_prec_type
|
||||
procedure, pass(prec) :: build => amg_zprecbld
|
||||
procedure, pass(prec) :: hierarchy_build => amg_z_hierarchy_bld
|
||||
procedure, pass(prec) :: hierarchy_rebuild => amg_z_hierarchy_rebld
|
||||
procedure, pass(prec) :: hierarchy_free => amg_z_hierarchy_free
|
||||
procedure, pass(prec) :: smoothers_build => amg_z_smoothers_bld
|
||||
procedure, pass(prec) :: smoothers_free => amg_z_smoothers_free
|
||||
procedure, pass(prec) :: descr => amg_zfile_prec_descr
|
||||
end type amg_zprec_type
|
||||
|
||||
@@ -345,6 +347,14 @@ module amg_z_prec_type
|
||||
end subroutine amg_z_smoothers_bld
|
||||
end interface amg_smoothers_bld
|
||||
|
||||
interface amg_smoothers_free
|
||||
module procedure amg_z_smoothers_free
|
||||
end interface amg_smoothers_free
|
||||
|
||||
interface amg_hierarchy_free
|
||||
module procedure amg_z_hierarchy_free
|
||||
end interface amg_hierarchy_free
|
||||
|
||||
contains
|
||||
!
|
||||
! Function returning a pointer to the smoother
|
||||
@@ -618,6 +628,68 @@ contains
|
||||
|
||||
end subroutine amg_z_prec_free
|
||||
|
||||
subroutine amg_z_smoothers_free(prec,info)
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
class(amg_zprec_type), intent(inout) :: prec
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: me,err_act,i
|
||||
character(len=20) :: name
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_z_smoothers_free'
|
||||
call psb_erractionsave(err_act)
|
||||
if (psb_errstatus_fatal()) then
|
||||
info = psb_err_internal_error_; goto 9999
|
||||
end if
|
||||
|
||||
if (allocated(prec%precv)) then
|
||||
do i=1,size(prec%precv)
|
||||
call prec%precv(i)%free_smoothers(info)
|
||||
end do
|
||||
end if
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_z_smoothers_free
|
||||
|
||||
subroutine amg_z_hierarchy_free(prec,info)
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
class(amg_zprec_type), intent(inout) :: prec
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: me,err_act,i
|
||||
character(len=20) :: name
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_z_hierarchy_free'
|
||||
call psb_erractionsave(err_act)
|
||||
if (psb_errstatus_fatal()) then
|
||||
info = psb_err_internal_error_; goto 9999
|
||||
end if
|
||||
|
||||
me=-1
|
||||
write(0,*) 'Missing implementation '
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
9999 call psb_error_handler(err_act)
|
||||
return
|
||||
|
||||
end subroutine amg_z_hierarchy_free
|
||||
|
||||
|
||||
!
|
||||
|
||||
@@ -76,7 +76,7 @@ subroutine amg_c_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
integer(psb_ipk_) :: nrow, ncol, nrl, nzl, ip, nzt, i, k
|
||||
integer(psb_lpk_) :: nrsave, ncsave, nzsave, nza
|
||||
logical, parameter :: do_timings=.false., oldstyle=.false., debug=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_cpytrans1=-1, idx_cpytrans2=-1
|
||||
|
||||
name='amg_ptap_bld'
|
||||
if(psb_get_errstatus().ne.0) return
|
||||
@@ -93,7 +93,11 @@ subroutine amg_c_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("SPMM_BLD: par_spspmm")
|
||||
& idx_spspmm = psb_get_timer_idx("PTAP_BLD: par_spspmm")
|
||||
if ((do_timings).and.(idx_cpytrans1==-1)) &
|
||||
& idx_cpytrans1 = psb_get_timer_idx("PTAP_BLD: cpy&trans1")
|
||||
if ((do_timings).and.(idx_cpytrans2==-1)) &
|
||||
& idx_cpytrans2 = psb_get_timer_idx("PTAP_BLD: cpy&trans2")
|
||||
|
||||
naggr = nlaggr(me+1)
|
||||
ntaggr = sum(nlaggr)
|
||||
@@ -128,6 +132,7 @@ subroutine amg_c_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
! Ok first product done.
|
||||
|
||||
if (present(desc_ax)) then
|
||||
if (do_timings) call psb_tic(idx_cpytrans1)
|
||||
block
|
||||
call coo_prol%cp_to_coo(coo_restr,info)
|
||||
call coo_restr%set_ncols(desc_ac%get_local_cols())
|
||||
@@ -137,7 +142,7 @@ subroutine amg_c_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
call coo_restr%set_ncols(desc_ax%get_local_cols())
|
||||
end block
|
||||
call csr_restr%cp_from_coo(coo_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_cpytrans1)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spcnv coo_restr')
|
||||
goto 9999
|
||||
@@ -167,27 +172,28 @@ subroutine amg_c_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call coo_restr%transp()
|
||||
nzl = coo_restr%get_nzeros()
|
||||
nrl = desc_ac%get_local_rows()
|
||||
i=0
|
||||
nrl = desc_ac%get_local_rows()
|
||||
call coo_restr%fix(info)
|
||||
i=coo_restr%get_nzeros()
|
||||
!
|
||||
! Only keep local rows
|
||||
!
|
||||
do k=1, nzl
|
||||
if ((1 <= coo_restr%ia(k)) .and.(coo_restr%ia(k) <= nrl)) then
|
||||
i = i+1
|
||||
coo_restr%val(i) = coo_restr%val(k)
|
||||
coo_restr%ia(i) = coo_restr%ia(k)
|
||||
coo_restr%ja(i) = coo_restr%ja(k)
|
||||
search: do k=i,1,-1
|
||||
if (coo_restr%ia(k) <= nrl) then
|
||||
call coo_restr%set_nzeros(k)
|
||||
exit search
|
||||
end if
|
||||
end do
|
||||
call coo_restr%set_nzeros(i)
|
||||
call coo_restr%fix(info)
|
||||
end do search
|
||||
|
||||
nzl = coo_restr%get_nzeros()
|
||||
call coo_restr%set_nrows(desc_ac%get_local_rows())
|
||||
call coo_restr%set_ncols(desc_a%get_local_cols())
|
||||
if (debug) call check_coo(me,trim(name)//' Check 2 on coo_restr:',coo_restr)
|
||||
if (do_timings) call psb_tic(idx_cpytrans2)
|
||||
|
||||
call csr_restr%cp_from_coo(coo_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_cpytrans2)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spcnv coo_restr')
|
||||
goto 9999
|
||||
|
||||
+218
-21
@@ -72,7 +72,9 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_c_inner_mod
|
||||
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
@@ -85,7 +87,7 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_), allocatable :: ils(:), neigh(:), irow(:), icol(:),&
|
||||
integer(psb_ipk_), allocatable :: neigh(:), irow(:), icol(:),&
|
||||
& ideg(:), idxs(:)
|
||||
integer(psb_lpk_), allocatable :: tmpaggr(:)
|
||||
complex(psb_spk_), allocatable :: val(:), diag(:)
|
||||
@@ -99,6 +101,9 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_) :: nrow, ncol, n_ne
|
||||
integer(psb_lpk_) :: nrglob
|
||||
character(len=20) :: name, ch_err
|
||||
integer(psb_ipk_), save :: idx_soc1_p1=-1, idx_soc1_p2=-1, idx_soc1_p3=-1
|
||||
integer(psb_ipk_), save :: idx_soc1_p0=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_soc1_map_bld'
|
||||
@@ -114,6 +119,14 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
nrglob = desc_a%get_global_rows()
|
||||
if ((do_timings).and.(idx_soc1_p0==-1)) &
|
||||
& idx_soc1_p0 = psb_get_timer_idx("SOC1_MAP: phase0")
|
||||
if ((do_timings).and.(idx_soc1_p1==-1)) &
|
||||
& idx_soc1_p1 = psb_get_timer_idx("SOC1_MAP: phase1")
|
||||
if ((do_timings).and.(idx_soc1_p2==-1)) &
|
||||
& idx_soc1_p2 = psb_get_timer_idx("SOC1_MAP: phase2")
|
||||
if ((do_timings).and.(idx_soc1_p3==-1)) &
|
||||
& idx_soc1_p3 = psb_get_timer_idx("SOC1_MAP: phase3")
|
||||
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -133,41 +146,204 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p0)
|
||||
call a%cp_to(acsr)
|
||||
if (do_timings) call psb_toc(idx_soc1_p0)
|
||||
if (clean_zeros) call acsr%clean_zeros(info)
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
!$omp parallel do private(i) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
idxs(i) = i
|
||||
idxs(i) = i
|
||||
end do
|
||||
else
|
||||
!$omp end parallel do
|
||||
else
|
||||
!$omp parallel do private(i) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
ideg(i) = acsr%irp(i+1) - acsr%irp(i)
|
||||
end do
|
||||
!$omp end parallel do
|
||||
call psb_msort(ideg,ix=idxs,dir=psb_sort_down_)
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p1)
|
||||
|
||||
!
|
||||
! Phase one: Start with disjoint groups.
|
||||
!
|
||||
naggr = 0
|
||||
icnt = 0
|
||||
#if defined(OPENMP)
|
||||
block
|
||||
integer(psb_ipk_), allocatable :: bnds(:), locnaggr(:)
|
||||
integer(psb_ipk_) :: myth,nths, kk
|
||||
! The parallelization makes use of a locaggr(:) array; each thread
|
||||
! keeps its own version of naggr, and when the loop ends, a prefix is applied
|
||||
! to locnaggr to determine:
|
||||
! 1. The total number of aggregaters NAGGR;
|
||||
! 2. How much should each thread shift its own aggregates
|
||||
! Part 2 requires to keep track of which thread defined each entry
|
||||
! of ilaggr(), so that each entry can be adjusted correctly: even
|
||||
! if an entry I belongs to the range BNDS(TH)>BNDS(TH+1)-1, it may have
|
||||
! been set because it is strongly connected to an entry J belonging to a
|
||||
! different thread.
|
||||
|
||||
!$omp parallel shared(bnds,idxs,locnaggr,ilaggr,nr,naggr,diag,theta,nths,info) &
|
||||
!$omp private(icol,val,myth,kk)
|
||||
block
|
||||
integer(psb_ipk_) :: ii,nlp,k,kp,n,ia,isz, nc, i,j,m, nz, ilg, ip, rsz
|
||||
integer(psb_lpk_) :: itmp
|
||||
!$omp master
|
||||
nths = omp_get_num_threads()
|
||||
allocate(bnds(0:nths),locnaggr(0:nths+1))
|
||||
locnaggr(:) = 0
|
||||
bnds(0) = 1
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
myth = omp_get_thread_num()
|
||||
rsz = nr/nths
|
||||
if (myth < mod(nr,nths)) rsz = rsz + 1
|
||||
bnds(myth+1) = rsz
|
||||
!$omp barrier
|
||||
!$omp master
|
||||
do i=1,nths
|
||||
bnds(i) = bnds(i) + bnds(i-1)
|
||||
end do
|
||||
info = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
|
||||
!$omp do schedule(static) private(disjoint)
|
||||
do kk=0, nths-1
|
||||
step1: do ii=bnds(kk), bnds(kk+1)-1
|
||||
i = idxs(ii)
|
||||
if (info /= 0) cycle step1
|
||||
if ((i<1).or.(i>nr)) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if ((nz<0).or.(nz>size(icol))) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
icol(1:nz) = acsr%ja(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
val(1:nz) = acsr%val(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
|
||||
!
|
||||
! Build the set of all strongly coupled nodes
|
||||
!
|
||||
ip = 0
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
! If any of the neighbours is already assigned,
|
||||
! we will not reset.
|
||||
if (j>nr) cycle step1
|
||||
if (ilaggr(j) > 0) cycle step1
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
enddo
|
||||
|
||||
!
|
||||
! If the whole strongly coupled neighborhood of I is
|
||||
! as yet unconnected, turn it into the next aggregate.
|
||||
! Same if ip==0 (in which case, neighborhood only
|
||||
! contains I even if it does not look like it from matrix)
|
||||
! The fact that DISJOINT is private and not under lock
|
||||
! generates a certain un-repeatability, in that between
|
||||
! computing DISJOINT and assigning, another thread might
|
||||
! alter the values of ILAGGR.
|
||||
! However, a certain unrepeatability is already present
|
||||
! because the sequence of aggregates is computed with a
|
||||
! different order than in serial mode.
|
||||
! In any case, even if the enteries of ILAGGR may be
|
||||
! overwritten, the important thing is that each entry is
|
||||
! consistent and they generate a correct aggregation map.
|
||||
!
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
!$omp end atomic
|
||||
cycle step1
|
||||
end if
|
||||
!$omp atomic write
|
||||
ilaggr(i) = itmp
|
||||
!$omp end atomic
|
||||
do k=1, ip
|
||||
!$omp atomic write
|
||||
ilaggr(icol(k)) = itmp
|
||||
!$omp end atomic
|
||||
end do
|
||||
end if
|
||||
end if
|
||||
enddo step1
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
!$omp master
|
||||
naggr = sum(locnaggr(0:nths-1))
|
||||
do i=1,nths
|
||||
locnaggr(i) = locnaggr(i) + locnaggr(i-1)
|
||||
end do
|
||||
do i=nths+1,1,-1
|
||||
locnaggr(i) = locnaggr(i-1)
|
||||
end do
|
||||
locnaggr(0) = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
!$omp do schedule(static)
|
||||
do kk=0, nths-1
|
||||
do ii=bnds(kk), bnds(kk+1)-1
|
||||
if (ilaggr(ii) > 0) then
|
||||
kp = mod(ilaggr(ii),nths)
|
||||
ilaggr(ii) = (ilaggr(ii)/nths)- (bnds(kp)-1) + locnaggr(kp)
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end do
|
||||
end block
|
||||
!$omp end parallel
|
||||
end block
|
||||
if (info /= 0) then
|
||||
if (info == 12345678) write(0,*) 'Overflow in encoding ILAGGR'
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
end if
|
||||
#else
|
||||
step1: do ii=1, nr
|
||||
if (info /= 0) cycle
|
||||
i = idxs(ii)
|
||||
if ((i<1).or.(i>nr)) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if ((nz<0).or.(nz>size(icol))) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
icol(1:nz) = acsr%ja(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
@@ -176,7 +352,7 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
! Build the set of all strongly coupled nodes
|
||||
!
|
||||
ip = 0
|
||||
ip = 0
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
if ((1<=j).and.(j<=nr)) then
|
||||
@@ -194,8 +370,7 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! contains I even if it does not look like it from matrix)
|
||||
!
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
icnt = icnt + 1
|
||||
if (disjoint) then
|
||||
naggr = naggr + 1
|
||||
do k=1, ip
|
||||
ilaggr(icol(k)) = naggr
|
||||
@@ -204,16 +379,22 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
endif
|
||||
enddo step1
|
||||
|
||||
#endif
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1:',count(ilaggr == -(nr+1))
|
||||
& ' Check 1:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc1_p1)
|
||||
if (do_timings) call psb_tic(idx_soc1_p2)
|
||||
!
|
||||
! Phase two: join the neighbours
|
||||
!
|
||||
!$omp workshare
|
||||
tmpaggr = ilaggr
|
||||
!$omp end workshare
|
||||
!$omp parallel do schedule(static) shared(tmpaggr,ilaggr,nr,naggr,diag,theta)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip,cpling)
|
||||
step2: do ii=1,nr
|
||||
i = idxs(ii)
|
||||
|
||||
@@ -244,8 +425,15 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
end if
|
||||
end do step2
|
||||
!$omp end parallel do
|
||||
if (do_timings) call psb_toc(idx_soc1_p2)
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1.5:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p3)
|
||||
!
|
||||
! Phase three: sweep over leftovers, if any
|
||||
!
|
||||
@@ -274,7 +462,6 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
enddo
|
||||
if (ip > 0) then
|
||||
icnt = icnt + 1
|
||||
naggr = naggr + 1
|
||||
ilaggr(i) = naggr
|
||||
do k=1, ip
|
||||
@@ -292,7 +479,10 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end do step3
|
||||
|
||||
! Any leftovers?
|
||||
!$omp parallel do schedule(static) shared(ilaggr,info)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip)
|
||||
do i=1, nr
|
||||
if (info /= 0) cycle
|
||||
if (ilaggr(i) < 0) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if (nz == 1) then
|
||||
@@ -303,15 +493,18 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! other processes.
|
||||
ilaggr(i) = -(nrglob+nr)
|
||||
else
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name,a_err='Fatal error: non-singleton leftovers')
|
||||
goto 9999
|
||||
cycle
|
||||
endif
|
||||
end if
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
if (info /= 0) goto 9999
|
||||
if (do_timings) call psb_toc(idx_soc1_p3)
|
||||
if (naggr > ncol) then
|
||||
!write(0,*) name,'Error : naggr > ncol',naggr,ncol
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Fatal error: naggr>ncol')
|
||||
goto 9999
|
||||
@@ -336,9 +529,13 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nlaggr(:) = 0
|
||||
nlaggr(me+1) = naggr
|
||||
call psb_sum(ctxt,nlaggr(1:np))
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 2:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
call acsr%free()
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
+205
-16
@@ -68,9 +68,12 @@
|
||||
!
|
||||
subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,info)
|
||||
|
||||
use psb_base_mod
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_c_inner_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
|
||||
implicit none
|
||||
|
||||
@@ -99,6 +102,9 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_) :: np, me
|
||||
integer(psb_ipk_) :: nrow, ncol, n_ne
|
||||
character(len=20) :: name, ch_err
|
||||
integer(psb_ipk_), save :: idx_soc2_p1=-1, idx_soc2_p2=-1, idx_soc2_p3=-1
|
||||
integer(psb_ipk_), save :: idx_soc2_p0=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_soc2_map_bld'
|
||||
@@ -114,6 +120,14 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
nrglob = desc_a%get_global_rows()
|
||||
if ((do_timings).and.(idx_soc2_p0==-1)) &
|
||||
& idx_soc2_p0 = psb_get_timer_idx("SOC2_MAP: phase0")
|
||||
if ((do_timings).and.(idx_soc2_p1==-1)) &
|
||||
& idx_soc2_p1 = psb_get_timer_idx("SOC2_MAP: phase1")
|
||||
if ((do_timings).and.(idx_soc2_p2==-1)) &
|
||||
& idx_soc2_p2 = psb_get_timer_idx("SOC2_MAP: phase2")
|
||||
if ((do_timings).and.(idx_soc2_p3==-1)) &
|
||||
& idx_soc2_p3 = psb_get_timer_idx("SOC2_MAP: phase3")
|
||||
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -125,6 +139,7 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc2_p0)
|
||||
diag = a%get_diag(info)
|
||||
if(info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
@@ -137,55 +152,217 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
call a%cp_to(muij)
|
||||
if (clean_zeros) call muij%clean_zeros(info)
|
||||
!$omp parallel do private(i,j,k) shared(nr,diag,muij) schedule(static)
|
||||
do i=1, nr
|
||||
do k=muij%irp(i),muij%irp(i+1)-1
|
||||
j = muij%ja(k)
|
||||
if (j<= nr) muij%val(k) = abs(muij%val(k))/sqrt(abs(diag(i)*diag(j)))
|
||||
end do
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
!
|
||||
! Compute the 1-neigbour; mark strong links with +1, weak links with -1
|
||||
!
|
||||
call s_neigh_coo%allocate(nr,nr,muij%get_nzeros())
|
||||
ip = 0
|
||||
!$omp parallel do private(i,j,k) shared(nr,diag,muij) schedule(static)
|
||||
do i=1, nr
|
||||
do k=muij%irp(i),muij%irp(i+1)-1
|
||||
j = muij%ja(k)
|
||||
s_neigh_coo%ia(k) = i
|
||||
s_neigh_coo%ja(k) = j
|
||||
if (j<=nr) then
|
||||
ip = ip + 1
|
||||
s_neigh_coo%ia(ip) = i
|
||||
s_neigh_coo%ja(ip) = j
|
||||
if (real(muij%val(k)) >= theta) then
|
||||
s_neigh_coo%val(ip) = sone
|
||||
s_neigh_coo%val(k) = sone
|
||||
else
|
||||
s_neigh_coo%val(ip) = -sone
|
||||
s_neigh_coo%val(k) = -sone
|
||||
end if
|
||||
else
|
||||
s_neigh_coo%val(k) = -sone
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end parallel do
|
||||
!write(*,*) 'S_NEIGH: ',nr,ip
|
||||
call s_neigh_coo%set_nzeros(ip)
|
||||
call s_neigh_coo%set_nzeros(muij%get_nzeros())
|
||||
call s_neigh%mv_from_coo(s_neigh_coo,info)
|
||||
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
|
||||
!$omp parallel do private(i) shared(ilaggr,idxs) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
idxs(i) = i
|
||||
end do
|
||||
!$omp end parallel do
|
||||
else
|
||||
!$omp parallel do private(i) shared(ilaggr,idxs,muij) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
ideg(i) = muij%irp(i+1) - muij%irp(i)
|
||||
end do
|
||||
!$omp end parallel do
|
||||
call psb_msort(ideg,ix=idxs,dir=psb_sort_down_)
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc2_p0)
|
||||
if (do_timings) call psb_tic(idx_soc2_p1)
|
||||
|
||||
!
|
||||
! Phase one: Start with disjoint groups.
|
||||
!
|
||||
naggr = 0
|
||||
#if defined(OPENMP)
|
||||
block
|
||||
integer(psb_ipk_), allocatable :: bnds(:), locnaggr(:)
|
||||
integer(psb_ipk_) :: myth,nths, kk
|
||||
! The parallelization makes use of a locaggr(:) array; each thread
|
||||
! keeps its own version of naggr, and when the loop ends, a prefix is applied
|
||||
! to locnaggr to determine:
|
||||
! 1. The total number of aggregaters NAGGR;
|
||||
! 2. How much should each thread shift its own aggregates
|
||||
! Part 2 requires to keep track of which thread defined each entry
|
||||
! of ilaggr(), so that each entry can be adjusted correctly: even
|
||||
! if an entry I belongs to the range BNDS(TH)>BNDS(TH+1)-1, it may have
|
||||
! been set because it is strongly connected to an entry J belonging to a
|
||||
! different thread.
|
||||
|
||||
!$omp parallel shared(s_neigh,bnds,idxs,locnaggr,ilaggr,nr,naggr,diag,theta,nths,info) &
|
||||
!$omp private(icol,val,myth,kk)
|
||||
block
|
||||
integer(psb_ipk_) :: ii,nlp,k,kp,n,ia,isz,nc,i,j,m,nz,ilg,ip,rsz,ip1,nzcnt
|
||||
integer(psb_lpk_) :: itmp
|
||||
!$omp master
|
||||
nths = omp_get_num_threads()
|
||||
allocate(bnds(0:nths),locnaggr(0:nths+1))
|
||||
locnaggr(:) = 0
|
||||
bnds(0) = 1
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
myth = omp_get_thread_num()
|
||||
rsz = nr/nths
|
||||
if (myth < mod(nr,nths)) rsz = rsz + 1
|
||||
bnds(myth+1) = rsz
|
||||
!$omp barrier
|
||||
!$omp master
|
||||
do i=1,nths
|
||||
bnds(i) = bnds(i) + bnds(i-1)
|
||||
end do
|
||||
info = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
|
||||
!$omp do schedule(static) private(disjoint)
|
||||
do kk=0, nths-1
|
||||
step1: do ii=bnds(kk), bnds(kk+1)-1
|
||||
i = idxs(ii)
|
||||
if (info /= 0) then
|
||||
write(0,*) ' Step1:',kk,ii,i,info
|
||||
cycle step1
|
||||
end if
|
||||
if ((i<1).or.(i>nr)) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
!
|
||||
! Get the 1-neighbourhood of I
|
||||
!
|
||||
ip1 = s_neigh%irp(i)
|
||||
nz = s_neigh%irp(i+1)-ip1
|
||||
!
|
||||
! If the neighbourhood only contains I, skip it
|
||||
!
|
||||
if (nz ==0) then
|
||||
ilaggr(i) = 0
|
||||
cycle step1
|
||||
end if
|
||||
if ((nz==1).and.(s_neigh%ja(ip1)==i)) then
|
||||
ilaggr(i) = 0
|
||||
cycle step1
|
||||
end if
|
||||
|
||||
nzcnt = count(real(s_neigh%val(ip1:ip1+nz-1)) > 0)
|
||||
icol(1:nzcnt) = pack(s_neigh%ja(ip1:ip1+nz-1),(real(s_neigh%val(ip1:ip1+nz-1)) > 0))
|
||||
disjoint = all(ilaggr(icol(1:nzcnt)) == -(nr+1))
|
||||
|
||||
!
|
||||
! If the whole strongly coupled neighborhood of I is
|
||||
! as yet unconnected, turn it into the next aggregate.
|
||||
! Same if ip==0 (in which case, neighborhood only
|
||||
! contains I even if it does not look like it from matrix)
|
||||
! The fact that DISJOINT is private and not under lock
|
||||
! generates a certain un-repeatability, in that between
|
||||
! computing DISJOINT and assigning, another thread might
|
||||
! alter the values of ILAGGR.
|
||||
! However, a certain unrepeatability is already present
|
||||
! because the sequence of aggregates is computed with a
|
||||
! different order than in serial mode.
|
||||
! In any case, even if the enteries of ILAGGR may be
|
||||
! overwritten, the important thing is that each entry is
|
||||
! consistent and they generate a correct aggregation map.
|
||||
!
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
!$omp end atomic
|
||||
cycle step1
|
||||
end if
|
||||
!$omp atomic write
|
||||
ilaggr(i) = itmp
|
||||
!$omp end atomic
|
||||
do k=1, nzcnt
|
||||
!$omp atomic write
|
||||
ilaggr(icol(k)) = itmp
|
||||
!$omp end atomic
|
||||
end do
|
||||
end if
|
||||
end if
|
||||
enddo step1
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
!$omp master
|
||||
naggr = sum(locnaggr(0:nths-1))
|
||||
do i=1,nths
|
||||
locnaggr(i) = locnaggr(i) + locnaggr(i-1)
|
||||
end do
|
||||
do i=nths+1,1,-1
|
||||
locnaggr(i) = locnaggr(i-1)
|
||||
end do
|
||||
locnaggr(0) = 0
|
||||
!write(0,*) 'LNAG ',locnaggr(nths+1)
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
!$omp do schedule(static)
|
||||
do kk=0, nths-1
|
||||
do ii=bnds(kk), bnds(kk+1)-1
|
||||
if (ilaggr(ii) > 0) then
|
||||
kp = mod(ilaggr(ii),nths)
|
||||
ilaggr(ii) = (ilaggr(ii)/nths)- (bnds(kp)-1) + locnaggr(kp)
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end do
|
||||
end block
|
||||
!$omp end parallel
|
||||
end block
|
||||
if (info /= 0) then
|
||||
if (info == 12345678) write(0,*) 'Overflow in encoding ILAGGR'
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
#else
|
||||
icnt = 0
|
||||
step1: do ii=1, nr
|
||||
i = idxs(ii)
|
||||
@@ -224,16 +401,21 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
endif
|
||||
enddo step1
|
||||
|
||||
#endif
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1:',count(ilaggr == -(nr+1))
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc2_p1)
|
||||
if (do_timings) call psb_tic(idx_soc2_p2)
|
||||
!
|
||||
! Phase two: join the neighbours
|
||||
!
|
||||
!$omp workshare
|
||||
tmpaggr = ilaggr
|
||||
!$omp end workshare
|
||||
!$omp parallel do schedule(static) shared(tmpaggr,ilaggr,nr,naggr,diag,muij,s_neigh)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip,cpling)
|
||||
step2: do ii=1,nr
|
||||
i = idxs(ii)
|
||||
|
||||
@@ -259,8 +441,9 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
end if
|
||||
end do step2
|
||||
|
||||
|
||||
!$omp end parallel do
|
||||
if (do_timings) call psb_toc(idx_soc2_p2)
|
||||
if (do_timings) call psb_tic(idx_soc2_p3)
|
||||
!
|
||||
! Phase three: sweep over leftovers, if any
|
||||
!
|
||||
@@ -294,6 +477,8 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end do step3
|
||||
|
||||
! Any leftovers?
|
||||
!$omp parallel do schedule(static) shared(ilaggr,s_neigh,info)&
|
||||
!$omp private(ii,i,j,k)
|
||||
do i=1, nr
|
||||
if (ilaggr(i) <= 0) then
|
||||
nz = (s_neigh%irp(i+1)-s_neigh%irp(i))
|
||||
@@ -305,13 +490,17 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! other processes.
|
||||
ilaggr(i) = -(nrglob+nr)
|
||||
else
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name,a_err='Fatal error: non-singleton leftovers')
|
||||
goto 9999
|
||||
cycle
|
||||
endif
|
||||
end if
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
if (info /= 0) goto 9999
|
||||
if (do_timings) call psb_toc(idx_soc2_p3)
|
||||
if (naggr > ncol) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Fatal error: naggr>ncol')
|
||||
@@ -76,7 +76,7 @@ subroutine amg_d_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
integer(psb_ipk_) :: nrow, ncol, nrl, nzl, ip, nzt, i, k
|
||||
integer(psb_lpk_) :: nrsave, ncsave, nzsave, nza
|
||||
logical, parameter :: do_timings=.false., oldstyle=.false., debug=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_cpytrans1=-1, idx_cpytrans2=-1
|
||||
|
||||
name='amg_ptap_bld'
|
||||
if(psb_get_errstatus().ne.0) return
|
||||
@@ -93,7 +93,11 @@ subroutine amg_d_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("SPMM_BLD: par_spspmm")
|
||||
& idx_spspmm = psb_get_timer_idx("PTAP_BLD: par_spspmm")
|
||||
if ((do_timings).and.(idx_cpytrans1==-1)) &
|
||||
& idx_cpytrans1 = psb_get_timer_idx("PTAP_BLD: cpy&trans1")
|
||||
if ((do_timings).and.(idx_cpytrans2==-1)) &
|
||||
& idx_cpytrans2 = psb_get_timer_idx("PTAP_BLD: cpy&trans2")
|
||||
|
||||
naggr = nlaggr(me+1)
|
||||
ntaggr = sum(nlaggr)
|
||||
@@ -128,6 +132,7 @@ subroutine amg_d_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
! Ok first product done.
|
||||
|
||||
if (present(desc_ax)) then
|
||||
if (do_timings) call psb_tic(idx_cpytrans1)
|
||||
block
|
||||
call coo_prol%cp_to_coo(coo_restr,info)
|
||||
call coo_restr%set_ncols(desc_ac%get_local_cols())
|
||||
@@ -137,7 +142,7 @@ subroutine amg_d_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
call coo_restr%set_ncols(desc_ax%get_local_cols())
|
||||
end block
|
||||
call csr_restr%cp_from_coo(coo_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_cpytrans1)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spcnv coo_restr')
|
||||
goto 9999
|
||||
@@ -167,27 +172,28 @@ subroutine amg_d_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call coo_restr%transp()
|
||||
nzl = coo_restr%get_nzeros()
|
||||
nrl = desc_ac%get_local_rows()
|
||||
i=0
|
||||
nrl = desc_ac%get_local_rows()
|
||||
call coo_restr%fix(info)
|
||||
i=coo_restr%get_nzeros()
|
||||
!
|
||||
! Only keep local rows
|
||||
!
|
||||
do k=1, nzl
|
||||
if ((1 <= coo_restr%ia(k)) .and.(coo_restr%ia(k) <= nrl)) then
|
||||
i = i+1
|
||||
coo_restr%val(i) = coo_restr%val(k)
|
||||
coo_restr%ia(i) = coo_restr%ia(k)
|
||||
coo_restr%ja(i) = coo_restr%ja(k)
|
||||
search: do k=i,1,-1
|
||||
if (coo_restr%ia(k) <= nrl) then
|
||||
call coo_restr%set_nzeros(k)
|
||||
exit search
|
||||
end if
|
||||
end do
|
||||
call coo_restr%set_nzeros(i)
|
||||
call coo_restr%fix(info)
|
||||
end do search
|
||||
|
||||
nzl = coo_restr%get_nzeros()
|
||||
call coo_restr%set_nrows(desc_ac%get_local_rows())
|
||||
call coo_restr%set_ncols(desc_a%get_local_cols())
|
||||
if (debug) call check_coo(me,trim(name)//' Check 2 on coo_restr:',coo_restr)
|
||||
if (do_timings) call psb_tic(idx_cpytrans2)
|
||||
|
||||
call csr_restr%cp_from_coo(coo_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_cpytrans2)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spcnv coo_restr')
|
||||
goto 9999
|
||||
|
||||
+218
-21
@@ -72,7 +72,9 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_inner_mod
|
||||
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
@@ -85,7 +87,7 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_), allocatable :: ils(:), neigh(:), irow(:), icol(:),&
|
||||
integer(psb_ipk_), allocatable :: neigh(:), irow(:), icol(:),&
|
||||
& ideg(:), idxs(:)
|
||||
integer(psb_lpk_), allocatable :: tmpaggr(:)
|
||||
real(psb_dpk_), allocatable :: val(:), diag(:)
|
||||
@@ -99,6 +101,9 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_) :: nrow, ncol, n_ne
|
||||
integer(psb_lpk_) :: nrglob
|
||||
character(len=20) :: name, ch_err
|
||||
integer(psb_ipk_), save :: idx_soc1_p1=-1, idx_soc1_p2=-1, idx_soc1_p3=-1
|
||||
integer(psb_ipk_), save :: idx_soc1_p0=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_soc1_map_bld'
|
||||
@@ -114,6 +119,14 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
nrglob = desc_a%get_global_rows()
|
||||
if ((do_timings).and.(idx_soc1_p0==-1)) &
|
||||
& idx_soc1_p0 = psb_get_timer_idx("SOC1_MAP: phase0")
|
||||
if ((do_timings).and.(idx_soc1_p1==-1)) &
|
||||
& idx_soc1_p1 = psb_get_timer_idx("SOC1_MAP: phase1")
|
||||
if ((do_timings).and.(idx_soc1_p2==-1)) &
|
||||
& idx_soc1_p2 = psb_get_timer_idx("SOC1_MAP: phase2")
|
||||
if ((do_timings).and.(idx_soc1_p3==-1)) &
|
||||
& idx_soc1_p3 = psb_get_timer_idx("SOC1_MAP: phase3")
|
||||
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -133,41 +146,204 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p0)
|
||||
call a%cp_to(acsr)
|
||||
if (do_timings) call psb_toc(idx_soc1_p0)
|
||||
if (clean_zeros) call acsr%clean_zeros(info)
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
!$omp parallel do private(i) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
idxs(i) = i
|
||||
idxs(i) = i
|
||||
end do
|
||||
else
|
||||
!$omp end parallel do
|
||||
else
|
||||
!$omp parallel do private(i) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
ideg(i) = acsr%irp(i+1) - acsr%irp(i)
|
||||
end do
|
||||
!$omp end parallel do
|
||||
call psb_msort(ideg,ix=idxs,dir=psb_sort_down_)
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p1)
|
||||
|
||||
!
|
||||
! Phase one: Start with disjoint groups.
|
||||
!
|
||||
naggr = 0
|
||||
icnt = 0
|
||||
#if defined(OPENMP)
|
||||
block
|
||||
integer(psb_ipk_), allocatable :: bnds(:), locnaggr(:)
|
||||
integer(psb_ipk_) :: myth,nths, kk
|
||||
! The parallelization makes use of a locaggr(:) array; each thread
|
||||
! keeps its own version of naggr, and when the loop ends, a prefix is applied
|
||||
! to locnaggr to determine:
|
||||
! 1. The total number of aggregaters NAGGR;
|
||||
! 2. How much should each thread shift its own aggregates
|
||||
! Part 2 requires to keep track of which thread defined each entry
|
||||
! of ilaggr(), so that each entry can be adjusted correctly: even
|
||||
! if an entry I belongs to the range BNDS(TH)>BNDS(TH+1)-1, it may have
|
||||
! been set because it is strongly connected to an entry J belonging to a
|
||||
! different thread.
|
||||
|
||||
!$omp parallel shared(bnds,idxs,locnaggr,ilaggr,nr,naggr,diag,theta,nths,info) &
|
||||
!$omp private(icol,val,myth,kk)
|
||||
block
|
||||
integer(psb_ipk_) :: ii,nlp,k,kp,n,ia,isz, nc, i,j,m, nz, ilg, ip, rsz
|
||||
integer(psb_lpk_) :: itmp
|
||||
!$omp master
|
||||
nths = omp_get_num_threads()
|
||||
allocate(bnds(0:nths),locnaggr(0:nths+1))
|
||||
locnaggr(:) = 0
|
||||
bnds(0) = 1
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
myth = omp_get_thread_num()
|
||||
rsz = nr/nths
|
||||
if (myth < mod(nr,nths)) rsz = rsz + 1
|
||||
bnds(myth+1) = rsz
|
||||
!$omp barrier
|
||||
!$omp master
|
||||
do i=1,nths
|
||||
bnds(i) = bnds(i) + bnds(i-1)
|
||||
end do
|
||||
info = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
|
||||
!$omp do schedule(static) private(disjoint)
|
||||
do kk=0, nths-1
|
||||
step1: do ii=bnds(kk), bnds(kk+1)-1
|
||||
i = idxs(ii)
|
||||
if (info /= 0) cycle step1
|
||||
if ((i<1).or.(i>nr)) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if ((nz<0).or.(nz>size(icol))) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
icol(1:nz) = acsr%ja(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
val(1:nz) = acsr%val(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
|
||||
!
|
||||
! Build the set of all strongly coupled nodes
|
||||
!
|
||||
ip = 0
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
! If any of the neighbours is already assigned,
|
||||
! we will not reset.
|
||||
if (j>nr) cycle step1
|
||||
if (ilaggr(j) > 0) cycle step1
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
enddo
|
||||
|
||||
!
|
||||
! If the whole strongly coupled neighborhood of I is
|
||||
! as yet unconnected, turn it into the next aggregate.
|
||||
! Same if ip==0 (in which case, neighborhood only
|
||||
! contains I even if it does not look like it from matrix)
|
||||
! The fact that DISJOINT is private and not under lock
|
||||
! generates a certain un-repeatability, in that between
|
||||
! computing DISJOINT and assigning, another thread might
|
||||
! alter the values of ILAGGR.
|
||||
! However, a certain unrepeatability is already present
|
||||
! because the sequence of aggregates is computed with a
|
||||
! different order than in serial mode.
|
||||
! In any case, even if the enteries of ILAGGR may be
|
||||
! overwritten, the important thing is that each entry is
|
||||
! consistent and they generate a correct aggregation map.
|
||||
!
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
!$omp end atomic
|
||||
cycle step1
|
||||
end if
|
||||
!$omp atomic write
|
||||
ilaggr(i) = itmp
|
||||
!$omp end atomic
|
||||
do k=1, ip
|
||||
!$omp atomic write
|
||||
ilaggr(icol(k)) = itmp
|
||||
!$omp end atomic
|
||||
end do
|
||||
end if
|
||||
end if
|
||||
enddo step1
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
!$omp master
|
||||
naggr = sum(locnaggr(0:nths-1))
|
||||
do i=1,nths
|
||||
locnaggr(i) = locnaggr(i) + locnaggr(i-1)
|
||||
end do
|
||||
do i=nths+1,1,-1
|
||||
locnaggr(i) = locnaggr(i-1)
|
||||
end do
|
||||
locnaggr(0) = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
!$omp do schedule(static)
|
||||
do kk=0, nths-1
|
||||
do ii=bnds(kk), bnds(kk+1)-1
|
||||
if (ilaggr(ii) > 0) then
|
||||
kp = mod(ilaggr(ii),nths)
|
||||
ilaggr(ii) = (ilaggr(ii)/nths)- (bnds(kp)-1) + locnaggr(kp)
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end do
|
||||
end block
|
||||
!$omp end parallel
|
||||
end block
|
||||
if (info /= 0) then
|
||||
if (info == 12345678) write(0,*) 'Overflow in encoding ILAGGR'
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
end if
|
||||
#else
|
||||
step1: do ii=1, nr
|
||||
if (info /= 0) cycle
|
||||
i = idxs(ii)
|
||||
if ((i<1).or.(i>nr)) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if ((nz<0).or.(nz>size(icol))) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
icol(1:nz) = acsr%ja(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
@@ -176,7 +352,7 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
! Build the set of all strongly coupled nodes
|
||||
!
|
||||
ip = 0
|
||||
ip = 0
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
if ((1<=j).and.(j<=nr)) then
|
||||
@@ -194,8 +370,7 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! contains I even if it does not look like it from matrix)
|
||||
!
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
icnt = icnt + 1
|
||||
if (disjoint) then
|
||||
naggr = naggr + 1
|
||||
do k=1, ip
|
||||
ilaggr(icol(k)) = naggr
|
||||
@@ -204,16 +379,22 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
endif
|
||||
enddo step1
|
||||
|
||||
#endif
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1:',count(ilaggr == -(nr+1))
|
||||
& ' Check 1:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc1_p1)
|
||||
if (do_timings) call psb_tic(idx_soc1_p2)
|
||||
!
|
||||
! Phase two: join the neighbours
|
||||
!
|
||||
!$omp workshare
|
||||
tmpaggr = ilaggr
|
||||
!$omp end workshare
|
||||
!$omp parallel do schedule(static) shared(tmpaggr,ilaggr,nr,naggr,diag,theta)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip,cpling)
|
||||
step2: do ii=1,nr
|
||||
i = idxs(ii)
|
||||
|
||||
@@ -244,8 +425,15 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
end if
|
||||
end do step2
|
||||
!$omp end parallel do
|
||||
if (do_timings) call psb_toc(idx_soc1_p2)
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1.5:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p3)
|
||||
!
|
||||
! Phase three: sweep over leftovers, if any
|
||||
!
|
||||
@@ -274,7 +462,6 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
enddo
|
||||
if (ip > 0) then
|
||||
icnt = icnt + 1
|
||||
naggr = naggr + 1
|
||||
ilaggr(i) = naggr
|
||||
do k=1, ip
|
||||
@@ -292,7 +479,10 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end do step3
|
||||
|
||||
! Any leftovers?
|
||||
!$omp parallel do schedule(static) shared(ilaggr,info)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip)
|
||||
do i=1, nr
|
||||
if (info /= 0) cycle
|
||||
if (ilaggr(i) < 0) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if (nz == 1) then
|
||||
@@ -303,15 +493,18 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! other processes.
|
||||
ilaggr(i) = -(nrglob+nr)
|
||||
else
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name,a_err='Fatal error: non-singleton leftovers')
|
||||
goto 9999
|
||||
cycle
|
||||
endif
|
||||
end if
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
if (info /= 0) goto 9999
|
||||
if (do_timings) call psb_toc(idx_soc1_p3)
|
||||
if (naggr > ncol) then
|
||||
!write(0,*) name,'Error : naggr > ncol',naggr,ncol
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Fatal error: naggr>ncol')
|
||||
goto 9999
|
||||
@@ -336,9 +529,13 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nlaggr(:) = 0
|
||||
nlaggr(me+1) = naggr
|
||||
call psb_sum(ctxt,nlaggr(1:np))
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 2:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
call acsr%free()
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
+205
-16
@@ -68,9 +68,12 @@
|
||||
!
|
||||
subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,info)
|
||||
|
||||
use psb_base_mod
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_inner_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
|
||||
implicit none
|
||||
|
||||
@@ -99,6 +102,9 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_) :: np, me
|
||||
integer(psb_ipk_) :: nrow, ncol, n_ne
|
||||
character(len=20) :: name, ch_err
|
||||
integer(psb_ipk_), save :: idx_soc2_p1=-1, idx_soc2_p2=-1, idx_soc2_p3=-1
|
||||
integer(psb_ipk_), save :: idx_soc2_p0=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_soc2_map_bld'
|
||||
@@ -114,6 +120,14 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
nrglob = desc_a%get_global_rows()
|
||||
if ((do_timings).and.(idx_soc2_p0==-1)) &
|
||||
& idx_soc2_p0 = psb_get_timer_idx("SOC2_MAP: phase0")
|
||||
if ((do_timings).and.(idx_soc2_p1==-1)) &
|
||||
& idx_soc2_p1 = psb_get_timer_idx("SOC2_MAP: phase1")
|
||||
if ((do_timings).and.(idx_soc2_p2==-1)) &
|
||||
& idx_soc2_p2 = psb_get_timer_idx("SOC2_MAP: phase2")
|
||||
if ((do_timings).and.(idx_soc2_p3==-1)) &
|
||||
& idx_soc2_p3 = psb_get_timer_idx("SOC2_MAP: phase3")
|
||||
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -125,6 +139,7 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc2_p0)
|
||||
diag = a%get_diag(info)
|
||||
if(info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
@@ -137,55 +152,217 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
call a%cp_to(muij)
|
||||
if (clean_zeros) call muij%clean_zeros(info)
|
||||
!$omp parallel do private(i,j,k) shared(nr,diag,muij) schedule(static)
|
||||
do i=1, nr
|
||||
do k=muij%irp(i),muij%irp(i+1)-1
|
||||
j = muij%ja(k)
|
||||
if (j<= nr) muij%val(k) = abs(muij%val(k))/sqrt(abs(diag(i)*diag(j)))
|
||||
end do
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
!
|
||||
! Compute the 1-neigbour; mark strong links with +1, weak links with -1
|
||||
!
|
||||
call s_neigh_coo%allocate(nr,nr,muij%get_nzeros())
|
||||
ip = 0
|
||||
!$omp parallel do private(i,j,k) shared(nr,diag,muij) schedule(static)
|
||||
do i=1, nr
|
||||
do k=muij%irp(i),muij%irp(i+1)-1
|
||||
j = muij%ja(k)
|
||||
s_neigh_coo%ia(k) = i
|
||||
s_neigh_coo%ja(k) = j
|
||||
if (j<=nr) then
|
||||
ip = ip + 1
|
||||
s_neigh_coo%ia(ip) = i
|
||||
s_neigh_coo%ja(ip) = j
|
||||
if (real(muij%val(k)) >= theta) then
|
||||
s_neigh_coo%val(ip) = done
|
||||
s_neigh_coo%val(k) = done
|
||||
else
|
||||
s_neigh_coo%val(ip) = -done
|
||||
s_neigh_coo%val(k) = -done
|
||||
end if
|
||||
else
|
||||
s_neigh_coo%val(k) = -done
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end parallel do
|
||||
!write(*,*) 'S_NEIGH: ',nr,ip
|
||||
call s_neigh_coo%set_nzeros(ip)
|
||||
call s_neigh_coo%set_nzeros(muij%get_nzeros())
|
||||
call s_neigh%mv_from_coo(s_neigh_coo,info)
|
||||
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
|
||||
!$omp parallel do private(i) shared(ilaggr,idxs) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
idxs(i) = i
|
||||
end do
|
||||
!$omp end parallel do
|
||||
else
|
||||
!$omp parallel do private(i) shared(ilaggr,idxs,muij) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
ideg(i) = muij%irp(i+1) - muij%irp(i)
|
||||
end do
|
||||
!$omp end parallel do
|
||||
call psb_msort(ideg,ix=idxs,dir=psb_sort_down_)
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc2_p0)
|
||||
if (do_timings) call psb_tic(idx_soc2_p1)
|
||||
|
||||
!
|
||||
! Phase one: Start with disjoint groups.
|
||||
!
|
||||
naggr = 0
|
||||
#if defined(OPENMP)
|
||||
block
|
||||
integer(psb_ipk_), allocatable :: bnds(:), locnaggr(:)
|
||||
integer(psb_ipk_) :: myth,nths, kk
|
||||
! The parallelization makes use of a locaggr(:) array; each thread
|
||||
! keeps its own version of naggr, and when the loop ends, a prefix is applied
|
||||
! to locnaggr to determine:
|
||||
! 1. The total number of aggregaters NAGGR;
|
||||
! 2. How much should each thread shift its own aggregates
|
||||
! Part 2 requires to keep track of which thread defined each entry
|
||||
! of ilaggr(), so that each entry can be adjusted correctly: even
|
||||
! if an entry I belongs to the range BNDS(TH)>BNDS(TH+1)-1, it may have
|
||||
! been set because it is strongly connected to an entry J belonging to a
|
||||
! different thread.
|
||||
|
||||
!$omp parallel shared(s_neigh,bnds,idxs,locnaggr,ilaggr,nr,naggr,diag,theta,nths,info) &
|
||||
!$omp private(icol,val,myth,kk)
|
||||
block
|
||||
integer(psb_ipk_) :: ii,nlp,k,kp,n,ia,isz,nc,i,j,m,nz,ilg,ip,rsz,ip1,nzcnt
|
||||
integer(psb_lpk_) :: itmp
|
||||
!$omp master
|
||||
nths = omp_get_num_threads()
|
||||
allocate(bnds(0:nths),locnaggr(0:nths+1))
|
||||
locnaggr(:) = 0
|
||||
bnds(0) = 1
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
myth = omp_get_thread_num()
|
||||
rsz = nr/nths
|
||||
if (myth < mod(nr,nths)) rsz = rsz + 1
|
||||
bnds(myth+1) = rsz
|
||||
!$omp barrier
|
||||
!$omp master
|
||||
do i=1,nths
|
||||
bnds(i) = bnds(i) + bnds(i-1)
|
||||
end do
|
||||
info = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
|
||||
!$omp do schedule(static) private(disjoint)
|
||||
do kk=0, nths-1
|
||||
step1: do ii=bnds(kk), bnds(kk+1)-1
|
||||
i = idxs(ii)
|
||||
if (info /= 0) then
|
||||
write(0,*) ' Step1:',kk,ii,i,info
|
||||
cycle step1
|
||||
end if
|
||||
if ((i<1).or.(i>nr)) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
!
|
||||
! Get the 1-neighbourhood of I
|
||||
!
|
||||
ip1 = s_neigh%irp(i)
|
||||
nz = s_neigh%irp(i+1)-ip1
|
||||
!
|
||||
! If the neighbourhood only contains I, skip it
|
||||
!
|
||||
if (nz ==0) then
|
||||
ilaggr(i) = 0
|
||||
cycle step1
|
||||
end if
|
||||
if ((nz==1).and.(s_neigh%ja(ip1)==i)) then
|
||||
ilaggr(i) = 0
|
||||
cycle step1
|
||||
end if
|
||||
|
||||
nzcnt = count(real(s_neigh%val(ip1:ip1+nz-1)) > 0)
|
||||
icol(1:nzcnt) = pack(s_neigh%ja(ip1:ip1+nz-1),(real(s_neigh%val(ip1:ip1+nz-1)) > 0))
|
||||
disjoint = all(ilaggr(icol(1:nzcnt)) == -(nr+1))
|
||||
|
||||
!
|
||||
! If the whole strongly coupled neighborhood of I is
|
||||
! as yet unconnected, turn it into the next aggregate.
|
||||
! Same if ip==0 (in which case, neighborhood only
|
||||
! contains I even if it does not look like it from matrix)
|
||||
! The fact that DISJOINT is private and not under lock
|
||||
! generates a certain un-repeatability, in that between
|
||||
! computing DISJOINT and assigning, another thread might
|
||||
! alter the values of ILAGGR.
|
||||
! However, a certain unrepeatability is already present
|
||||
! because the sequence of aggregates is computed with a
|
||||
! different order than in serial mode.
|
||||
! In any case, even if the enteries of ILAGGR may be
|
||||
! overwritten, the important thing is that each entry is
|
||||
! consistent and they generate a correct aggregation map.
|
||||
!
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
!$omp end atomic
|
||||
cycle step1
|
||||
end if
|
||||
!$omp atomic write
|
||||
ilaggr(i) = itmp
|
||||
!$omp end atomic
|
||||
do k=1, nzcnt
|
||||
!$omp atomic write
|
||||
ilaggr(icol(k)) = itmp
|
||||
!$omp end atomic
|
||||
end do
|
||||
end if
|
||||
end if
|
||||
enddo step1
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
!$omp master
|
||||
naggr = sum(locnaggr(0:nths-1))
|
||||
do i=1,nths
|
||||
locnaggr(i) = locnaggr(i) + locnaggr(i-1)
|
||||
end do
|
||||
do i=nths+1,1,-1
|
||||
locnaggr(i) = locnaggr(i-1)
|
||||
end do
|
||||
locnaggr(0) = 0
|
||||
!write(0,*) 'LNAG ',locnaggr(nths+1)
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
!$omp do schedule(static)
|
||||
do kk=0, nths-1
|
||||
do ii=bnds(kk), bnds(kk+1)-1
|
||||
if (ilaggr(ii) > 0) then
|
||||
kp = mod(ilaggr(ii),nths)
|
||||
ilaggr(ii) = (ilaggr(ii)/nths)- (bnds(kp)-1) + locnaggr(kp)
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end do
|
||||
end block
|
||||
!$omp end parallel
|
||||
end block
|
||||
if (info /= 0) then
|
||||
if (info == 12345678) write(0,*) 'Overflow in encoding ILAGGR'
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
#else
|
||||
icnt = 0
|
||||
step1: do ii=1, nr
|
||||
i = idxs(ii)
|
||||
@@ -224,16 +401,21 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
endif
|
||||
enddo step1
|
||||
|
||||
#endif
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1:',count(ilaggr == -(nr+1))
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc2_p1)
|
||||
if (do_timings) call psb_tic(idx_soc2_p2)
|
||||
!
|
||||
! Phase two: join the neighbours
|
||||
!
|
||||
!$omp workshare
|
||||
tmpaggr = ilaggr
|
||||
!$omp end workshare
|
||||
!$omp parallel do schedule(static) shared(tmpaggr,ilaggr,nr,naggr,diag,muij,s_neigh)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip,cpling)
|
||||
step2: do ii=1,nr
|
||||
i = idxs(ii)
|
||||
|
||||
@@ -259,8 +441,9 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
end if
|
||||
end do step2
|
||||
|
||||
|
||||
!$omp end parallel do
|
||||
if (do_timings) call psb_toc(idx_soc2_p2)
|
||||
if (do_timings) call psb_tic(idx_soc2_p3)
|
||||
!
|
||||
! Phase three: sweep over leftovers, if any
|
||||
!
|
||||
@@ -294,6 +477,8 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end do step3
|
||||
|
||||
! Any leftovers?
|
||||
!$omp parallel do schedule(static) shared(ilaggr,s_neigh,info)&
|
||||
!$omp private(ii,i,j,k)
|
||||
do i=1, nr
|
||||
if (ilaggr(i) <= 0) then
|
||||
nz = (s_neigh%irp(i+1)-s_neigh%irp(i))
|
||||
@@ -305,13 +490,17 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! other processes.
|
||||
ilaggr(i) = -(nrglob+nr)
|
||||
else
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name,a_err='Fatal error: non-singleton leftovers')
|
||||
goto 9999
|
||||
cycle
|
||||
endif
|
||||
end if
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
if (info /= 0) goto 9999
|
||||
if (do_timings) call psb_toc(idx_soc2_p3)
|
||||
if (naggr > ncol) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Fatal error: naggr>ncol')
|
||||
@@ -76,7 +76,7 @@ subroutine amg_s_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
integer(psb_ipk_) :: nrow, ncol, nrl, nzl, ip, nzt, i, k
|
||||
integer(psb_lpk_) :: nrsave, ncsave, nzsave, nza
|
||||
logical, parameter :: do_timings=.false., oldstyle=.false., debug=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_cpytrans1=-1, idx_cpytrans2=-1
|
||||
|
||||
name='amg_ptap_bld'
|
||||
if(psb_get_errstatus().ne.0) return
|
||||
@@ -93,7 +93,11 @@ subroutine amg_s_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("SPMM_BLD: par_spspmm")
|
||||
& idx_spspmm = psb_get_timer_idx("PTAP_BLD: par_spspmm")
|
||||
if ((do_timings).and.(idx_cpytrans1==-1)) &
|
||||
& idx_cpytrans1 = psb_get_timer_idx("PTAP_BLD: cpy&trans1")
|
||||
if ((do_timings).and.(idx_cpytrans2==-1)) &
|
||||
& idx_cpytrans2 = psb_get_timer_idx("PTAP_BLD: cpy&trans2")
|
||||
|
||||
naggr = nlaggr(me+1)
|
||||
ntaggr = sum(nlaggr)
|
||||
@@ -128,6 +132,7 @@ subroutine amg_s_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
! Ok first product done.
|
||||
|
||||
if (present(desc_ax)) then
|
||||
if (do_timings) call psb_tic(idx_cpytrans1)
|
||||
block
|
||||
call coo_prol%cp_to_coo(coo_restr,info)
|
||||
call coo_restr%set_ncols(desc_ac%get_local_cols())
|
||||
@@ -137,7 +142,7 @@ subroutine amg_s_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
call coo_restr%set_ncols(desc_ax%get_local_cols())
|
||||
end block
|
||||
call csr_restr%cp_from_coo(coo_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_cpytrans1)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spcnv coo_restr')
|
||||
goto 9999
|
||||
@@ -167,27 +172,28 @@ subroutine amg_s_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call coo_restr%transp()
|
||||
nzl = coo_restr%get_nzeros()
|
||||
nrl = desc_ac%get_local_rows()
|
||||
i=0
|
||||
nrl = desc_ac%get_local_rows()
|
||||
call coo_restr%fix(info)
|
||||
i=coo_restr%get_nzeros()
|
||||
!
|
||||
! Only keep local rows
|
||||
!
|
||||
do k=1, nzl
|
||||
if ((1 <= coo_restr%ia(k)) .and.(coo_restr%ia(k) <= nrl)) then
|
||||
i = i+1
|
||||
coo_restr%val(i) = coo_restr%val(k)
|
||||
coo_restr%ia(i) = coo_restr%ia(k)
|
||||
coo_restr%ja(i) = coo_restr%ja(k)
|
||||
search: do k=i,1,-1
|
||||
if (coo_restr%ia(k) <= nrl) then
|
||||
call coo_restr%set_nzeros(k)
|
||||
exit search
|
||||
end if
|
||||
end do
|
||||
call coo_restr%set_nzeros(i)
|
||||
call coo_restr%fix(info)
|
||||
end do search
|
||||
|
||||
nzl = coo_restr%get_nzeros()
|
||||
call coo_restr%set_nrows(desc_ac%get_local_rows())
|
||||
call coo_restr%set_ncols(desc_a%get_local_cols())
|
||||
if (debug) call check_coo(me,trim(name)//' Check 2 on coo_restr:',coo_restr)
|
||||
if (do_timings) call psb_tic(idx_cpytrans2)
|
||||
|
||||
call csr_restr%cp_from_coo(coo_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_cpytrans2)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spcnv coo_restr')
|
||||
goto 9999
|
||||
|
||||
+218
-21
@@ -72,7 +72,9 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_s_inner_mod
|
||||
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
@@ -85,7 +87,7 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_), allocatable :: ils(:), neigh(:), irow(:), icol(:),&
|
||||
integer(psb_ipk_), allocatable :: neigh(:), irow(:), icol(:),&
|
||||
& ideg(:), idxs(:)
|
||||
integer(psb_lpk_), allocatable :: tmpaggr(:)
|
||||
real(psb_spk_), allocatable :: val(:), diag(:)
|
||||
@@ -99,6 +101,9 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_) :: nrow, ncol, n_ne
|
||||
integer(psb_lpk_) :: nrglob
|
||||
character(len=20) :: name, ch_err
|
||||
integer(psb_ipk_), save :: idx_soc1_p1=-1, idx_soc1_p2=-1, idx_soc1_p3=-1
|
||||
integer(psb_ipk_), save :: idx_soc1_p0=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_soc1_map_bld'
|
||||
@@ -114,6 +119,14 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
nrglob = desc_a%get_global_rows()
|
||||
if ((do_timings).and.(idx_soc1_p0==-1)) &
|
||||
& idx_soc1_p0 = psb_get_timer_idx("SOC1_MAP: phase0")
|
||||
if ((do_timings).and.(idx_soc1_p1==-1)) &
|
||||
& idx_soc1_p1 = psb_get_timer_idx("SOC1_MAP: phase1")
|
||||
if ((do_timings).and.(idx_soc1_p2==-1)) &
|
||||
& idx_soc1_p2 = psb_get_timer_idx("SOC1_MAP: phase2")
|
||||
if ((do_timings).and.(idx_soc1_p3==-1)) &
|
||||
& idx_soc1_p3 = psb_get_timer_idx("SOC1_MAP: phase3")
|
||||
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -133,41 +146,204 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p0)
|
||||
call a%cp_to(acsr)
|
||||
if (do_timings) call psb_toc(idx_soc1_p0)
|
||||
if (clean_zeros) call acsr%clean_zeros(info)
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
!$omp parallel do private(i) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
idxs(i) = i
|
||||
idxs(i) = i
|
||||
end do
|
||||
else
|
||||
!$omp end parallel do
|
||||
else
|
||||
!$omp parallel do private(i) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
ideg(i) = acsr%irp(i+1) - acsr%irp(i)
|
||||
end do
|
||||
!$omp end parallel do
|
||||
call psb_msort(ideg,ix=idxs,dir=psb_sort_down_)
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p1)
|
||||
|
||||
!
|
||||
! Phase one: Start with disjoint groups.
|
||||
!
|
||||
naggr = 0
|
||||
icnt = 0
|
||||
#if defined(OPENMP)
|
||||
block
|
||||
integer(psb_ipk_), allocatable :: bnds(:), locnaggr(:)
|
||||
integer(psb_ipk_) :: myth,nths, kk
|
||||
! The parallelization makes use of a locaggr(:) array; each thread
|
||||
! keeps its own version of naggr, and when the loop ends, a prefix is applied
|
||||
! to locnaggr to determine:
|
||||
! 1. The total number of aggregaters NAGGR;
|
||||
! 2. How much should each thread shift its own aggregates
|
||||
! Part 2 requires to keep track of which thread defined each entry
|
||||
! of ilaggr(), so that each entry can be adjusted correctly: even
|
||||
! if an entry I belongs to the range BNDS(TH)>BNDS(TH+1)-1, it may have
|
||||
! been set because it is strongly connected to an entry J belonging to a
|
||||
! different thread.
|
||||
|
||||
!$omp parallel shared(bnds,idxs,locnaggr,ilaggr,nr,naggr,diag,theta,nths,info) &
|
||||
!$omp private(icol,val,myth,kk)
|
||||
block
|
||||
integer(psb_ipk_) :: ii,nlp,k,kp,n,ia,isz, nc, i,j,m, nz, ilg, ip, rsz
|
||||
integer(psb_lpk_) :: itmp
|
||||
!$omp master
|
||||
nths = omp_get_num_threads()
|
||||
allocate(bnds(0:nths),locnaggr(0:nths+1))
|
||||
locnaggr(:) = 0
|
||||
bnds(0) = 1
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
myth = omp_get_thread_num()
|
||||
rsz = nr/nths
|
||||
if (myth < mod(nr,nths)) rsz = rsz + 1
|
||||
bnds(myth+1) = rsz
|
||||
!$omp barrier
|
||||
!$omp master
|
||||
do i=1,nths
|
||||
bnds(i) = bnds(i) + bnds(i-1)
|
||||
end do
|
||||
info = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
|
||||
!$omp do schedule(static) private(disjoint)
|
||||
do kk=0, nths-1
|
||||
step1: do ii=bnds(kk), bnds(kk+1)-1
|
||||
i = idxs(ii)
|
||||
if (info /= 0) cycle step1
|
||||
if ((i<1).or.(i>nr)) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if ((nz<0).or.(nz>size(icol))) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
icol(1:nz) = acsr%ja(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
val(1:nz) = acsr%val(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
|
||||
!
|
||||
! Build the set of all strongly coupled nodes
|
||||
!
|
||||
ip = 0
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
! If any of the neighbours is already assigned,
|
||||
! we will not reset.
|
||||
if (j>nr) cycle step1
|
||||
if (ilaggr(j) > 0) cycle step1
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
enddo
|
||||
|
||||
!
|
||||
! If the whole strongly coupled neighborhood of I is
|
||||
! as yet unconnected, turn it into the next aggregate.
|
||||
! Same if ip==0 (in which case, neighborhood only
|
||||
! contains I even if it does not look like it from matrix)
|
||||
! The fact that DISJOINT is private and not under lock
|
||||
! generates a certain un-repeatability, in that between
|
||||
! computing DISJOINT and assigning, another thread might
|
||||
! alter the values of ILAGGR.
|
||||
! However, a certain unrepeatability is already present
|
||||
! because the sequence of aggregates is computed with a
|
||||
! different order than in serial mode.
|
||||
! In any case, even if the enteries of ILAGGR may be
|
||||
! overwritten, the important thing is that each entry is
|
||||
! consistent and they generate a correct aggregation map.
|
||||
!
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
!$omp end atomic
|
||||
cycle step1
|
||||
end if
|
||||
!$omp atomic write
|
||||
ilaggr(i) = itmp
|
||||
!$omp end atomic
|
||||
do k=1, ip
|
||||
!$omp atomic write
|
||||
ilaggr(icol(k)) = itmp
|
||||
!$omp end atomic
|
||||
end do
|
||||
end if
|
||||
end if
|
||||
enddo step1
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
!$omp master
|
||||
naggr = sum(locnaggr(0:nths-1))
|
||||
do i=1,nths
|
||||
locnaggr(i) = locnaggr(i) + locnaggr(i-1)
|
||||
end do
|
||||
do i=nths+1,1,-1
|
||||
locnaggr(i) = locnaggr(i-1)
|
||||
end do
|
||||
locnaggr(0) = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
!$omp do schedule(static)
|
||||
do kk=0, nths-1
|
||||
do ii=bnds(kk), bnds(kk+1)-1
|
||||
if (ilaggr(ii) > 0) then
|
||||
kp = mod(ilaggr(ii),nths)
|
||||
ilaggr(ii) = (ilaggr(ii)/nths)- (bnds(kp)-1) + locnaggr(kp)
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end do
|
||||
end block
|
||||
!$omp end parallel
|
||||
end block
|
||||
if (info /= 0) then
|
||||
if (info == 12345678) write(0,*) 'Overflow in encoding ILAGGR'
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
end if
|
||||
#else
|
||||
step1: do ii=1, nr
|
||||
if (info /= 0) cycle
|
||||
i = idxs(ii)
|
||||
if ((i<1).or.(i>nr)) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if ((nz<0).or.(nz>size(icol))) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
icol(1:nz) = acsr%ja(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
@@ -176,7 +352,7 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
! Build the set of all strongly coupled nodes
|
||||
!
|
||||
ip = 0
|
||||
ip = 0
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
if ((1<=j).and.(j<=nr)) then
|
||||
@@ -194,8 +370,7 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! contains I even if it does not look like it from matrix)
|
||||
!
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
icnt = icnt + 1
|
||||
if (disjoint) then
|
||||
naggr = naggr + 1
|
||||
do k=1, ip
|
||||
ilaggr(icol(k)) = naggr
|
||||
@@ -204,16 +379,22 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
endif
|
||||
enddo step1
|
||||
|
||||
#endif
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1:',count(ilaggr == -(nr+1))
|
||||
& ' Check 1:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc1_p1)
|
||||
if (do_timings) call psb_tic(idx_soc1_p2)
|
||||
!
|
||||
! Phase two: join the neighbours
|
||||
!
|
||||
!$omp workshare
|
||||
tmpaggr = ilaggr
|
||||
!$omp end workshare
|
||||
!$omp parallel do schedule(static) shared(tmpaggr,ilaggr,nr,naggr,diag,theta)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip,cpling)
|
||||
step2: do ii=1,nr
|
||||
i = idxs(ii)
|
||||
|
||||
@@ -244,8 +425,15 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
end if
|
||||
end do step2
|
||||
!$omp end parallel do
|
||||
if (do_timings) call psb_toc(idx_soc1_p2)
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1.5:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p3)
|
||||
!
|
||||
! Phase three: sweep over leftovers, if any
|
||||
!
|
||||
@@ -274,7 +462,6 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
enddo
|
||||
if (ip > 0) then
|
||||
icnt = icnt + 1
|
||||
naggr = naggr + 1
|
||||
ilaggr(i) = naggr
|
||||
do k=1, ip
|
||||
@@ -292,7 +479,10 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end do step3
|
||||
|
||||
! Any leftovers?
|
||||
!$omp parallel do schedule(static) shared(ilaggr,info)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip)
|
||||
do i=1, nr
|
||||
if (info /= 0) cycle
|
||||
if (ilaggr(i) < 0) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if (nz == 1) then
|
||||
@@ -303,15 +493,18 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! other processes.
|
||||
ilaggr(i) = -(nrglob+nr)
|
||||
else
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name,a_err='Fatal error: non-singleton leftovers')
|
||||
goto 9999
|
||||
cycle
|
||||
endif
|
||||
end if
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
if (info /= 0) goto 9999
|
||||
if (do_timings) call psb_toc(idx_soc1_p3)
|
||||
if (naggr > ncol) then
|
||||
!write(0,*) name,'Error : naggr > ncol',naggr,ncol
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Fatal error: naggr>ncol')
|
||||
goto 9999
|
||||
@@ -336,9 +529,13 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nlaggr(:) = 0
|
||||
nlaggr(me+1) = naggr
|
||||
call psb_sum(ctxt,nlaggr(1:np))
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 2:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
call acsr%free()
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
+205
-16
@@ -68,9 +68,12 @@
|
||||
!
|
||||
subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,info)
|
||||
|
||||
use psb_base_mod
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_s_inner_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
|
||||
implicit none
|
||||
|
||||
@@ -99,6 +102,9 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_) :: np, me
|
||||
integer(psb_ipk_) :: nrow, ncol, n_ne
|
||||
character(len=20) :: name, ch_err
|
||||
integer(psb_ipk_), save :: idx_soc2_p1=-1, idx_soc2_p2=-1, idx_soc2_p3=-1
|
||||
integer(psb_ipk_), save :: idx_soc2_p0=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_soc2_map_bld'
|
||||
@@ -114,6 +120,14 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
nrglob = desc_a%get_global_rows()
|
||||
if ((do_timings).and.(idx_soc2_p0==-1)) &
|
||||
& idx_soc2_p0 = psb_get_timer_idx("SOC2_MAP: phase0")
|
||||
if ((do_timings).and.(idx_soc2_p1==-1)) &
|
||||
& idx_soc2_p1 = psb_get_timer_idx("SOC2_MAP: phase1")
|
||||
if ((do_timings).and.(idx_soc2_p2==-1)) &
|
||||
& idx_soc2_p2 = psb_get_timer_idx("SOC2_MAP: phase2")
|
||||
if ((do_timings).and.(idx_soc2_p3==-1)) &
|
||||
& idx_soc2_p3 = psb_get_timer_idx("SOC2_MAP: phase3")
|
||||
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -125,6 +139,7 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc2_p0)
|
||||
diag = a%get_diag(info)
|
||||
if(info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
@@ -137,55 +152,217 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
call a%cp_to(muij)
|
||||
if (clean_zeros) call muij%clean_zeros(info)
|
||||
!$omp parallel do private(i,j,k) shared(nr,diag,muij) schedule(static)
|
||||
do i=1, nr
|
||||
do k=muij%irp(i),muij%irp(i+1)-1
|
||||
j = muij%ja(k)
|
||||
if (j<= nr) muij%val(k) = abs(muij%val(k))/sqrt(abs(diag(i)*diag(j)))
|
||||
end do
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
!
|
||||
! Compute the 1-neigbour; mark strong links with +1, weak links with -1
|
||||
!
|
||||
call s_neigh_coo%allocate(nr,nr,muij%get_nzeros())
|
||||
ip = 0
|
||||
!$omp parallel do private(i,j,k) shared(nr,diag,muij) schedule(static)
|
||||
do i=1, nr
|
||||
do k=muij%irp(i),muij%irp(i+1)-1
|
||||
j = muij%ja(k)
|
||||
s_neigh_coo%ia(k) = i
|
||||
s_neigh_coo%ja(k) = j
|
||||
if (j<=nr) then
|
||||
ip = ip + 1
|
||||
s_neigh_coo%ia(ip) = i
|
||||
s_neigh_coo%ja(ip) = j
|
||||
if (real(muij%val(k)) >= theta) then
|
||||
s_neigh_coo%val(ip) = sone
|
||||
s_neigh_coo%val(k) = sone
|
||||
else
|
||||
s_neigh_coo%val(ip) = -sone
|
||||
s_neigh_coo%val(k) = -sone
|
||||
end if
|
||||
else
|
||||
s_neigh_coo%val(k) = -sone
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end parallel do
|
||||
!write(*,*) 'S_NEIGH: ',nr,ip
|
||||
call s_neigh_coo%set_nzeros(ip)
|
||||
call s_neigh_coo%set_nzeros(muij%get_nzeros())
|
||||
call s_neigh%mv_from_coo(s_neigh_coo,info)
|
||||
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
|
||||
!$omp parallel do private(i) shared(ilaggr,idxs) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
idxs(i) = i
|
||||
end do
|
||||
!$omp end parallel do
|
||||
else
|
||||
!$omp parallel do private(i) shared(ilaggr,idxs,muij) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
ideg(i) = muij%irp(i+1) - muij%irp(i)
|
||||
end do
|
||||
!$omp end parallel do
|
||||
call psb_msort(ideg,ix=idxs,dir=psb_sort_down_)
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc2_p0)
|
||||
if (do_timings) call psb_tic(idx_soc2_p1)
|
||||
|
||||
!
|
||||
! Phase one: Start with disjoint groups.
|
||||
!
|
||||
naggr = 0
|
||||
#if defined(OPENMP)
|
||||
block
|
||||
integer(psb_ipk_), allocatable :: bnds(:), locnaggr(:)
|
||||
integer(psb_ipk_) :: myth,nths, kk
|
||||
! The parallelization makes use of a locaggr(:) array; each thread
|
||||
! keeps its own version of naggr, and when the loop ends, a prefix is applied
|
||||
! to locnaggr to determine:
|
||||
! 1. The total number of aggregaters NAGGR;
|
||||
! 2. How much should each thread shift its own aggregates
|
||||
! Part 2 requires to keep track of which thread defined each entry
|
||||
! of ilaggr(), so that each entry can be adjusted correctly: even
|
||||
! if an entry I belongs to the range BNDS(TH)>BNDS(TH+1)-1, it may have
|
||||
! been set because it is strongly connected to an entry J belonging to a
|
||||
! different thread.
|
||||
|
||||
!$omp parallel shared(s_neigh,bnds,idxs,locnaggr,ilaggr,nr,naggr,diag,theta,nths,info) &
|
||||
!$omp private(icol,val,myth,kk)
|
||||
block
|
||||
integer(psb_ipk_) :: ii,nlp,k,kp,n,ia,isz,nc,i,j,m,nz,ilg,ip,rsz,ip1,nzcnt
|
||||
integer(psb_lpk_) :: itmp
|
||||
!$omp master
|
||||
nths = omp_get_num_threads()
|
||||
allocate(bnds(0:nths),locnaggr(0:nths+1))
|
||||
locnaggr(:) = 0
|
||||
bnds(0) = 1
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
myth = omp_get_thread_num()
|
||||
rsz = nr/nths
|
||||
if (myth < mod(nr,nths)) rsz = rsz + 1
|
||||
bnds(myth+1) = rsz
|
||||
!$omp barrier
|
||||
!$omp master
|
||||
do i=1,nths
|
||||
bnds(i) = bnds(i) + bnds(i-1)
|
||||
end do
|
||||
info = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
|
||||
!$omp do schedule(static) private(disjoint)
|
||||
do kk=0, nths-1
|
||||
step1: do ii=bnds(kk), bnds(kk+1)-1
|
||||
i = idxs(ii)
|
||||
if (info /= 0) then
|
||||
write(0,*) ' Step1:',kk,ii,i,info
|
||||
cycle step1
|
||||
end if
|
||||
if ((i<1).or.(i>nr)) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
!
|
||||
! Get the 1-neighbourhood of I
|
||||
!
|
||||
ip1 = s_neigh%irp(i)
|
||||
nz = s_neigh%irp(i+1)-ip1
|
||||
!
|
||||
! If the neighbourhood only contains I, skip it
|
||||
!
|
||||
if (nz ==0) then
|
||||
ilaggr(i) = 0
|
||||
cycle step1
|
||||
end if
|
||||
if ((nz==1).and.(s_neigh%ja(ip1)==i)) then
|
||||
ilaggr(i) = 0
|
||||
cycle step1
|
||||
end if
|
||||
|
||||
nzcnt = count(real(s_neigh%val(ip1:ip1+nz-1)) > 0)
|
||||
icol(1:nzcnt) = pack(s_neigh%ja(ip1:ip1+nz-1),(real(s_neigh%val(ip1:ip1+nz-1)) > 0))
|
||||
disjoint = all(ilaggr(icol(1:nzcnt)) == -(nr+1))
|
||||
|
||||
!
|
||||
! If the whole strongly coupled neighborhood of I is
|
||||
! as yet unconnected, turn it into the next aggregate.
|
||||
! Same if ip==0 (in which case, neighborhood only
|
||||
! contains I even if it does not look like it from matrix)
|
||||
! The fact that DISJOINT is private and not under lock
|
||||
! generates a certain un-repeatability, in that between
|
||||
! computing DISJOINT and assigning, another thread might
|
||||
! alter the values of ILAGGR.
|
||||
! However, a certain unrepeatability is already present
|
||||
! because the sequence of aggregates is computed with a
|
||||
! different order than in serial mode.
|
||||
! In any case, even if the enteries of ILAGGR may be
|
||||
! overwritten, the important thing is that each entry is
|
||||
! consistent and they generate a correct aggregation map.
|
||||
!
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
!$omp end atomic
|
||||
cycle step1
|
||||
end if
|
||||
!$omp atomic write
|
||||
ilaggr(i) = itmp
|
||||
!$omp end atomic
|
||||
do k=1, nzcnt
|
||||
!$omp atomic write
|
||||
ilaggr(icol(k)) = itmp
|
||||
!$omp end atomic
|
||||
end do
|
||||
end if
|
||||
end if
|
||||
enddo step1
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
!$omp master
|
||||
naggr = sum(locnaggr(0:nths-1))
|
||||
do i=1,nths
|
||||
locnaggr(i) = locnaggr(i) + locnaggr(i-1)
|
||||
end do
|
||||
do i=nths+1,1,-1
|
||||
locnaggr(i) = locnaggr(i-1)
|
||||
end do
|
||||
locnaggr(0) = 0
|
||||
!write(0,*) 'LNAG ',locnaggr(nths+1)
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
!$omp do schedule(static)
|
||||
do kk=0, nths-1
|
||||
do ii=bnds(kk), bnds(kk+1)-1
|
||||
if (ilaggr(ii) > 0) then
|
||||
kp = mod(ilaggr(ii),nths)
|
||||
ilaggr(ii) = (ilaggr(ii)/nths)- (bnds(kp)-1) + locnaggr(kp)
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end do
|
||||
end block
|
||||
!$omp end parallel
|
||||
end block
|
||||
if (info /= 0) then
|
||||
if (info == 12345678) write(0,*) 'Overflow in encoding ILAGGR'
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
#else
|
||||
icnt = 0
|
||||
step1: do ii=1, nr
|
||||
i = idxs(ii)
|
||||
@@ -224,16 +401,21 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
endif
|
||||
enddo step1
|
||||
|
||||
#endif
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1:',count(ilaggr == -(nr+1))
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc2_p1)
|
||||
if (do_timings) call psb_tic(idx_soc2_p2)
|
||||
!
|
||||
! Phase two: join the neighbours
|
||||
!
|
||||
!$omp workshare
|
||||
tmpaggr = ilaggr
|
||||
!$omp end workshare
|
||||
!$omp parallel do schedule(static) shared(tmpaggr,ilaggr,nr,naggr,diag,muij,s_neigh)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip,cpling)
|
||||
step2: do ii=1,nr
|
||||
i = idxs(ii)
|
||||
|
||||
@@ -259,8 +441,9 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
end if
|
||||
end do step2
|
||||
|
||||
|
||||
!$omp end parallel do
|
||||
if (do_timings) call psb_toc(idx_soc2_p2)
|
||||
if (do_timings) call psb_tic(idx_soc2_p3)
|
||||
!
|
||||
! Phase three: sweep over leftovers, if any
|
||||
!
|
||||
@@ -294,6 +477,8 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end do step3
|
||||
|
||||
! Any leftovers?
|
||||
!$omp parallel do schedule(static) shared(ilaggr,s_neigh,info)&
|
||||
!$omp private(ii,i,j,k)
|
||||
do i=1, nr
|
||||
if (ilaggr(i) <= 0) then
|
||||
nz = (s_neigh%irp(i+1)-s_neigh%irp(i))
|
||||
@@ -305,13 +490,17 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! other processes.
|
||||
ilaggr(i) = -(nrglob+nr)
|
||||
else
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name,a_err='Fatal error: non-singleton leftovers')
|
||||
goto 9999
|
||||
cycle
|
||||
endif
|
||||
end if
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
if (info /= 0) goto 9999
|
||||
if (do_timings) call psb_toc(idx_soc2_p3)
|
||||
if (naggr > ncol) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Fatal error: naggr>ncol')
|
||||
@@ -76,7 +76,7 @@ subroutine amg_z_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
integer(psb_ipk_) :: nrow, ncol, nrl, nzl, ip, nzt, i, k
|
||||
integer(psb_lpk_) :: nrsave, ncsave, nzsave, nza
|
||||
logical, parameter :: do_timings=.false., oldstyle=.false., debug=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_cpytrans1=-1, idx_cpytrans2=-1
|
||||
|
||||
name='amg_ptap_bld'
|
||||
if(psb_get_errstatus().ne.0) return
|
||||
@@ -93,7 +93,11 @@ subroutine amg_z_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("SPMM_BLD: par_spspmm")
|
||||
& idx_spspmm = psb_get_timer_idx("PTAP_BLD: par_spspmm")
|
||||
if ((do_timings).and.(idx_cpytrans1==-1)) &
|
||||
& idx_cpytrans1 = psb_get_timer_idx("PTAP_BLD: cpy&trans1")
|
||||
if ((do_timings).and.(idx_cpytrans2==-1)) &
|
||||
& idx_cpytrans2 = psb_get_timer_idx("PTAP_BLD: cpy&trans2")
|
||||
|
||||
naggr = nlaggr(me+1)
|
||||
ntaggr = sum(nlaggr)
|
||||
@@ -128,6 +132,7 @@ subroutine amg_z_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
! Ok first product done.
|
||||
|
||||
if (present(desc_ax)) then
|
||||
if (do_timings) call psb_tic(idx_cpytrans1)
|
||||
block
|
||||
call coo_prol%cp_to_coo(coo_restr,info)
|
||||
call coo_restr%set_ncols(desc_ac%get_local_cols())
|
||||
@@ -137,7 +142,7 @@ subroutine amg_z_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
call coo_restr%set_ncols(desc_ax%get_local_cols())
|
||||
end block
|
||||
call csr_restr%cp_from_coo(coo_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_cpytrans1)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spcnv coo_restr')
|
||||
goto 9999
|
||||
@@ -167,27 +172,28 @@ subroutine amg_z_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call coo_restr%transp()
|
||||
nzl = coo_restr%get_nzeros()
|
||||
nrl = desc_ac%get_local_rows()
|
||||
i=0
|
||||
nrl = desc_ac%get_local_rows()
|
||||
call coo_restr%fix(info)
|
||||
i=coo_restr%get_nzeros()
|
||||
!
|
||||
! Only keep local rows
|
||||
!
|
||||
do k=1, nzl
|
||||
if ((1 <= coo_restr%ia(k)) .and.(coo_restr%ia(k) <= nrl)) then
|
||||
i = i+1
|
||||
coo_restr%val(i) = coo_restr%val(k)
|
||||
coo_restr%ia(i) = coo_restr%ia(k)
|
||||
coo_restr%ja(i) = coo_restr%ja(k)
|
||||
search: do k=i,1,-1
|
||||
if (coo_restr%ia(k) <= nrl) then
|
||||
call coo_restr%set_nzeros(k)
|
||||
exit search
|
||||
end if
|
||||
end do
|
||||
call coo_restr%set_nzeros(i)
|
||||
call coo_restr%fix(info)
|
||||
end do search
|
||||
|
||||
nzl = coo_restr%get_nzeros()
|
||||
call coo_restr%set_nrows(desc_ac%get_local_rows())
|
||||
call coo_restr%set_ncols(desc_a%get_local_cols())
|
||||
if (debug) call check_coo(me,trim(name)//' Check 2 on coo_restr:',coo_restr)
|
||||
if (do_timings) call psb_tic(idx_cpytrans2)
|
||||
|
||||
call csr_restr%cp_from_coo(coo_restr,info)
|
||||
|
||||
if (do_timings) call psb_toc(idx_cpytrans2)
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='spcnv coo_restr')
|
||||
goto 9999
|
||||
|
||||
+218
-21
@@ -72,7 +72,9 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_z_inner_mod
|
||||
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
@@ -85,7 +87,7 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
! Local variables
|
||||
integer(psb_ipk_), allocatable :: ils(:), neigh(:), irow(:), icol(:),&
|
||||
integer(psb_ipk_), allocatable :: neigh(:), irow(:), icol(:),&
|
||||
& ideg(:), idxs(:)
|
||||
integer(psb_lpk_), allocatable :: tmpaggr(:)
|
||||
complex(psb_dpk_), allocatable :: val(:), diag(:)
|
||||
@@ -99,6 +101,9 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_) :: nrow, ncol, n_ne
|
||||
integer(psb_lpk_) :: nrglob
|
||||
character(len=20) :: name, ch_err
|
||||
integer(psb_ipk_), save :: idx_soc1_p1=-1, idx_soc1_p2=-1, idx_soc1_p3=-1
|
||||
integer(psb_ipk_), save :: idx_soc1_p0=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_soc1_map_bld'
|
||||
@@ -114,6 +119,14 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
nrglob = desc_a%get_global_rows()
|
||||
if ((do_timings).and.(idx_soc1_p0==-1)) &
|
||||
& idx_soc1_p0 = psb_get_timer_idx("SOC1_MAP: phase0")
|
||||
if ((do_timings).and.(idx_soc1_p1==-1)) &
|
||||
& idx_soc1_p1 = psb_get_timer_idx("SOC1_MAP: phase1")
|
||||
if ((do_timings).and.(idx_soc1_p2==-1)) &
|
||||
& idx_soc1_p2 = psb_get_timer_idx("SOC1_MAP: phase2")
|
||||
if ((do_timings).and.(idx_soc1_p3==-1)) &
|
||||
& idx_soc1_p3 = psb_get_timer_idx("SOC1_MAP: phase3")
|
||||
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -133,41 +146,204 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p0)
|
||||
call a%cp_to(acsr)
|
||||
if (do_timings) call psb_toc(idx_soc1_p0)
|
||||
if (clean_zeros) call acsr%clean_zeros(info)
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
!$omp parallel do private(i) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
idxs(i) = i
|
||||
idxs(i) = i
|
||||
end do
|
||||
else
|
||||
!$omp end parallel do
|
||||
else
|
||||
!$omp parallel do private(i) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
ideg(i) = acsr%irp(i+1) - acsr%irp(i)
|
||||
end do
|
||||
!$omp end parallel do
|
||||
call psb_msort(ideg,ix=idxs,dir=psb_sort_down_)
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p1)
|
||||
|
||||
!
|
||||
! Phase one: Start with disjoint groups.
|
||||
!
|
||||
naggr = 0
|
||||
icnt = 0
|
||||
#if defined(OPENMP)
|
||||
block
|
||||
integer(psb_ipk_), allocatable :: bnds(:), locnaggr(:)
|
||||
integer(psb_ipk_) :: myth,nths, kk
|
||||
! The parallelization makes use of a locaggr(:) array; each thread
|
||||
! keeps its own version of naggr, and when the loop ends, a prefix is applied
|
||||
! to locnaggr to determine:
|
||||
! 1. The total number of aggregaters NAGGR;
|
||||
! 2. How much should each thread shift its own aggregates
|
||||
! Part 2 requires to keep track of which thread defined each entry
|
||||
! of ilaggr(), so that each entry can be adjusted correctly: even
|
||||
! if an entry I belongs to the range BNDS(TH)>BNDS(TH+1)-1, it may have
|
||||
! been set because it is strongly connected to an entry J belonging to a
|
||||
! different thread.
|
||||
|
||||
!$omp parallel shared(bnds,idxs,locnaggr,ilaggr,nr,naggr,diag,theta,nths,info) &
|
||||
!$omp private(icol,val,myth,kk)
|
||||
block
|
||||
integer(psb_ipk_) :: ii,nlp,k,kp,n,ia,isz, nc, i,j,m, nz, ilg, ip, rsz
|
||||
integer(psb_lpk_) :: itmp
|
||||
!$omp master
|
||||
nths = omp_get_num_threads()
|
||||
allocate(bnds(0:nths),locnaggr(0:nths+1))
|
||||
locnaggr(:) = 0
|
||||
bnds(0) = 1
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
myth = omp_get_thread_num()
|
||||
rsz = nr/nths
|
||||
if (myth < mod(nr,nths)) rsz = rsz + 1
|
||||
bnds(myth+1) = rsz
|
||||
!$omp barrier
|
||||
!$omp master
|
||||
do i=1,nths
|
||||
bnds(i) = bnds(i) + bnds(i-1)
|
||||
end do
|
||||
info = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
|
||||
!$omp do schedule(static) private(disjoint)
|
||||
do kk=0, nths-1
|
||||
step1: do ii=bnds(kk), bnds(kk+1)-1
|
||||
i = idxs(ii)
|
||||
if (info /= 0) cycle step1
|
||||
if ((i<1).or.(i>nr)) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if ((nz<0).or.(nz>size(icol))) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
icol(1:nz) = acsr%ja(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
val(1:nz) = acsr%val(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
|
||||
!
|
||||
! Build the set of all strongly coupled nodes
|
||||
!
|
||||
ip = 0
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
! If any of the neighbours is already assigned,
|
||||
! we will not reset.
|
||||
if (j>nr) cycle step1
|
||||
if (ilaggr(j) > 0) cycle step1
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
enddo
|
||||
|
||||
!
|
||||
! If the whole strongly coupled neighborhood of I is
|
||||
! as yet unconnected, turn it into the next aggregate.
|
||||
! Same if ip==0 (in which case, neighborhood only
|
||||
! contains I even if it does not look like it from matrix)
|
||||
! The fact that DISJOINT is private and not under lock
|
||||
! generates a certain un-repeatability, in that between
|
||||
! computing DISJOINT and assigning, another thread might
|
||||
! alter the values of ILAGGR.
|
||||
! However, a certain unrepeatability is already present
|
||||
! because the sequence of aggregates is computed with a
|
||||
! different order than in serial mode.
|
||||
! In any case, even if the enteries of ILAGGR may be
|
||||
! overwritten, the important thing is that each entry is
|
||||
! consistent and they generate a correct aggregation map.
|
||||
!
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
!$omp end atomic
|
||||
cycle step1
|
||||
end if
|
||||
!$omp atomic write
|
||||
ilaggr(i) = itmp
|
||||
!$omp end atomic
|
||||
do k=1, ip
|
||||
!$omp atomic write
|
||||
ilaggr(icol(k)) = itmp
|
||||
!$omp end atomic
|
||||
end do
|
||||
end if
|
||||
end if
|
||||
enddo step1
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
!$omp master
|
||||
naggr = sum(locnaggr(0:nths-1))
|
||||
do i=1,nths
|
||||
locnaggr(i) = locnaggr(i) + locnaggr(i-1)
|
||||
end do
|
||||
do i=nths+1,1,-1
|
||||
locnaggr(i) = locnaggr(i-1)
|
||||
end do
|
||||
locnaggr(0) = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
!$omp do schedule(static)
|
||||
do kk=0, nths-1
|
||||
do ii=bnds(kk), bnds(kk+1)-1
|
||||
if (ilaggr(ii) > 0) then
|
||||
kp = mod(ilaggr(ii),nths)
|
||||
ilaggr(ii) = (ilaggr(ii)/nths)- (bnds(kp)-1) + locnaggr(kp)
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end do
|
||||
end block
|
||||
!$omp end parallel
|
||||
end block
|
||||
if (info /= 0) then
|
||||
if (info == 12345678) write(0,*) 'Overflow in encoding ILAGGR'
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
end if
|
||||
#else
|
||||
step1: do ii=1, nr
|
||||
if (info /= 0) cycle
|
||||
i = idxs(ii)
|
||||
if ((i<1).or.(i>nr)) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if ((nz<0).or.(nz>size(icol))) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
icol(1:nz) = acsr%ja(acsr%irp(i):acsr%irp(i+1)-1)
|
||||
@@ -176,7 +352,7 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
! Build the set of all strongly coupled nodes
|
||||
!
|
||||
ip = 0
|
||||
ip = 0
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
if ((1<=j).and.(j<=nr)) then
|
||||
@@ -194,8 +370,7 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! contains I even if it does not look like it from matrix)
|
||||
!
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
icnt = icnt + 1
|
||||
if (disjoint) then
|
||||
naggr = naggr + 1
|
||||
do k=1, ip
|
||||
ilaggr(icol(k)) = naggr
|
||||
@@ -204,16 +379,22 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
endif
|
||||
enddo step1
|
||||
|
||||
#endif
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1:',count(ilaggr == -(nr+1))
|
||||
& ' Check 1:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc1_p1)
|
||||
if (do_timings) call psb_tic(idx_soc1_p2)
|
||||
!
|
||||
! Phase two: join the neighbours
|
||||
!
|
||||
!$omp workshare
|
||||
tmpaggr = ilaggr
|
||||
!$omp end workshare
|
||||
!$omp parallel do schedule(static) shared(tmpaggr,ilaggr,nr,naggr,diag,theta)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip,cpling)
|
||||
step2: do ii=1,nr
|
||||
i = idxs(ii)
|
||||
|
||||
@@ -244,8 +425,15 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
end if
|
||||
end do step2
|
||||
!$omp end parallel do
|
||||
if (do_timings) call psb_toc(idx_soc1_p2)
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1.5:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc1_p3)
|
||||
!
|
||||
! Phase three: sweep over leftovers, if any
|
||||
!
|
||||
@@ -274,7 +462,6 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
enddo
|
||||
if (ip > 0) then
|
||||
icnt = icnt + 1
|
||||
naggr = naggr + 1
|
||||
ilaggr(i) = naggr
|
||||
do k=1, ip
|
||||
@@ -292,7 +479,10 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end do step3
|
||||
|
||||
! Any leftovers?
|
||||
!$omp parallel do schedule(static) shared(ilaggr,info)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip)
|
||||
do i=1, nr
|
||||
if (info /= 0) cycle
|
||||
if (ilaggr(i) < 0) then
|
||||
nz = (acsr%irp(i+1)-acsr%irp(i))
|
||||
if (nz == 1) then
|
||||
@@ -303,15 +493,18 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! other processes.
|
||||
ilaggr(i) = -(nrglob+nr)
|
||||
else
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name,a_err='Fatal error: non-singleton leftovers')
|
||||
goto 9999
|
||||
cycle
|
||||
endif
|
||||
end if
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
if (info /= 0) goto 9999
|
||||
if (do_timings) call psb_toc(idx_soc1_p3)
|
||||
if (naggr > ncol) then
|
||||
!write(0,*) name,'Error : naggr > ncol',naggr,ncol
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Fatal error: naggr>ncol')
|
||||
goto 9999
|
||||
@@ -336,9 +529,13 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nlaggr(:) = 0
|
||||
nlaggr(me+1) = naggr
|
||||
call psb_sum(ctxt,nlaggr(1:np))
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 2:',naggr,count(ilaggr(1:nr) == -(nr+1)), count(ilaggr(1:nr)>0),&
|
||||
& count(ilaggr(1:nr) == -(nr+1))+count(ilaggr(1:nr)>0),nr
|
||||
end if
|
||||
|
||||
call acsr%free()
|
||||
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
+205
-16
@@ -68,9 +68,12 @@
|
||||
!
|
||||
subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,info)
|
||||
|
||||
use psb_base_mod
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_z_inner_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
|
||||
implicit none
|
||||
|
||||
@@ -99,6 +102,9 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
integer(psb_ipk_) :: np, me
|
||||
integer(psb_ipk_) :: nrow, ncol, n_ne
|
||||
character(len=20) :: name, ch_err
|
||||
integer(psb_ipk_), save :: idx_soc2_p1=-1, idx_soc2_p2=-1, idx_soc2_p3=-1
|
||||
integer(psb_ipk_), save :: idx_soc2_p0=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
name = 'amg_soc2_map_bld'
|
||||
@@ -114,6 +120,14 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
nrglob = desc_a%get_global_rows()
|
||||
if ((do_timings).and.(idx_soc2_p0==-1)) &
|
||||
& idx_soc2_p0 = psb_get_timer_idx("SOC2_MAP: phase0")
|
||||
if ((do_timings).and.(idx_soc2_p1==-1)) &
|
||||
& idx_soc2_p1 = psb_get_timer_idx("SOC2_MAP: phase1")
|
||||
if ((do_timings).and.(idx_soc2_p2==-1)) &
|
||||
& idx_soc2_p2 = psb_get_timer_idx("SOC2_MAP: phase2")
|
||||
if ((do_timings).and.(idx_soc2_p3==-1)) &
|
||||
& idx_soc2_p3 = psb_get_timer_idx("SOC2_MAP: phase3")
|
||||
|
||||
nr = a%get_nrows()
|
||||
nc = a%get_ncols()
|
||||
@@ -125,6 +139,7 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_tic(idx_soc2_p0)
|
||||
diag = a%get_diag(info)
|
||||
if(info /= psb_success_) then
|
||||
info=psb_err_from_subroutine_
|
||||
@@ -137,55 +152,217 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
call a%cp_to(muij)
|
||||
if (clean_zeros) call muij%clean_zeros(info)
|
||||
!$omp parallel do private(i,j,k) shared(nr,diag,muij) schedule(static)
|
||||
do i=1, nr
|
||||
do k=muij%irp(i),muij%irp(i+1)-1
|
||||
j = muij%ja(k)
|
||||
if (j<= nr) muij%val(k) = abs(muij%val(k))/sqrt(abs(diag(i)*diag(j)))
|
||||
end do
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
!
|
||||
! Compute the 1-neigbour; mark strong links with +1, weak links with -1
|
||||
!
|
||||
call s_neigh_coo%allocate(nr,nr,muij%get_nzeros())
|
||||
ip = 0
|
||||
!$omp parallel do private(i,j,k) shared(nr,diag,muij) schedule(static)
|
||||
do i=1, nr
|
||||
do k=muij%irp(i),muij%irp(i+1)-1
|
||||
j = muij%ja(k)
|
||||
s_neigh_coo%ia(k) = i
|
||||
s_neigh_coo%ja(k) = j
|
||||
if (j<=nr) then
|
||||
ip = ip + 1
|
||||
s_neigh_coo%ia(ip) = i
|
||||
s_neigh_coo%ja(ip) = j
|
||||
if (real(muij%val(k)) >= theta) then
|
||||
s_neigh_coo%val(ip) = done
|
||||
s_neigh_coo%val(k) = done
|
||||
else
|
||||
s_neigh_coo%val(ip) = -done
|
||||
s_neigh_coo%val(k) = -done
|
||||
end if
|
||||
else
|
||||
s_neigh_coo%val(k) = -done
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end parallel do
|
||||
!write(*,*) 'S_NEIGH: ',nr,ip
|
||||
call s_neigh_coo%set_nzeros(ip)
|
||||
call s_neigh_coo%set_nzeros(muij%get_nzeros())
|
||||
call s_neigh%mv_from_coo(s_neigh_coo,info)
|
||||
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
if (iorder == amg_aggr_ord_nat_) then
|
||||
|
||||
!$omp parallel do private(i) shared(ilaggr,idxs) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
idxs(i) = i
|
||||
end do
|
||||
!$omp end parallel do
|
||||
else
|
||||
!$omp parallel do private(i) shared(ilaggr,idxs,muij) schedule(static)
|
||||
do i=1, nr
|
||||
ilaggr(i) = -(nr+1)
|
||||
ideg(i) = muij%irp(i+1) - muij%irp(i)
|
||||
end do
|
||||
!$omp end parallel do
|
||||
call psb_msort(ideg,ix=idxs,dir=psb_sort_down_)
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc2_p0)
|
||||
if (do_timings) call psb_tic(idx_soc2_p1)
|
||||
|
||||
!
|
||||
! Phase one: Start with disjoint groups.
|
||||
!
|
||||
naggr = 0
|
||||
#if defined(OPENMP)
|
||||
block
|
||||
integer(psb_ipk_), allocatable :: bnds(:), locnaggr(:)
|
||||
integer(psb_ipk_) :: myth,nths, kk
|
||||
! The parallelization makes use of a locaggr(:) array; each thread
|
||||
! keeps its own version of naggr, and when the loop ends, a prefix is applied
|
||||
! to locnaggr to determine:
|
||||
! 1. The total number of aggregaters NAGGR;
|
||||
! 2. How much should each thread shift its own aggregates
|
||||
! Part 2 requires to keep track of which thread defined each entry
|
||||
! of ilaggr(), so that each entry can be adjusted correctly: even
|
||||
! if an entry I belongs to the range BNDS(TH)>BNDS(TH+1)-1, it may have
|
||||
! been set because it is strongly connected to an entry J belonging to a
|
||||
! different thread.
|
||||
|
||||
!$omp parallel shared(s_neigh,bnds,idxs,locnaggr,ilaggr,nr,naggr,diag,theta,nths,info) &
|
||||
!$omp private(icol,val,myth,kk)
|
||||
block
|
||||
integer(psb_ipk_) :: ii,nlp,k,kp,n,ia,isz,nc,i,j,m,nz,ilg,ip,rsz,ip1,nzcnt
|
||||
integer(psb_lpk_) :: itmp
|
||||
!$omp master
|
||||
nths = omp_get_num_threads()
|
||||
allocate(bnds(0:nths),locnaggr(0:nths+1))
|
||||
locnaggr(:) = 0
|
||||
bnds(0) = 1
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
myth = omp_get_thread_num()
|
||||
rsz = nr/nths
|
||||
if (myth < mod(nr,nths)) rsz = rsz + 1
|
||||
bnds(myth+1) = rsz
|
||||
!$omp barrier
|
||||
!$omp master
|
||||
do i=1,nths
|
||||
bnds(i) = bnds(i) + bnds(i-1)
|
||||
end do
|
||||
info = 0
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
|
||||
!$omp do schedule(static) private(disjoint)
|
||||
do kk=0, nths-1
|
||||
step1: do ii=bnds(kk), bnds(kk+1)-1
|
||||
i = idxs(ii)
|
||||
if (info /= 0) then
|
||||
write(0,*) ' Step1:',kk,ii,i,info
|
||||
cycle step1
|
||||
end if
|
||||
if ((i<1).or.(i>nr)) then
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name)
|
||||
cycle step1
|
||||
!goto 9999
|
||||
end if
|
||||
|
||||
|
||||
if (ilaggr(i) == -(nr+1)) then
|
||||
!
|
||||
! Get the 1-neighbourhood of I
|
||||
!
|
||||
ip1 = s_neigh%irp(i)
|
||||
nz = s_neigh%irp(i+1)-ip1
|
||||
!
|
||||
! If the neighbourhood only contains I, skip it
|
||||
!
|
||||
if (nz ==0) then
|
||||
ilaggr(i) = 0
|
||||
cycle step1
|
||||
end if
|
||||
if ((nz==1).and.(s_neigh%ja(ip1)==i)) then
|
||||
ilaggr(i) = 0
|
||||
cycle step1
|
||||
end if
|
||||
|
||||
nzcnt = count(real(s_neigh%val(ip1:ip1+nz-1)) > 0)
|
||||
icol(1:nzcnt) = pack(s_neigh%ja(ip1:ip1+nz-1),(real(s_neigh%val(ip1:ip1+nz-1)) > 0))
|
||||
disjoint = all(ilaggr(icol(1:nzcnt)) == -(nr+1))
|
||||
|
||||
!
|
||||
! If the whole strongly coupled neighborhood of I is
|
||||
! as yet unconnected, turn it into the next aggregate.
|
||||
! Same if ip==0 (in which case, neighborhood only
|
||||
! contains I even if it does not look like it from matrix)
|
||||
! The fact that DISJOINT is private and not under lock
|
||||
! generates a certain un-repeatability, in that between
|
||||
! computing DISJOINT and assigning, another thread might
|
||||
! alter the values of ILAGGR.
|
||||
! However, a certain unrepeatability is already present
|
||||
! because the sequence of aggregates is computed with a
|
||||
! different order than in serial mode.
|
||||
! In any case, even if the enteries of ILAGGR may be
|
||||
! overwritten, the important thing is that each entry is
|
||||
! consistent and they generate a correct aggregation map.
|
||||
!
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
!$omp end atomic
|
||||
cycle step1
|
||||
end if
|
||||
!$omp atomic write
|
||||
ilaggr(i) = itmp
|
||||
!$omp end atomic
|
||||
do k=1, nzcnt
|
||||
!$omp atomic write
|
||||
ilaggr(icol(k)) = itmp
|
||||
!$omp end atomic
|
||||
end do
|
||||
end if
|
||||
end if
|
||||
enddo step1
|
||||
end do
|
||||
!$omp end do
|
||||
|
||||
!$omp master
|
||||
naggr = sum(locnaggr(0:nths-1))
|
||||
do i=1,nths
|
||||
locnaggr(i) = locnaggr(i) + locnaggr(i-1)
|
||||
end do
|
||||
do i=nths+1,1,-1
|
||||
locnaggr(i) = locnaggr(i-1)
|
||||
end do
|
||||
locnaggr(0) = 0
|
||||
!write(0,*) 'LNAG ',locnaggr(nths+1)
|
||||
!$omp end master
|
||||
!$omp barrier
|
||||
!$omp do schedule(static)
|
||||
do kk=0, nths-1
|
||||
do ii=bnds(kk), bnds(kk+1)-1
|
||||
if (ilaggr(ii) > 0) then
|
||||
kp = mod(ilaggr(ii),nths)
|
||||
ilaggr(ii) = (ilaggr(ii)/nths)- (bnds(kp)-1) + locnaggr(kp)
|
||||
end if
|
||||
end do
|
||||
end do
|
||||
!$omp end do
|
||||
end block
|
||||
!$omp end parallel
|
||||
end block
|
||||
if (info /= 0) then
|
||||
if (info == 12345678) write(0,*) 'Overflow in encoding ILAGGR'
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name)
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
#else
|
||||
icnt = 0
|
||||
step1: do ii=1, nr
|
||||
i = idxs(ii)
|
||||
@@ -224,16 +401,21 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
endif
|
||||
enddo step1
|
||||
|
||||
#endif
|
||||
if (debug_level >= psb_debug_outer_) then
|
||||
write(debug_unit,*) me,' ',trim(name),&
|
||||
& ' Check 1:',count(ilaggr == -(nr+1))
|
||||
end if
|
||||
|
||||
if (do_timings) call psb_toc(idx_soc2_p1)
|
||||
if (do_timings) call psb_tic(idx_soc2_p2)
|
||||
!
|
||||
! Phase two: join the neighbours
|
||||
!
|
||||
!$omp workshare
|
||||
tmpaggr = ilaggr
|
||||
!$omp end workshare
|
||||
!$omp parallel do schedule(static) shared(tmpaggr,ilaggr,nr,naggr,diag,muij,s_neigh)&
|
||||
!$omp private(ii,i,j,k,nz,icol,val,ip,cpling)
|
||||
step2: do ii=1,nr
|
||||
i = idxs(ii)
|
||||
|
||||
@@ -259,8 +441,9 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end if
|
||||
end if
|
||||
end do step2
|
||||
|
||||
|
||||
!$omp end parallel do
|
||||
if (do_timings) call psb_toc(idx_soc2_p2)
|
||||
if (do_timings) call psb_tic(idx_soc2_p3)
|
||||
!
|
||||
! Phase three: sweep over leftovers, if any
|
||||
!
|
||||
@@ -294,6 +477,8 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
end do step3
|
||||
|
||||
! Any leftovers?
|
||||
!$omp parallel do schedule(static) shared(ilaggr,s_neigh,info)&
|
||||
!$omp private(ii,i,j,k)
|
||||
do i=1, nr
|
||||
if (ilaggr(i) <= 0) then
|
||||
nz = (s_neigh%irp(i+1)-s_neigh%irp(i))
|
||||
@@ -305,13 +490,17 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! other processes.
|
||||
ilaggr(i) = -(nrglob+nr)
|
||||
else
|
||||
!$omp atomic write
|
||||
info=psb_err_internal_error_
|
||||
!$omp end atomic
|
||||
call psb_errpush(info,name,a_err='Fatal error: non-singleton leftovers')
|
||||
goto 9999
|
||||
cycle
|
||||
endif
|
||||
end if
|
||||
end do
|
||||
|
||||
!$omp end parallel do
|
||||
if (info /= 0) goto 9999
|
||||
if (do_timings) call psb_toc(idx_soc2_p3)
|
||||
if (naggr > ncol) then
|
||||
info=psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='Fatal error: naggr>ncol')
|
||||
@@ -17,6 +17,7 @@ amg_c_base_onelev_csetr.o \
|
||||
amg_c_base_onelev_descr.o \
|
||||
amg_c_base_onelev_dump.o \
|
||||
amg_c_base_onelev_free.o \
|
||||
amg_c_base_onelev_free_smoothers.o \
|
||||
amg_c_base_onelev_mat_asb.o \
|
||||
amg_c_base_onelev_setag.o \
|
||||
amg_c_base_onelev_setsm.o \
|
||||
@@ -32,6 +33,7 @@ amg_d_base_onelev_csetr.o \
|
||||
amg_d_base_onelev_descr.o \
|
||||
amg_d_base_onelev_dump.o \
|
||||
amg_d_base_onelev_free.o \
|
||||
amg_d_base_onelev_free_smoothers.o \
|
||||
amg_d_base_onelev_mat_asb.o \
|
||||
amg_d_base_onelev_setag.o \
|
||||
amg_d_base_onelev_setsm.o \
|
||||
@@ -47,6 +49,7 @@ amg_s_base_onelev_csetr.o \
|
||||
amg_s_base_onelev_descr.o \
|
||||
amg_s_base_onelev_dump.o \
|
||||
amg_s_base_onelev_free.o \
|
||||
amg_s_base_onelev_free_smoothers.o \
|
||||
amg_s_base_onelev_mat_asb.o \
|
||||
amg_s_base_onelev_setag.o \
|
||||
amg_s_base_onelev_setsm.o \
|
||||
@@ -62,6 +65,7 @@ amg_z_base_onelev_csetr.o \
|
||||
amg_z_base_onelev_descr.o \
|
||||
amg_z_base_onelev_dump.o \
|
||||
amg_z_base_onelev_free.o \
|
||||
amg_z_base_onelev_free_smoothers.o \
|
||||
amg_z_base_onelev_mat_asb.o \
|
||||
amg_z_base_onelev_setag.o \
|
||||
amg_z_base_onelev_setsm.o \
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
!
|
||||
!
|
||||
! AMG4PSBLAS version 1.0
|
||||
! Algebraic Multigrid Package
|
||||
! based on PSBLAS (Parallel Sparse BLAS version 3.7)
|
||||
!
|
||||
! (C) Copyright 2021
|
||||
!
|
||||
! Salvatore Filippone
|
||||
! Pasqua D'Ambra
|
||||
! Fabio Durastante
|
||||
!
|
||||
! Redistribution and use in source and binary forms, with or without
|
||||
! modification, are permitted provided that the following conditions
|
||||
! are met:
|
||||
! 1. Redistributions of source code must retain the above copyright
|
||||
! notice, this list of conditions and the following disclaimer.
|
||||
! 2. Redistributions in binary form must reproduce the above copyright
|
||||
! notice, this list of conditions, and the following disclaimer in the
|
||||
! documentation and/or other materials provided with the distribution.
|
||||
! 3. The name of the AMG4PSBLAS group or the names of its contributors may
|
||||
! not be used to endorse or promote products derived from this
|
||||
! software without specific written permission.
|
||||
!
|
||||
! THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
! ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
|
||||
! TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
! PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AMG4PSBLAS GROUP OR ITS CONTRIBUTORS
|
||||
! BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
! CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
! SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
! INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
! CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
!
|
||||
subroutine amg_c_base_onelev_free_smoothers(lv,info)
|
||||
|
||||
use psb_base_mod
|
||||
use amg_c_onelev_mod, amg_protect_name => amg_c_base_onelev_free_smoothers
|
||||
implicit none
|
||||
|
||||
class(amg_c_onelev_type), intent(inout) :: lv
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
integer(psb_ipk_) :: i
|
||||
|
||||
info = psb_success_
|
||||
|
||||
! We might just deallocate the top level array, except
|
||||
! that there may be inner objects containing C pointers,
|
||||
! e.g. UMFPACK, SLU or CUDA stuff.
|
||||
! We really need FINALs.
|
||||
if (allocated(lv%sm)) &
|
||||
& call lv%sm%free(info)
|
||||
|
||||
if (allocated(lv%sm2a)) &
|
||||
& call lv%sm2a%free(info)
|
||||
|
||||
end subroutine amg_c_base_onelev_free_smoothers
|
||||
@@ -0,0 +1,60 @@
|
||||
!
|
||||
!
|
||||
! AMG4PSBLAS version 1.0
|
||||
! Algebraic Multigrid Package
|
||||
! based on PSBLAS (Parallel Sparse BLAS version 3.7)
|
||||
!
|
||||
! (C) Copyright 2021
|
||||
!
|
||||
! Salvatore Filippone
|
||||
! Pasqua D'Ambra
|
||||
! Fabio Durastante
|
||||
!
|
||||
! Redistribution and use in source and binary forms, with or without
|
||||
! modification, are permitted provided that the following conditions
|
||||
! are met:
|
||||
! 1. Redistributions of source code must retain the above copyright
|
||||
! notice, this list of conditions and the following disclaimer.
|
||||
! 2. Redistributions in binary form must reproduce the above copyright
|
||||
! notice, this list of conditions, and the following disclaimer in the
|
||||
! documentation and/or other materials provided with the distribution.
|
||||
! 3. The name of the AMG4PSBLAS group or the names of its contributors may
|
||||
! not be used to endorse or promote products derived from this
|
||||
! software without specific written permission.
|
||||
!
|
||||
! THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
! ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
|
||||
! TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
! PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AMG4PSBLAS GROUP OR ITS CONTRIBUTORS
|
||||
! BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
! CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
! SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
! INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
! CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
!
|
||||
subroutine amg_d_base_onelev_free_smoothers(lv,info)
|
||||
|
||||
use psb_base_mod
|
||||
use amg_d_onelev_mod, amg_protect_name => amg_d_base_onelev_free_smoothers
|
||||
implicit none
|
||||
|
||||
class(amg_d_onelev_type), intent(inout) :: lv
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
integer(psb_ipk_) :: i
|
||||
|
||||
info = psb_success_
|
||||
|
||||
! We might just deallocate the top level array, except
|
||||
! that there may be inner objects containing C pointers,
|
||||
! e.g. UMFPACK, SLU or CUDA stuff.
|
||||
! We really need FINALs.
|
||||
if (allocated(lv%sm)) &
|
||||
& call lv%sm%free(info)
|
||||
|
||||
if (allocated(lv%sm2a)) &
|
||||
& call lv%sm2a%free(info)
|
||||
|
||||
end subroutine amg_d_base_onelev_free_smoothers
|
||||
@@ -0,0 +1,60 @@
|
||||
!
|
||||
!
|
||||
! AMG4PSBLAS version 1.0
|
||||
! Algebraic Multigrid Package
|
||||
! based on PSBLAS (Parallel Sparse BLAS version 3.7)
|
||||
!
|
||||
! (C) Copyright 2021
|
||||
!
|
||||
! Salvatore Filippone
|
||||
! Pasqua D'Ambra
|
||||
! Fabio Durastante
|
||||
!
|
||||
! Redistribution and use in source and binary forms, with or without
|
||||
! modification, are permitted provided that the following conditions
|
||||
! are met:
|
||||
! 1. Redistributions of source code must retain the above copyright
|
||||
! notice, this list of conditions and the following disclaimer.
|
||||
! 2. Redistributions in binary form must reproduce the above copyright
|
||||
! notice, this list of conditions, and the following disclaimer in the
|
||||
! documentation and/or other materials provided with the distribution.
|
||||
! 3. The name of the AMG4PSBLAS group or the names of its contributors may
|
||||
! not be used to endorse or promote products derived from this
|
||||
! software without specific written permission.
|
||||
!
|
||||
! THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
! ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
|
||||
! TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
! PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AMG4PSBLAS GROUP OR ITS CONTRIBUTORS
|
||||
! BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
! CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
! SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
! INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
! CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
!
|
||||
subroutine amg_s_base_onelev_free_smoothers(lv,info)
|
||||
|
||||
use psb_base_mod
|
||||
use amg_s_onelev_mod, amg_protect_name => amg_s_base_onelev_free_smoothers
|
||||
implicit none
|
||||
|
||||
class(amg_s_onelev_type), intent(inout) :: lv
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
integer(psb_ipk_) :: i
|
||||
|
||||
info = psb_success_
|
||||
|
||||
! We might just deallocate the top level array, except
|
||||
! that there may be inner objects containing C pointers,
|
||||
! e.g. UMFPACK, SLU or CUDA stuff.
|
||||
! We really need FINALs.
|
||||
if (allocated(lv%sm)) &
|
||||
& call lv%sm%free(info)
|
||||
|
||||
if (allocated(lv%sm2a)) &
|
||||
& call lv%sm2a%free(info)
|
||||
|
||||
end subroutine amg_s_base_onelev_free_smoothers
|
||||
@@ -0,0 +1,60 @@
|
||||
!
|
||||
!
|
||||
! AMG4PSBLAS version 1.0
|
||||
! Algebraic Multigrid Package
|
||||
! based on PSBLAS (Parallel Sparse BLAS version 3.7)
|
||||
!
|
||||
! (C) Copyright 2021
|
||||
!
|
||||
! Salvatore Filippone
|
||||
! Pasqua D'Ambra
|
||||
! Fabio Durastante
|
||||
!
|
||||
! Redistribution and use in source and binary forms, with or without
|
||||
! modification, are permitted provided that the following conditions
|
||||
! are met:
|
||||
! 1. Redistributions of source code must retain the above copyright
|
||||
! notice, this list of conditions and the following disclaimer.
|
||||
! 2. Redistributions in binary form must reproduce the above copyright
|
||||
! notice, this list of conditions, and the following disclaimer in the
|
||||
! documentation and/or other materials provided with the distribution.
|
||||
! 3. The name of the AMG4PSBLAS group or the names of its contributors may
|
||||
! not be used to endorse or promote products derived from this
|
||||
! software without specific written permission.
|
||||
!
|
||||
! THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
! ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
|
||||
! TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
! PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AMG4PSBLAS GROUP OR ITS CONTRIBUTORS
|
||||
! BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||
! CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||
! SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||
! INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||
! CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
!
|
||||
subroutine amg_z_base_onelev_free_smoothers(lv,info)
|
||||
|
||||
use psb_base_mod
|
||||
use amg_z_onelev_mod, amg_protect_name => amg_z_base_onelev_free_smoothers
|
||||
implicit none
|
||||
|
||||
class(amg_z_onelev_type), intent(inout) :: lv
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
integer(psb_ipk_) :: i
|
||||
|
||||
info = psb_success_
|
||||
|
||||
! We might just deallocate the top level array, except
|
||||
! that there may be inner objects containing C pointers,
|
||||
! e.g. UMFPACK, SLU or CUDA stuff.
|
||||
! We really need FINALs.
|
||||
if (allocated(lv%sm)) &
|
||||
& call lv%sm%free(info)
|
||||
|
||||
if (allocated(lv%sm2a)) &
|
||||
& call lv%sm2a%free(info)
|
||||
|
||||
end subroutine amg_z_base_onelev_free_smoothers
|
||||
@@ -56,6 +56,8 @@ subroutine amg_c_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='d_bwgs_solver_bld', ch_err
|
||||
integer(psb_ipk_), save :: idx_tril=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -65,6 +67,8 @@ subroutine amg_c_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
call psb_info(ctxt, me, np)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),' start'
|
||||
if ((do_timings).and.(idx_tril==-1)) &
|
||||
& idx_tril = psb_get_timer_idx("BWGS_BLD: tril")
|
||||
|
||||
|
||||
n_row = desc_a%get_local_rows()
|
||||
@@ -77,7 +81,10 @@ subroutine amg_c_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
! This cuts out the off-diagonal part, because it's supposed to
|
||||
! be handled by the outer Jacobi smoother.
|
||||
!
|
||||
!write(0,*) 'Calling A%TRIL in bwgs_solver_bld'
|
||||
if (do_timings) call psb_tic(idx_tril)
|
||||
call a%tril(sv%l,info,diag=-ione,jmax=nrow_a,u=sv%u)
|
||||
if (do_timings) call psb_toc(idx_tril)
|
||||
|
||||
else
|
||||
|
||||
|
||||
@@ -56,6 +56,8 @@ subroutine amg_c_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='c_gs_solver_bld', ch_err
|
||||
integer(psb_ipk_), save :: idx_tril=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -65,6 +67,8 @@ subroutine amg_c_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
call psb_info(ctxt, me, np)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),' start'
|
||||
if ((do_timings).and.(idx_tril==-1)) &
|
||||
& idx_tril = psb_get_timer_idx("GS_BLD: tril")
|
||||
|
||||
|
||||
n_row = desc_a%get_local_rows()
|
||||
@@ -76,9 +80,12 @@ subroutine amg_c_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
!
|
||||
! This cuts out the off-diagonal part, because it's supposed to
|
||||
! be handled by the outer Jacobi smoother.
|
||||
!
|
||||
!
|
||||
!write(0,*) 'Calling A%TRIL in gs_solver_bld'
|
||||
if (do_timings) call psb_tic(idx_tril)
|
||||
call a%tril(sv%l,info,diag=izero,jmax=nrow_a,u=sv%u)
|
||||
|
||||
if (do_timings) call psb_toc(idx_tril)
|
||||
!write(0,*) 'From A%TRIL in gs_solver_bld',a%get_nzeros(),sv%l%get_nzeros(),sv%u%get_nzeros()
|
||||
else
|
||||
|
||||
info = psb_err_missing_override_method_
|
||||
|
||||
@@ -56,6 +56,8 @@ subroutine amg_d_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='d_bwgs_solver_bld', ch_err
|
||||
integer(psb_ipk_), save :: idx_tril=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -65,6 +67,8 @@ subroutine amg_d_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
call psb_info(ctxt, me, np)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),' start'
|
||||
if ((do_timings).and.(idx_tril==-1)) &
|
||||
& idx_tril = psb_get_timer_idx("BWGS_BLD: tril")
|
||||
|
||||
|
||||
n_row = desc_a%get_local_rows()
|
||||
@@ -77,7 +81,10 @@ subroutine amg_d_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
! This cuts out the off-diagonal part, because it's supposed to
|
||||
! be handled by the outer Jacobi smoother.
|
||||
!
|
||||
!write(0,*) 'Calling A%TRIL in bwgs_solver_bld'
|
||||
if (do_timings) call psb_tic(idx_tril)
|
||||
call a%tril(sv%l,info,diag=-ione,jmax=nrow_a,u=sv%u)
|
||||
if (do_timings) call psb_toc(idx_tril)
|
||||
|
||||
else
|
||||
|
||||
|
||||
@@ -56,6 +56,8 @@ subroutine amg_d_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='d_gs_solver_bld', ch_err
|
||||
integer(psb_ipk_), save :: idx_tril=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -65,6 +67,8 @@ subroutine amg_d_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
call psb_info(ctxt, me, np)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),' start'
|
||||
if ((do_timings).and.(idx_tril==-1)) &
|
||||
& idx_tril = psb_get_timer_idx("GS_BLD: tril")
|
||||
|
||||
|
||||
n_row = desc_a%get_local_rows()
|
||||
@@ -76,9 +80,12 @@ subroutine amg_d_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
!
|
||||
! This cuts out the off-diagonal part, because it's supposed to
|
||||
! be handled by the outer Jacobi smoother.
|
||||
!
|
||||
!
|
||||
!write(0,*) 'Calling A%TRIL in gs_solver_bld'
|
||||
if (do_timings) call psb_tic(idx_tril)
|
||||
call a%tril(sv%l,info,diag=izero,jmax=nrow_a,u=sv%u)
|
||||
|
||||
if (do_timings) call psb_toc(idx_tril)
|
||||
!write(0,*) 'From A%TRIL in gs_solver_bld',a%get_nzeros(),sv%l%get_nzeros(),sv%u%get_nzeros()
|
||||
else
|
||||
|
||||
info = psb_err_missing_override_method_
|
||||
|
||||
@@ -56,6 +56,8 @@ subroutine amg_s_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='d_bwgs_solver_bld', ch_err
|
||||
integer(psb_ipk_), save :: idx_tril=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -65,6 +67,8 @@ subroutine amg_s_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
call psb_info(ctxt, me, np)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),' start'
|
||||
if ((do_timings).and.(idx_tril==-1)) &
|
||||
& idx_tril = psb_get_timer_idx("BWGS_BLD: tril")
|
||||
|
||||
|
||||
n_row = desc_a%get_local_rows()
|
||||
@@ -77,7 +81,10 @@ subroutine amg_s_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
! This cuts out the off-diagonal part, because it's supposed to
|
||||
! be handled by the outer Jacobi smoother.
|
||||
!
|
||||
!write(0,*) 'Calling A%TRIL in bwgs_solver_bld'
|
||||
if (do_timings) call psb_tic(idx_tril)
|
||||
call a%tril(sv%l,info,diag=-ione,jmax=nrow_a,u=sv%u)
|
||||
if (do_timings) call psb_toc(idx_tril)
|
||||
|
||||
else
|
||||
|
||||
|
||||
@@ -56,6 +56,8 @@ subroutine amg_s_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='s_gs_solver_bld', ch_err
|
||||
integer(psb_ipk_), save :: idx_tril=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -65,6 +67,8 @@ subroutine amg_s_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
call psb_info(ctxt, me, np)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),' start'
|
||||
if ((do_timings).and.(idx_tril==-1)) &
|
||||
& idx_tril = psb_get_timer_idx("GS_BLD: tril")
|
||||
|
||||
|
||||
n_row = desc_a%get_local_rows()
|
||||
@@ -76,9 +80,12 @@ subroutine amg_s_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
!
|
||||
! This cuts out the off-diagonal part, because it's supposed to
|
||||
! be handled by the outer Jacobi smoother.
|
||||
!
|
||||
!
|
||||
!write(0,*) 'Calling A%TRIL in gs_solver_bld'
|
||||
if (do_timings) call psb_tic(idx_tril)
|
||||
call a%tril(sv%l,info,diag=izero,jmax=nrow_a,u=sv%u)
|
||||
|
||||
if (do_timings) call psb_toc(idx_tril)
|
||||
!write(0,*) 'From A%TRIL in gs_solver_bld',a%get_nzeros(),sv%l%get_nzeros(),sv%u%get_nzeros()
|
||||
else
|
||||
|
||||
info = psb_err_missing_override_method_
|
||||
|
||||
@@ -56,6 +56,8 @@ subroutine amg_z_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='d_bwgs_solver_bld', ch_err
|
||||
integer(psb_ipk_), save :: idx_tril=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -65,6 +67,8 @@ subroutine amg_z_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
call psb_info(ctxt, me, np)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),' start'
|
||||
if ((do_timings).and.(idx_tril==-1)) &
|
||||
& idx_tril = psb_get_timer_idx("BWGS_BLD: tril")
|
||||
|
||||
|
||||
n_row = desc_a%get_local_rows()
|
||||
@@ -77,7 +81,10 @@ subroutine amg_z_bwgs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
! This cuts out the off-diagonal part, because it's supposed to
|
||||
! be handled by the outer Jacobi smoother.
|
||||
!
|
||||
!write(0,*) 'Calling A%TRIL in bwgs_solver_bld'
|
||||
if (do_timings) call psb_tic(idx_tril)
|
||||
call a%tril(sv%l,info,diag=-ione,jmax=nrow_a,u=sv%u)
|
||||
if (do_timings) call psb_toc(idx_tril)
|
||||
|
||||
else
|
||||
|
||||
|
||||
@@ -56,6 +56,8 @@ subroutine amg_z_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='z_gs_solver_bld', ch_err
|
||||
integer(psb_ipk_), save :: idx_tril=-1
|
||||
logical, parameter :: do_timings=.true.
|
||||
|
||||
info=psb_success_
|
||||
call psb_erractionsave(err_act)
|
||||
@@ -65,6 +67,8 @@ subroutine amg_z_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
call psb_info(ctxt, me, np)
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),' start'
|
||||
if ((do_timings).and.(idx_tril==-1)) &
|
||||
& idx_tril = psb_get_timer_idx("GS_BLD: tril")
|
||||
|
||||
|
||||
n_row = desc_a%get_local_rows()
|
||||
@@ -76,9 +80,12 @@ subroutine amg_z_gs_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
!
|
||||
! This cuts out the off-diagonal part, because it's supposed to
|
||||
! be handled by the outer Jacobi smoother.
|
||||
!
|
||||
!
|
||||
!write(0,*) 'Calling A%TRIL in gs_solver_bld'
|
||||
if (do_timings) call psb_tic(idx_tril)
|
||||
call a%tril(sv%l,info,diag=izero,jmax=nrow_a,u=sv%u)
|
||||
|
||||
if (do_timings) call psb_toc(idx_tril)
|
||||
!write(0,*) 'From A%TRIL in gs_solver_bld',a%get_nzeros(),sv%l%get_nzeros(),sv%u%get_nzeros()
|
||||
else
|
||||
|
||||
info = psb_err_missing_override_method_
|
||||
|
||||
@@ -73,6 +73,9 @@ program amg_d_pde2d
|
||||
use amg_d_pde2d_exp_mod
|
||||
use amg_d_pde2d_box_mod
|
||||
use amg_d_genpde_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! input parameters
|
||||
@@ -93,7 +96,7 @@ program amg_d_pde2d
|
||||
type(psb_d_vect_type) :: x,b,r
|
||||
! parallel environment
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: iam, np
|
||||
integer(psb_ipk_) :: iam, np, nth
|
||||
|
||||
! solver parameters
|
||||
integer(psb_ipk_) :: iter, itmax,itrace, istopc, irst, nlv
|
||||
@@ -197,6 +200,15 @@ program amg_d_pde2d
|
||||
|
||||
call psb_init(ctxt)
|
||||
call psb_info(ctxt,iam,np)
|
||||
#if defined(OPENMP)
|
||||
!$OMP parallel shared(nth)
|
||||
!$OMP master
|
||||
nth = omp_get_num_threads()
|
||||
!$OMP end master
|
||||
!$OMP end parallel
|
||||
#else
|
||||
nth = 1
|
||||
#endif
|
||||
|
||||
if (iam < 0) then
|
||||
! This should not happen, but just in case
|
||||
@@ -451,12 +463,14 @@ program amg_d_pde2d
|
||||
call psb_sum(ctxt,precsize)
|
||||
call prec%descr(info,iout=psb_out_unit)
|
||||
if (iam == psb_root_) then
|
||||
write(psb_out_unit,'("Computed solution on ",i8," processors")') np
|
||||
write(psb_out_unit,'("Computed solution on ",i8," process(es)")') np
|
||||
write(psb_out_unit,'("Number of threads : ",i12)') nth
|
||||
write(psb_out_unit,'("Total number of tasks : ",i12)') nth*np
|
||||
write(psb_out_unit,'("Linear system size : ",i12)') system_size
|
||||
write(psb_out_unit,'("PDE Coefficients : ",a)') trim(pdecoeff)
|
||||
write(psb_out_unit,'("Krylov method : ",a)') trim(s_choice%kmethd)
|
||||
write(psb_out_unit,'("Preconditioner : ",a)') trim(p_choice%descr)
|
||||
write(psb_out_unit,'("Iterations to convergence : ",i12)') iter
|
||||
write(psb_out_unit,'("PDE Coefficients : ",a)') trim(pdecoeff)
|
||||
write(psb_out_unit,'("Krylov method : ",a)') trim(s_choice%kmethd)
|
||||
write(psb_out_unit,'("Preconditioner : ",a)') trim(p_choice%descr)
|
||||
write(psb_out_unit,'("Iterations to convergence : ",i12)') iter
|
||||
write(psb_out_unit,'("Relative error estimate on exit : ",es12.5)') err
|
||||
write(psb_out_unit,'("Number of levels in hierarchy : ",i12)') prec%get_nlevs()
|
||||
write(psb_out_unit,'("Time to build hierarchy : ",es12.5)') thier
|
||||
@@ -74,6 +74,9 @@ program amg_d_pde3d
|
||||
use amg_d_pde3d_exp_mod
|
||||
use amg_d_pde3d_gauss_mod
|
||||
use amg_d_genpde_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! input parameters
|
||||
@@ -94,7 +97,7 @@ program amg_d_pde3d
|
||||
type(psb_d_vect_type) :: x,b,r
|
||||
! parallel environment
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: iam, np
|
||||
integer(psb_ipk_) :: iam, np, nth
|
||||
|
||||
! solver parameters
|
||||
integer(psb_ipk_) :: iter, itmax,itrace, istopc, irst, nlv
|
||||
@@ -192,12 +195,21 @@ program amg_d_pde3d
|
||||
! other variables
|
||||
integer(psb_ipk_) :: info, i, k
|
||||
character(len=20) :: name,ch_err
|
||||
|
||||
type(psb_d_csr_sparse_mat) :: amold
|
||||
info=psb_success_
|
||||
|
||||
|
||||
call psb_init(ctxt)
|
||||
call psb_info(ctxt,iam,np)
|
||||
#if defined(OPENMP)
|
||||
!$OMP parallel shared(nth)
|
||||
!$OMP master
|
||||
nth = omp_get_num_threads()
|
||||
!$OMP end master
|
||||
!$OMP end parallel
|
||||
#else
|
||||
nth = 1
|
||||
#endif
|
||||
|
||||
if (iam < 0) then
|
||||
! This should not happen, but just in case
|
||||
@@ -390,7 +402,7 @@ program amg_d_pde3d
|
||||
end if
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
call prec%smoothers_build(a,desc_a,info)
|
||||
call prec%smoothers_build(a,desc_a,info,amold=amold)
|
||||
tprec = psb_wtime()-t1
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='amg_smoothers_bld')
|
||||
@@ -455,7 +467,9 @@ program amg_d_pde3d
|
||||
call psb_sum(ctxt,precsize)
|
||||
call prec%descr(info,iout=psb_out_unit)
|
||||
if (iam == psb_root_) then
|
||||
write(psb_out_unit,'("Computed solution on ",i8," processors")') np
|
||||
write(psb_out_unit,'("Computed solution on ",i8," process(es)")') np
|
||||
write(psb_out_unit,'("Number of threads : ",i12)') nth
|
||||
write(psb_out_unit,'("Total number of tasks : ",i12)') nth*np
|
||||
write(psb_out_unit,'("Linear system size : ",i12)') system_size
|
||||
write(psb_out_unit,'("PDE Coefficients : ",a)') trim(pdecoeff)
|
||||
write(psb_out_unit,'("Krylov method : ",a)') trim(s_choice%kmethd)
|
||||
@@ -478,7 +492,7 @@ program amg_d_pde3d
|
||||
write(psb_out_unit,'("Storage format for DESC_A : ",a )') desc_a%get_fmt()
|
||||
|
||||
end if
|
||||
|
||||
call psb_print_timers(ctxt)
|
||||
!
|
||||
! cleanup storage and exit
|
||||
!
|
||||
@@ -73,6 +73,9 @@ program amg_s_pde2d
|
||||
use amg_s_pde2d_exp_mod
|
||||
use amg_s_pde2d_box_mod
|
||||
use amg_s_genpde_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! input parameters
|
||||
@@ -93,7 +96,7 @@ program amg_s_pde2d
|
||||
type(psb_s_vect_type) :: x,b,r
|
||||
! parallel environment
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: iam, np
|
||||
integer(psb_ipk_) :: iam, np, nth
|
||||
|
||||
! solver parameters
|
||||
integer(psb_ipk_) :: iter, itmax,itrace, istopc, irst, nlv
|
||||
@@ -197,6 +200,15 @@ program amg_s_pde2d
|
||||
|
||||
call psb_init(ctxt)
|
||||
call psb_info(ctxt,iam,np)
|
||||
#if defined(OPENMP)
|
||||
!$OMP parallel shared(nth)
|
||||
!$OMP master
|
||||
nth = omp_get_num_threads()
|
||||
!$OMP end master
|
||||
!$OMP end parallel
|
||||
#else
|
||||
nth = 1
|
||||
#endif
|
||||
|
||||
if (iam < 0) then
|
||||
! This should not happen, but just in case
|
||||
@@ -451,12 +463,14 @@ program amg_s_pde2d
|
||||
call psb_sum(ctxt,precsize)
|
||||
call prec%descr(info,iout=psb_out_unit)
|
||||
if (iam == psb_root_) then
|
||||
write(psb_out_unit,'("Computed solution on ",i8," processors")') np
|
||||
write(psb_out_unit,'("Computed solution on ",i8," process(es)")') np
|
||||
write(psb_out_unit,'("Number of threads : ",i12)') nth
|
||||
write(psb_out_unit,'("Total number of tasks : ",i12)') nth*np
|
||||
write(psb_out_unit,'("Linear system size : ",i12)') system_size
|
||||
write(psb_out_unit,'("PDE Coefficients : ",a)') trim(pdecoeff)
|
||||
write(psb_out_unit,'("Krylov method : ",a)') trim(s_choice%kmethd)
|
||||
write(psb_out_unit,'("Preconditioner : ",a)') trim(p_choice%descr)
|
||||
write(psb_out_unit,'("Iterations to convergence : ",i12)') iter
|
||||
write(psb_out_unit,'("PDE Coefficients : ",a)') trim(pdecoeff)
|
||||
write(psb_out_unit,'("Krylov method : ",a)') trim(s_choice%kmethd)
|
||||
write(psb_out_unit,'("Preconditioner : ",a)') trim(p_choice%descr)
|
||||
write(psb_out_unit,'("Iterations to convergence : ",i12)') iter
|
||||
write(psb_out_unit,'("Relative error estimate on exit : ",es12.5)') err
|
||||
write(psb_out_unit,'("Number of levels in hierarchy : ",i12)') prec%get_nlevs()
|
||||
write(psb_out_unit,'("Time to build hierarchy : ",es12.5)') thier
|
||||
@@ -74,6 +74,9 @@ program amg_s_pde3d
|
||||
use amg_s_pde3d_exp_mod
|
||||
use amg_s_pde3d_gauss_mod
|
||||
use amg_s_genpde_mod
|
||||
#if defined(OPENMP)
|
||||
use omp_lib
|
||||
#endif
|
||||
implicit none
|
||||
|
||||
! input parameters
|
||||
@@ -94,7 +97,7 @@ program amg_s_pde3d
|
||||
type(psb_s_vect_type) :: x,b,r
|
||||
! parallel environment
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: iam, np
|
||||
integer(psb_ipk_) :: iam, np, nth
|
||||
|
||||
! solver parameters
|
||||
integer(psb_ipk_) :: iter, itmax,itrace, istopc, irst, nlv
|
||||
@@ -192,12 +195,21 @@ program amg_s_pde3d
|
||||
! other variables
|
||||
integer(psb_ipk_) :: info, i, k
|
||||
character(len=20) :: name,ch_err
|
||||
|
||||
type(psb_s_csr_sparse_mat) :: amold
|
||||
info=psb_success_
|
||||
|
||||
|
||||
call psb_init(ctxt)
|
||||
call psb_info(ctxt,iam,np)
|
||||
#if defined(OPENMP)
|
||||
!$OMP parallel shared(nth)
|
||||
!$OMP master
|
||||
nth = omp_get_num_threads()
|
||||
!$OMP end master
|
||||
!$OMP end parallel
|
||||
#else
|
||||
nth = 1
|
||||
#endif
|
||||
|
||||
if (iam < 0) then
|
||||
! This should not happen, but just in case
|
||||
@@ -390,7 +402,7 @@ program amg_s_pde3d
|
||||
end if
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
call prec%smoothers_build(a,desc_a,info)
|
||||
call prec%smoothers_build(a,desc_a,info,amold=amold)
|
||||
tprec = psb_wtime()-t1
|
||||
if (info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='amg_smoothers_bld')
|
||||
@@ -455,7 +467,9 @@ program amg_s_pde3d
|
||||
call psb_sum(ctxt,precsize)
|
||||
call prec%descr(info,iout=psb_out_unit)
|
||||
if (iam == psb_root_) then
|
||||
write(psb_out_unit,'("Computed solution on ",i8," processors")') np
|
||||
write(psb_out_unit,'("Computed solution on ",i8," process(es)")') np
|
||||
write(psb_out_unit,'("Number of threads : ",i12)') nth
|
||||
write(psb_out_unit,'("Total number of tasks : ",i12)') nth*np
|
||||
write(psb_out_unit,'("Linear system size : ",i12)') system_size
|
||||
write(psb_out_unit,'("PDE Coefficients : ",a)') trim(pdecoeff)
|
||||
write(psb_out_unit,'("Krylov method : ",a)') trim(s_choice%kmethd)
|
||||
@@ -478,7 +492,7 @@ program amg_s_pde3d
|
||||
write(psb_out_unit,'("Storage format for DESC_A : ",a )') desc_a%get_fmt()
|
||||
|
||||
end if
|
||||
|
||||
call psb_print_timers(ctxt)
|
||||
!
|
||||
! cleanup storage and exit
|
||||
!
|
||||
@@ -1,6 +1,6 @@
|
||||
%%%%%%%%%%% General arguments % Lines starting with % are ignored.
|
||||
CSR ! Storage format CSR COO JAD
|
||||
0200 ! IDIM; domain size. Linear system size is IDIM**3
|
||||
0200 ! IDIM; domain size. Linear system size is IDIM**3
|
||||
CONST ! PDECOEFF: CONST, EXP, GAUSS Coefficients of the PDE
|
||||
BICGSTAB ! Iterative method: BiCGSTAB BiCGSTABL BiCG CG CGS FCG GCR RGMRES
|
||||
2 ! ISTOPC
|
||||
@@ -9,7 +9,7 @@ BICGSTAB ! Iterative method: BiCGSTAB BiCGSTABL BiCG CG CGS F
|
||||
30 ! IRST (restart for RGMRES and BiCGSTABL)
|
||||
1.d-6 ! EPS
|
||||
%%%%%%%%%%% Main preconditioner choices %%%%%%%%%%%%%%%%
|
||||
ML-VCYCLE-BJAC-D-BJAC ! Longer descriptive name for preconditioner (up to 20 chars)
|
||||
ML-VBM-VCYCLE-FBGS-D-BJAC ! Longer descriptive name for preconditioner (up to 20 chars)
|
||||
ML ! Preconditioner type: NONE JACOBI GS FBGS BJAC AS ML
|
||||
%%%%%%%%%%% First smoother (for all levels but coarsest) %%%%%%%%%%%%%%%%
|
||||
FBGS ! Smoother type JACOBI FBGS GS BWGS BJAC AS. For 1-level, repeats previous.
|
||||
@@ -39,8 +39,8 @@ VCYCLE ! Type of multilevel CYCLE: VCYCLE WCYCLE KCYCLE MUL
|
||||
-3 ! Max Number of levels in a multilevel preconditioner; if <0, lib default
|
||||
-3 ! Target coarse matrix size per process; if <0, lib default
|
||||
SMOOTHED ! Type of aggregation: SMOOTHED UNSMOOTHED
|
||||
COUPLED ! Parallel aggregation: DEC, SYMDEC, COUPLED
|
||||
MATCHBOXP ! aggregation measure SOC1, MATCHBOXP
|
||||
DEC ! Parallel aggregation: DEC, SYMDEC, COUPLED
|
||||
SOC1 ! aggregation measure SOC1, MATCHBOXP
|
||||
8 ! Requested size of the aggregates for MATCHBOXP
|
||||
NATURAL ! Ordering of aggregation NATURAL DEGREE
|
||||
-1.5 ! Coarsening ratio, if < 0 use library default
|
||||
|
||||
Reference in New Issue
Block a user