mirror of
https://github.com/sfilippone/amg4psblas.git
synced 2026-10-06 22:55:12 +00:00
Compare commits
86
Commits
Poly-novrl
...
repackage
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1aef82023c | ||
|
|
161b93da64 | ||
|
|
8d4af1ba9f | ||
|
|
cd87bdb0c1 | ||
|
|
53afa87814 | ||
|
|
13c99a0c3f | ||
|
|
1aa7c8db59 | ||
|
|
36b57eec24 | ||
|
|
417f8beaf9 | ||
|
|
d66bf1e2f8 | ||
|
|
c590a4088a | ||
|
|
80dfd1ad3b | ||
|
|
857844474a | ||
|
|
a85997926c | ||
|
|
70fb39ae55 | ||
|
|
82de529c54 | ||
|
|
db9757a45e | ||
|
|
5b654cc221 | ||
|
|
d518f5eac7 | ||
|
|
053d5c4bc0 | ||
|
|
032543d625 | ||
|
|
07149a02ad | ||
|
|
08a0c744b1 | ||
|
|
60324084d8 | ||
|
|
83ba79d7ae | ||
|
|
b704d50df1 | ||
|
|
68a9cceaa0 | ||
|
|
aec5a52c7f | ||
|
|
8966ecb4a6 | ||
|
|
3e9a5c0c5b | ||
|
|
dfc261cf34 | ||
|
|
cab98295e2 | ||
|
|
5a83c63810 | ||
|
|
2e43f55455 | ||
|
|
244fcda207 | ||
|
|
ca6fce0765 | ||
|
|
b6f92354d3 | ||
|
|
1b7fe6a9a7 | ||
|
|
14ea4d9c15 | ||
|
|
3ee333baac | ||
|
|
ecb41dfbbf | ||
|
|
33ac3f786b | ||
|
|
474c6a3634 | ||
|
|
c1e8bc0c57 | ||
|
|
2f5072166d | ||
|
|
89e2d53e8b | ||
|
|
bfe0a32e09 | ||
|
|
e88d176fed | ||
|
|
c96727a97c | ||
|
|
6362db0cc5 | ||
|
|
9239b16175 | ||
|
|
96a700cb9d | ||
|
|
41d91120d4 | ||
|
|
5d20407b15 | ||
|
|
322e3f65d1 | ||
|
|
3ff1ad9372 | ||
|
|
818ead5878 | ||
|
|
803d311d1c | ||
|
|
677e4fe6bc | ||
|
|
02a83575a2 | ||
|
|
cfbec1f6ea | ||
|
|
e11a134a1f | ||
|
|
6d05120930 | ||
|
|
bd2d1e3b26 | ||
|
|
c9605d1b29 | ||
|
|
67594f8b07 | ||
|
|
301fb57bb1 | ||
|
|
13eee99ea3 | ||
|
|
fb802c62cd | ||
|
|
767b606bb2 | ||
|
|
8492c07521 | ||
|
|
17698c2725 | ||
|
|
897c5229a6 | ||
|
|
ab5eaac5ed | ||
|
|
234071869d | ||
|
|
3e3b343131 | ||
|
|
5790aa0cbd | ||
|
|
a17f503486 | ||
|
|
74dccb6c44 | ||
|
|
e83bde6896 | ||
|
|
83d435b49e | ||
|
|
2ef4459b18 | ||
|
|
c2fd0ac66d | ||
|
|
5387e206b1 | ||
|
|
fc34385341 | ||
|
|
5fbdfb1436 |
@@ -20,3 +20,5 @@ autom4te.cache
|
||||
# the executable from tests
|
||||
runs
|
||||
|
||||
# Documentation temporary files
|
||||
docs/src/userguide.pdf
|
||||
|
||||
+1
-1
@@ -75,7 +75,7 @@ CDEFINES=$(AMGCDEFINES)
|
||||
AMGFDEFINES=@AMGFDEFINES@ $(PSBFDEFINES)
|
||||
FDEFINES=$(AMGFDEFINES)
|
||||
|
||||
CXXDEFINES=@AMGCXXDEFINES@
|
||||
CXXDEFINES=@AMGCXXDEFINES@ $(PSBCXXDEFINES)
|
||||
|
||||
@COMPILERULES@
|
||||
|
||||
|
||||
+1
-1
@@ -49,7 +49,7 @@ PSBLAS_INCLUDES=@PSBLAS_INCLUDES@
|
||||
PSBLAS_LIBS=@PSBLAS_LIBS@
|
||||
PSBBASEMODNAME=psb_base_mod
|
||||
PSBPRECMODNAME=psb_prec_mod
|
||||
PSBMETHDMODNAME=psb_krylov_mod
|
||||
PSBMETHDMODNAME=psb_linsolve_mod
|
||||
PSBUTILMODNAME=psb_util_mod
|
||||
|
||||
|
||||
|
||||
@@ -3,9 +3,9 @@ include Make.inc
|
||||
|
||||
all: objs lib
|
||||
|
||||
objs: amgp cbnd
|
||||
objs: libdir amgp cbnd
|
||||
|
||||
lib: libdir objs
|
||||
lib: objs
|
||||
cd amgprec && $(MAKE) lib
|
||||
cd cbind && $(MAKE) lib
|
||||
|
||||
@@ -46,6 +46,7 @@ cleanlib:
|
||||
|
||||
veryclean: cleanlib
|
||||
(cd amgprec && $(MAKE) veryclean)
|
||||
(cd cbind && $(MAKE) veryclean)
|
||||
(cd samples/simple/fileread && $(MAKE) clean)
|
||||
(cd samples/simple/pdegen && $(MAKE) clean)
|
||||
(cd samples/advanced/fileread && $(MAKE) clean)
|
||||
@@ -56,3 +57,4 @@ check: all
|
||||
|
||||
clean:
|
||||
(cd amgprec && $(MAKE) clean)
|
||||
(cd cbind && $(MAKE) clean)
|
||||
|
||||
@@ -1,54 +1,76 @@
|
||||
AMG4PSBLAS
|
||||
Algebraic Multigrid Package based on PSBLAS (Parallel Sparse BLAS version 3.8)
|
||||
|
||||
Salvatore Filippone (University of Rome Tor Vergata and IAC-CNR)
|
||||
Pasqua D'Ambra (IAC-CNR, Naples, IT)
|
||||
Fabio Durastante (IAC-CNR, Naples, IT)
|
||||
# AMG4PSBLAS v1.2
|
||||
Algebraic Multigrid Package based on [PSBLAS](https://github.com/sfilippone/psblas3) (Parallel Sparse BLAS version 3.9)
|
||||
|
||||
---------------------------------------------------------------------
|
||||
AMG4PSBLAS is a package of parallel algebraic multilevel preconditioners included in the PSCToolkit (Parallel Sparse Computation Toolkit) software framework.
|
||||
|
||||
AMG4PSBLAS is a package of Algebraic MultiGrid (AMG)
|
||||
preconditioners for the iterative solution of large and sparse linear systems.
|
||||
It is a progress of a software development project started in 2007, named MLD2P4, which originally implemented a multilevel version of some domain decomposition preconditioners of additive-Schwarz type and was based on a parallel decoupled version of the well known smoothed aggregation method to generate the multilevel hierarchy of coarser matrices.
|
||||
|
||||
It is an evolution of MLD2P4 (see LICENSE.MLD2P4), but it has been
|
||||
thoroughly reworked, and it is sufficiently different to warrant a new
|
||||
project name.
|
||||
In the last years the package was extended for including new algorithms and functionalities for the setup and application new AMG preconditioners with the final aims of improving efficiency and scalability when tens of thousands cores are used and of boosting reliability in dealing with general symmetric positive definite linear systems.
|
||||
|
||||
It is an evolution of MLD2P4 (see [LICENSE.MLD2P4](LICENSE.MLD2P4)), but due to the significant number of changes and the increase in scope, we decided to rename the package as AMG4PSBLAS.
|
||||
|
||||
MAIN REFERENCES:
|
||||
AMG4PSBLAS has been designed to provide scalable and easy-to-use preconditioners in the context of the PSBLAS (Parallel Sparse Basic Linear Algebra Subprograms) computational framework and can be used in conjuction with the Krylov solvers available in this framework. Our package is based on a completely algebraic approach; therefore users level interfaces assume that the system matrix and preconditioners are represented as PSBLAS distributed sparse matrices.
|
||||
|
||||
|
||||
AMG4PSBLAS enables the user to easily specify different features of an algebraic multilevel preconditioner, thus allowing to experiment with different preconditioners for the problem and parallel computers at hand.
|
||||
|
||||
P. D'Ambra, D. di Serafino, S. Filippone,
|
||||
MLD2P4: a Package of Parallel Algebraic Multilevel Domain Decomposition
|
||||
Preconditioners in Fortran 95,
|
||||
ACM Transactions on Mathematical Software, 37 (3), 2010, art. 30,
|
||||
doi: 10.1145/1824801.1824808.
|
||||
The package employs object-oriented design techniques in Fortran 2008, with interfaces to additional third party libraries such as MUMPS, UMFPACK, SuperLU, and SuperLU_Dist, which can be exploited in building multilevel preconditioners. The parallel implementation is based on a Single Program Multiple Data (SPMD) paradigm; the inter-process communication is based on MPI and is managed mainly through PSBLAS.
|
||||
|
||||
## Main Refrerences:
|
||||
|
||||
TO COMPILE
|
||||
The main reference for this project is
|
||||
> D'Ambra, P., Durastante, F., & Filippone, S. (2021). AMG preconditioners for linear solvers towards extreme scale. SIAM Journal on Scientific Computing, 43(5), S679-S703.
|
||||
|
||||
AMG4PSBLAS is the suite of preconditioners for the Parallel Sparse Computation Toolkit ([PSCToolkit](https://psctoolkit.github.io/)) suite of libraries. See the paper:
|
||||
> D’Ambra, P., Durastante, F., & Filippone, S. (2023). Parallel Sparse Computation Toolkit. Software Impacts, 15, 100463.
|
||||
|
||||
The main reference for features inherited from MLD2P4 is
|
||||
> P. D'Ambra, D. di Serafino, S. Filippone,
|
||||
> MLD2P4: a Package of Parallel Algebraic Multilevel Domain Decomposition
|
||||
> Preconditioners in Fortran 95,
|
||||
> ACM Transactions on Mathematical Software, 37 (3), 2010, art. 30,
|
||||
> doi: 10.1145/1824801.1824808.
|
||||
|
||||
## Installing
|
||||
|
||||
Installation requires having a working version of the [PSBLAS](https://github.com/sfilippone/psblas3) library installed.
|
||||
AMG4PSBLAS has several interfaces to third-party libraries that can be used in the construction and application phases of preconditioners.
|
||||
In particular, it is possible to link AMG4PSBLAS with the libraries: MUMPS, SuperLU, SuperLU_Dist, UMFPACK. This is _not mandatory_ and the library can run
|
||||
in isolation and without these features.
|
||||
|
||||
0. Unpack the tar file in a directory of your choice (preferrably
|
||||
outside the main PSBLAS directory).
|
||||
1. run configure --with-psblas=<ABSOLUTE path of the PSBLAS install directory>
|
||||
1. run configure `--with-psblas=<ABSOLUTE path of the PSBLAS install directory>`
|
||||
adding the options for MUMPS, SuperLU, SuperLU_Dist, UMFPACK as desired.
|
||||
See MLD2P4 User's and Reference Guide (Section 3) for details.
|
||||
2. Tweak Make.inc if you are not satisfied.
|
||||
3. make;
|
||||
See [AMG4PSBLAS User's and Reference Guide](docs/amg4psblas_1.0-guide.pdf) (Section 3) for details.
|
||||
2. Tweak `Make.inc` if you are not satisfied.
|
||||
3. run `make`;
|
||||
4. Go into the test subdirectory and build the examples of your choice.
|
||||
5. (if desired): make install
|
||||
5. (if desired): `make install`
|
||||
|
||||
>[!CAUTION]
|
||||
>The single precision version is supported only by MUMPS and SuperLU;
|
||||
>thus, even if you specify at configure time to use UMFPACK or SuperLU_Dist,
|
||||
>the corresponding preconditioner options will be available only from
|
||||
>the double precision version.
|
||||
|
||||
NOTES
|
||||
### CUDA, OpeMP, OpenACC
|
||||
|
||||
- The single precision version is supported only by MUMPS and SuperLU;
|
||||
thus, even if you specify at configure time to use UMFPACK or SuperLU_Dist,
|
||||
the corresponding preconditioner options will be available only from
|
||||
the double precision version.
|
||||
CUDA, OpenMP and OpenACC features are transparently inherited by PSBLAS installation. If PSBLAS has been configured (and installed) with these supports then AMG4PSBLAS will transparently inherit them. It will then be possible to move the computation to GPU accelerator simply by selecting the appropriate variable types. If these have not been activated or installed for PSBLAS then they will not be available for AMG4PSBLAS either and the operation will be purely on CPU/MPI.
|
||||
|
||||
### EoCoE - Software as service portal
|
||||
|
||||
In the European project “Energy oriented Center of Excellence: toward exascale for energy” we made available a software as service portal: [https://eocoe.psnc.pl/](https://eocoe.psnc.pl/). This permits to test several cutting-edge computational methods for accelerating the transition to the production, storage and management of clean, decarbonized energy. Among them you have the possibility of running PSBLAS+AMG4PSBLAS on some test problems to become familiar with using the software.
|
||||
|
||||
## TODO and bugs
|
||||
|
||||
- [X] Fix all reamining bugs. Bugs? We dont' have any ! 🤓
|
||||
|
||||
> [!NOTE]
|
||||
> To report bugs 🐛 or issues ❓ please use the [GitHub issue system](https://github.com/sfilippone/amg4psblas/issues).
|
||||
|
||||
## The AMG4PSBLAS team.
|
||||
|
||||
- Pasqua D'Ambra (IAC-CNR, Naples, IT)
|
||||
- Fabio Durastante (University of Pisa and IAC-CNR, IT)
|
||||
- Salvatore Filippone (University of Rome Tor Vergata and IAC-CNR, IT)
|
||||
|
||||
The AMG4PSBLAS team.
|
||||
---------------
|
||||
Salvatore Filippone
|
||||
Pasqua D'Ambra
|
||||
Fabio Durastante
|
||||
|
||||
@@ -288,17 +288,19 @@ module amg_base_prec_type
|
||||
!
|
||||
! Legal values for entry: amg_aggr_prol_
|
||||
!
|
||||
integer(psb_ipk_), parameter :: amg_no_smooth_ = 0
|
||||
integer(psb_ipk_), parameter :: amg_smooth_prol_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_min_energy_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_no_smooth_ = 0
|
||||
integer(psb_ipk_), parameter :: amg_smooth_prol_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_l1_smooth_prol_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_min_energy_ = 3
|
||||
! Disabling min_energy for the time being.
|
||||
integer(psb_ipk_), parameter :: amg_max_aggr_prol_=amg_smooth_prol_
|
||||
integer(psb_ipk_), parameter :: amg_max_aggr_prol_= amg_l1_smooth_prol_
|
||||
!
|
||||
! Legal values for entry: amg_aggr_filter_
|
||||
!
|
||||
integer(psb_ipk_), parameter :: amg_no_filter_mat_ = 0
|
||||
integer(psb_ipk_), parameter :: amg_filter_mat_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_max_filter_mat_ = amg_filter_mat_
|
||||
integer(psb_ipk_), parameter :: amg_no_filter_mat_ = 0
|
||||
integer(psb_ipk_), parameter :: amg_filter_mat_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_filter_prow_mat_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_max_filter_mat_ = amg_filter_prow_mat_
|
||||
!
|
||||
! Legal values for entry: amg_aggr_ord_
|
||||
!
|
||||
@@ -323,9 +325,10 @@ module amg_base_prec_type
|
||||
!
|
||||
! Legal values for entry: amg_poly_variant_
|
||||
!
|
||||
integer(psb_ipk_), parameter :: amg_poly_lottes_ = 0
|
||||
integer(psb_ipk_), parameter :: amg_poly_lottes_beta_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_poly_new_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_cheb_4_ = 0
|
||||
integer(psb_ipk_), parameter :: amg_cheb_4_opt_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_cheb_1_opt_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_poly_dbg_ = 8
|
||||
|
||||
integer(psb_ipk_), parameter :: amg_poly_rho_est_power_ = 0
|
||||
|
||||
@@ -375,10 +378,11 @@ module amg_base_prec_type
|
||||
character(len=19), parameter, private :: &
|
||||
& eigen_estimates(0:0)=(/'infinity norm '/)
|
||||
character(len=15), parameter, private :: &
|
||||
& aggr_prols(0:3)=(/'unsmoothed ','smoothed ',&
|
||||
& 'min energy ','bizr. smoothed'/)
|
||||
& aggr_prols(0:4)=(/'unsmoothed ','smoothed ',&
|
||||
& 'l1-smoothed ','min energy ','bizr. smoothed'/)
|
||||
character(len=15), parameter, private :: &
|
||||
& aggr_filters(0:1)=(/'no filtering ','filtering '/)
|
||||
& aggr_filters(0:2)=(/'no filtering ','filtering ',&
|
||||
& 'filtering rsum'/)
|
||||
character(len=15), parameter, private :: &
|
||||
& matrix_names(0:1)=(/'distributed ','replicated '/)
|
||||
character(len=18), parameter, private :: &
|
||||
@@ -547,6 +551,8 @@ contains
|
||||
val = amg_no_smooth_
|
||||
case('SMOOTHED')
|
||||
val = amg_smooth_prol_
|
||||
case('L1-SMOOTHED','L1SMOOTHED')
|
||||
val = amg_l1_smooth_prol_
|
||||
case('MINENERGY')
|
||||
val = amg_min_energy_
|
||||
case('NOPREC')
|
||||
@@ -569,12 +575,14 @@ contains
|
||||
val = amg_as_
|
||||
case('POLY')
|
||||
val = amg_poly_
|
||||
case('POLY_LOTTES')
|
||||
val = amg_poly_lottes_
|
||||
case('POLY_LOTTES_BETA')
|
||||
val = amg_poly_lottes_beta_
|
||||
case('POLY_NEW')
|
||||
val = amg_poly_new_
|
||||
case('CHEB_4')
|
||||
val = amg_cheb_4_
|
||||
case('CHEB_4_OPT')
|
||||
val = amg_cheb_4_opt_
|
||||
case('CHEB_1_OPT')
|
||||
val = amg_cheb_1_opt_
|
||||
case('POLY_DBG')
|
||||
val = amg_poly_dbg_
|
||||
case('POLY_RHO_EST_POWER')
|
||||
val = amg_poly_rho_est_power_
|
||||
case('A_NORMI')
|
||||
@@ -585,6 +593,8 @@ contains
|
||||
val = amg_eig_est_
|
||||
case('FILTER')
|
||||
val = amg_filter_mat_
|
||||
case('FILTERROWSUM')
|
||||
val = amg_filter_prow_mat_
|
||||
case('NOFILTER','NO_FILTER')
|
||||
val = amg_no_filter_mat_
|
||||
case('OUTER_SWEEPS')
|
||||
|
||||
@@ -91,9 +91,9 @@ module amg_c_ainv_solver
|
||||
import :: psb_desc_type, psb_cspmat_type, psb_c_base_sparse_mat, &
|
||||
& amg_c_base_solver_type, psb_dpk_, amg_c_ainv_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_c_ainv_solver_type), intent(inout) :: sv
|
||||
class(amg_c_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_c_ainv_solver_type), intent(inout) :: sv
|
||||
class(amg_c_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_c_ainv_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -109,11 +109,12 @@ module amg_c_inner_mod
|
||||
end interface amg_map_to_tprol
|
||||
|
||||
abstract interface
|
||||
subroutine amg_caggrmat_var_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
subroutine amg_caggrmat_var_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: psb_cspmat_type, psb_desc_type, psb_spk_, psb_ipk_, psb_lpk_, psb_lcspmat_type
|
||||
import :: amg_c_onelev_type, amg_sml_parms
|
||||
implicit none
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_cspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
|
||||
@@ -79,9 +79,9 @@ module amg_c_invk_solver
|
||||
import :: psb_desc_type, psb_cspmat_type, psb_c_base_sparse_mat, &
|
||||
& amg_c_base_solver_type, psb_spk_, amg_c_invk_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_c_invk_solver_type), intent(inout) :: sv
|
||||
class(amg_c_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_c_invk_solver_type), intent(inout) :: sv
|
||||
class(amg_c_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_c_invk_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -79,9 +79,9 @@ module amg_c_invt_solver
|
||||
import :: psb_desc_type, psb_cspmat_type, psb_c_base_sparse_mat, &
|
||||
& amg_c_base_solver_type, psb_spk_, amg_c_invt_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_c_invt_solver_type), intent(inout) :: sv
|
||||
class(amg_c_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_c_invt_solver_type), intent(inout) :: sv
|
||||
class(amg_c_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_c_invt_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -203,8 +203,8 @@ module amg_c_jac_smoother
|
||||
subroutine amg_c_jac_smoother_clone_settings(sm,smout,info)
|
||||
import :: amg_c_jac_smoother_type, psb_spk_, &
|
||||
& amg_c_base_smoother_type, psb_ipk_
|
||||
class(amg_c_jac_smoother_type), intent(inout) :: sm
|
||||
class(amg_c_base_smoother_type), allocatable, intent(inout) :: smout
|
||||
class(amg_c_jac_smoother_type), intent(inout) :: sm
|
||||
class(amg_c_base_smoother_type), intent(inout) :: smout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_c_jac_smoother_clone_settings
|
||||
end interface
|
||||
|
||||
@@ -91,9 +91,9 @@ module amg_d_ainv_solver
|
||||
import :: psb_desc_type, psb_dspmat_type, psb_d_base_sparse_mat, &
|
||||
& amg_d_base_solver_type, psb_dpk_, amg_d_ainv_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_d_ainv_solver_type), intent(inout) :: sv
|
||||
class(amg_d_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_d_ainv_solver_type), intent(inout) :: sv
|
||||
class(amg_d_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_ainv_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -109,11 +109,12 @@ module amg_d_inner_mod
|
||||
end interface amg_map_to_tprol
|
||||
|
||||
abstract interface
|
||||
subroutine amg_daggrmat_var_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
subroutine amg_daggrmat_var_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: psb_dspmat_type, psb_desc_type, psb_dpk_, psb_ipk_, psb_lpk_, psb_ldspmat_type
|
||||
import :: amg_d_onelev_type, amg_dml_parms
|
||||
implicit none
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
|
||||
@@ -79,9 +79,9 @@ module amg_d_invk_solver
|
||||
import :: psb_desc_type, psb_dspmat_type, psb_d_base_sparse_mat, &
|
||||
& amg_d_base_solver_type, psb_dpk_, amg_d_invk_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_d_invk_solver_type), intent(inout) :: sv
|
||||
class(amg_d_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_d_invk_solver_type), intent(inout) :: sv
|
||||
class(amg_d_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_invk_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -79,9 +79,9 @@ module amg_d_invt_solver
|
||||
import :: psb_desc_type, psb_dspmat_type, psb_d_base_sparse_mat, &
|
||||
& amg_d_base_solver_type, psb_dpk_, amg_d_invt_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_d_invt_solver_type), intent(inout) :: sv
|
||||
class(amg_d_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_d_invt_solver_type), intent(inout) :: sv
|
||||
class(amg_d_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_invt_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -203,8 +203,8 @@ module amg_d_jac_smoother
|
||||
subroutine amg_d_jac_smoother_clone_settings(sm,smout,info)
|
||||
import :: amg_d_jac_smoother_type, psb_dpk_, &
|
||||
& amg_d_base_smoother_type, psb_ipk_
|
||||
class(amg_d_jac_smoother_type), intent(inout) :: sm
|
||||
class(amg_d_base_smoother_type), allocatable, intent(inout) :: smout
|
||||
class(amg_d_jac_smoother_type), intent(inout) :: sm
|
||||
class(amg_d_base_smoother_type), intent(inout) :: smout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_jac_smoother_clone_settings
|
||||
end interface
|
||||
|
||||
@@ -244,11 +244,12 @@ module amg_d_parmatch_aggregator_mod
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
subroutine amg_d_parmatch_unsmth_bld(dol1smoothing,ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: amg_d_parmatch_aggregator_type, psb_desc_type, psb_dspmat_type,&
|
||||
& psb_ldspmat_type, psb_dpk_, psb_ipk_, psb_lpk_, amg_dml_parms, amg_daggr_data
|
||||
implicit none
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
class(amg_d_parmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
@@ -262,11 +263,12 @@ module amg_d_parmatch_aggregator_mod
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
subroutine amg_d_parmatch_smth_bld(dol1smoothing,ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: amg_d_parmatch_aggregator_type, psb_desc_type, psb_dspmat_type,&
|
||||
& psb_ldspmat_type, psb_dpk_, psb_ipk_, psb_lpk_, amg_dml_parms, amg_daggr_data
|
||||
implicit none
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
class(amg_d_parmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
|
||||
@@ -192,8 +192,8 @@ module amg_d_poly_smoother
|
||||
subroutine amg_d_poly_smoother_clone_settings(sm,smout,info)
|
||||
import :: amg_d_poly_smoother_type, psb_dpk_, &
|
||||
& amg_d_base_smoother_type, psb_ipk_
|
||||
class(amg_d_poly_smoother_type), intent(inout) :: sm
|
||||
class(amg_d_base_smoother_type), allocatable, intent(inout) :: smout
|
||||
class(amg_d_poly_smoother_type), intent(inout) :: sm
|
||||
class(amg_d_base_smoother_type), intent(inout) :: smout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_d_poly_smoother_clone_settings
|
||||
end interface
|
||||
@@ -322,7 +322,7 @@ contains
|
||||
!
|
||||
sm%pdegree = 1
|
||||
sm%rho_ba = -done
|
||||
sm%variant = amg_poly_lottes_
|
||||
sm%variant = amg_cheb_4_
|
||||
sm%rho_estimate = amg_poly_rho_est_power_
|
||||
sm%rho_estimate_iterations = 20
|
||||
if (allocated(sm%sv)) then
|
||||
|
||||
@@ -91,9 +91,9 @@ module amg_s_ainv_solver
|
||||
import :: psb_desc_type, psb_sspmat_type, psb_s_base_sparse_mat, &
|
||||
& amg_s_base_solver_type, psb_dpk_, amg_s_ainv_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_s_ainv_solver_type), intent(inout) :: sv
|
||||
class(amg_s_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_s_ainv_solver_type), intent(inout) :: sv
|
||||
class(amg_s_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_s_ainv_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -109,11 +109,12 @@ module amg_s_inner_mod
|
||||
end interface amg_map_to_tprol
|
||||
|
||||
abstract interface
|
||||
subroutine amg_saggrmat_var_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
subroutine amg_saggrmat_var_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: psb_sspmat_type, psb_desc_type, psb_spk_, psb_ipk_, psb_lpk_, psb_lsspmat_type
|
||||
import :: amg_s_onelev_type, amg_sml_parms
|
||||
implicit none
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_sspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
|
||||
@@ -79,9 +79,9 @@ module amg_s_invk_solver
|
||||
import :: psb_desc_type, psb_sspmat_type, psb_s_base_sparse_mat, &
|
||||
& amg_s_base_solver_type, psb_spk_, amg_s_invk_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_s_invk_solver_type), intent(inout) :: sv
|
||||
class(amg_s_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_s_invk_solver_type), intent(inout) :: sv
|
||||
class(amg_s_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_s_invk_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -79,9 +79,9 @@ module amg_s_invt_solver
|
||||
import :: psb_desc_type, psb_sspmat_type, psb_s_base_sparse_mat, &
|
||||
& amg_s_base_solver_type, psb_spk_, amg_s_invt_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_s_invt_solver_type), intent(inout) :: sv
|
||||
class(amg_s_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_s_invt_solver_type), intent(inout) :: sv
|
||||
class(amg_s_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_s_invt_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -203,8 +203,8 @@ module amg_s_jac_smoother
|
||||
subroutine amg_s_jac_smoother_clone_settings(sm,smout,info)
|
||||
import :: amg_s_jac_smoother_type, psb_spk_, &
|
||||
& amg_s_base_smoother_type, psb_ipk_
|
||||
class(amg_s_jac_smoother_type), intent(inout) :: sm
|
||||
class(amg_s_base_smoother_type), allocatable, intent(inout) :: smout
|
||||
class(amg_s_jac_smoother_type), intent(inout) :: sm
|
||||
class(amg_s_base_smoother_type), intent(inout) :: smout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_s_jac_smoother_clone_settings
|
||||
end interface
|
||||
|
||||
@@ -244,11 +244,12 @@ module amg_s_parmatch_aggregator_mod
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_s_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
subroutine amg_s_parmatch_unsmth_bld(dol1smoothing,ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: amg_s_parmatch_aggregator_type, psb_desc_type, psb_sspmat_type,&
|
||||
& psb_lsspmat_type, psb_dpk_, psb_ipk_, psb_lpk_, amg_sml_parms, amg_saggr_data
|
||||
implicit none
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
class(amg_s_parmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(psb_sspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
@@ -262,11 +263,12 @@ module amg_s_parmatch_aggregator_mod
|
||||
end interface
|
||||
|
||||
interface
|
||||
subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
subroutine amg_s_parmatch_smth_bld(dol1smoothing,ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: amg_s_parmatch_aggregator_type, psb_desc_type, psb_sspmat_type,&
|
||||
& psb_lsspmat_type, psb_dpk_, psb_ipk_, psb_lpk_, amg_sml_parms, amg_saggr_data
|
||||
implicit none
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
class(amg_s_parmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(psb_sspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
|
||||
@@ -192,8 +192,8 @@ module amg_s_poly_smoother
|
||||
subroutine amg_s_poly_smoother_clone_settings(sm,smout,info)
|
||||
import :: amg_s_poly_smoother_type, psb_spk_, &
|
||||
& amg_s_base_smoother_type, psb_ipk_
|
||||
class(amg_s_poly_smoother_type), intent(inout) :: sm
|
||||
class(amg_s_base_smoother_type), allocatable, intent(inout) :: smout
|
||||
class(amg_s_poly_smoother_type), intent(inout) :: sm
|
||||
class(amg_s_base_smoother_type), intent(inout) :: smout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_s_poly_smoother_clone_settings
|
||||
end interface
|
||||
@@ -322,7 +322,7 @@ contains
|
||||
!
|
||||
sm%pdegree = 1
|
||||
sm%rho_ba = -sone
|
||||
sm%variant = amg_poly_lottes_
|
||||
sm%variant = amg_cheb_4_
|
||||
sm%rho_estimate = amg_poly_rho_est_power_
|
||||
sm%rho_estimate_iterations = 20
|
||||
if (allocated(sm%sv)) then
|
||||
|
||||
@@ -91,9 +91,9 @@ module amg_z_ainv_solver
|
||||
import :: psb_desc_type, psb_zspmat_type, psb_z_base_sparse_mat, &
|
||||
& amg_z_base_solver_type, psb_dpk_, amg_z_ainv_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_z_ainv_solver_type), intent(inout) :: sv
|
||||
class(amg_z_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_z_ainv_solver_type), intent(inout) :: sv
|
||||
class(amg_z_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_z_ainv_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -109,11 +109,12 @@ module amg_z_inner_mod
|
||||
end interface amg_map_to_tprol
|
||||
|
||||
abstract interface
|
||||
subroutine amg_zaggrmat_var_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
subroutine amg_zaggrmat_var_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
import :: psb_zspmat_type, psb_desc_type, psb_dpk_, psb_ipk_, psb_lpk_, psb_lzspmat_type
|
||||
import :: amg_z_onelev_type, amg_dml_parms
|
||||
implicit none
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_zspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
|
||||
@@ -79,9 +79,9 @@ module amg_z_invk_solver
|
||||
import :: psb_desc_type, psb_zspmat_type, psb_z_base_sparse_mat, &
|
||||
& amg_z_base_solver_type, psb_dpk_, amg_z_invk_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_z_invk_solver_type), intent(inout) :: sv
|
||||
class(amg_z_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_z_invk_solver_type), intent(inout) :: sv
|
||||
class(amg_z_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_z_invk_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -79,9 +79,9 @@ module amg_z_invt_solver
|
||||
import :: psb_desc_type, psb_zspmat_type, psb_z_base_sparse_mat, &
|
||||
& amg_z_base_solver_type, psb_dpk_, amg_z_invt_solver_type, psb_ipk_
|
||||
Implicit None
|
||||
class(amg_z_invt_solver_type), intent(inout) :: sv
|
||||
class(amg_z_base_solver_type), allocatable, intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
class(amg_z_invt_solver_type), intent(inout) :: sv
|
||||
class(amg_z_base_solver_type), intent(inout) :: svout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_z_invt_solver_clone_settings
|
||||
end interface
|
||||
|
||||
|
||||
@@ -203,8 +203,8 @@ module amg_z_jac_smoother
|
||||
subroutine amg_z_jac_smoother_clone_settings(sm,smout,info)
|
||||
import :: amg_z_jac_smoother_type, psb_dpk_, &
|
||||
& amg_z_base_smoother_type, psb_ipk_
|
||||
class(amg_z_jac_smoother_type), intent(inout) :: sm
|
||||
class(amg_z_base_smoother_type), allocatable, intent(inout) :: smout
|
||||
class(amg_z_jac_smoother_type), intent(inout) :: sm
|
||||
class(amg_z_base_smoother_type), intent(inout) :: smout
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
end subroutine amg_z_jac_smoother_clone_settings
|
||||
end interface
|
||||
|
||||
@@ -67,14 +67,13 @@ void dMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
#endif
|
||||
|
||||
|
||||
#define TIME_TRACKER
|
||||
#ifdef TIME_TRACKER
|
||||
double tmr = MPI_Wtime();
|
||||
#endif
|
||||
#undef TIME_TRACKER
|
||||
#ifdef TIME_TRACKER
|
||||
double tmr = MPI_Wtime();
|
||||
#endif
|
||||
|
||||
// Rimosso per tornare al vecchio matching #define OMP
|
||||
#ifdef OMP
|
||||
fprintf(stderr,"Warning: using buggy OpenMP matching!\n");
|
||||
#if defined(OPENMP)
|
||||
//fprintf(stderr,"Warning: using buggy OpenMP matching!\n");
|
||||
dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(NLVer, NLEdge,
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
@@ -93,11 +92,11 @@ void dMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
#endif
|
||||
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
tmr = MPI_Wtime() - tmr;
|
||||
fprintf(stderr, "Elaboration time: %f for %ld nodes\n", tmr, NLVer);
|
||||
#endif
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
tmr = MPI_Wtime() - tmr;
|
||||
fprintf(stderr, "Elaboration time: %f for %ld nodes\n", tmr, NLVer);
|
||||
#endif
|
||||
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -115,13 +114,24 @@ void sMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
fprintf(stderr,"MatchBoxPC: rank %d nlver %ld nledge %ld [ %ld %ld ]\n",
|
||||
myRank,NLVer, NLEdge,verDistance[0],verDistance[1]);
|
||||
#endif
|
||||
#if defined(OPENMP)
|
||||
//fprintf(stderr,"Warning: using buggy OpenMP matching!\n");
|
||||
salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(NLVer, NLEdge,
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
myRank, numProcs, C_comm,
|
||||
msgIndSent, msgActualSent, msgPercent,
|
||||
ph0_time, ph1_time, ph2_time,
|
||||
ph1_card, ph2_card );
|
||||
#else
|
||||
salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(NLVer, NLEdge,
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
myRank, numProcs, C_comm,
|
||||
msgIndSent, msgActualSent, msgPercent,
|
||||
ph0_time, ph1_time, ph2_time,
|
||||
ph1_card, ph2_card );
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
myRank, numProcs, C_comm,
|
||||
msgIndSent, msgActualSent, msgPercent,
|
||||
ph0_time, ph1_time, ph2_time,
|
||||
ph1_card, ph2_card );
|
||||
#endif
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -59,7 +59,7 @@
|
||||
#include <assert.h>
|
||||
#include <map>
|
||||
#include <vector>
|
||||
#ifdef OMP
|
||||
#ifdef OPENMP
|
||||
// OpenMP is included and used if and only if the OpenMP version of the matching
|
||||
// is required
|
||||
#include "omp.h"
|
||||
@@ -82,6 +82,8 @@ const int BundleTag = 9; // Predefined tag
|
||||
|
||||
static vector<MilanLongInt> DEFAULT_VECTOR;
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
// MPI type map
|
||||
template <typename T>
|
||||
MPI_Datatype TypeMap();
|
||||
@@ -93,6 +95,7 @@ template <>
|
||||
inline MPI_Datatype TypeMap<double>() { return MPI_DOUBLE; }
|
||||
template <>
|
||||
inline MPI_Datatype TypeMap<float>() { return MPI_FLOAT; }
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C"
|
||||
@@ -178,264 +181,416 @@ extern "C"
|
||||
#define MilanRealMin MINUS_INFINITY
|
||||
#endif
|
||||
|
||||
#ifdef OMP
|
||||
/* These functions are only used in the experimental OMP implementation, if that
|
||||
is disabled there is no reason to actually compile or reference them. */
|
||||
|
||||
// Function of find the owner of a ghost vertex using binary search:
|
||||
MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs);
|
||||
// Function of find the owner of a ghost vertex using binary search:
|
||||
MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs);
|
||||
|
||||
MilanLongInt firstComputeCandidateMateD(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight);
|
||||
|
||||
MilanLongInt firstComputeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight);
|
||||
|
||||
void queuesTransfer(vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
bool isAlreadyMatched(MilanLongInt node,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
MilanLongInt computeCandidateMateD(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt StartIndex, MilanLongInt EndIndex,
|
||||
MilanLongInt *numGhostEdgesPtr,
|
||||
MilanLongInt *numGhostVerticesPtr,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &Counter,
|
||||
vector<MilanLongInt> &verGhostPtr,
|
||||
vector<MilanLongInt> &verGhostInd,
|
||||
vector<MilanLongInt> &tempCounter,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Message,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
MilanLongInt *&candidateMate,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void clean(MilanLongInt NLVer,
|
||||
MilanInt myRank,
|
||||
MilanLongInt MessageIndex,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus,
|
||||
MilanInt BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
MilanLongInt msgActual,
|
||||
MilanLongInt *msgActualSent,
|
||||
MilanLongInt msgInd,
|
||||
MilanLongInt *msgIndSent,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanReal *msgPercent);
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_BD(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *candidateMate);
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_BD(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
MilanLongInt *SPtr);
|
||||
|
||||
void processMatchedVerticesD(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void processMatchedVerticesAndSendMessagesD(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message);
|
||||
|
||||
void sendBundledMessages(MilanLongInt *numGhostEdgesPtr,
|
||||
MilanInt *BufferSizePtr,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanLongInt *MessageIndexPtr,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus);
|
||||
|
||||
void processMessagesD(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *SPtr,
|
||||
vector<MilanLongInt> &U);
|
||||
|
||||
void extractUChunk(
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU);
|
||||
|
||||
void queuesTransfer(vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
MilanLongInt firstComputeCandidateMateS(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanFloat *edgeLocWeight);
|
||||
|
||||
bool isAlreadyMatched(MilanLongInt node,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
MilanLongInt computeCandidateMateS(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_BS(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *candidateMate);
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_BS(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
void processMatchedVerticesS(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void processMatchedVerticesAndSendMessagesS(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message);
|
||||
|
||||
void processMessagesS(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *SPtr,
|
||||
vector<MilanLongInt> &U);
|
||||
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
|
||||
MilanLongInt computeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt StartIndex, MilanLongInt EndIndex,
|
||||
MilanLongInt *numGhostEdgesPtr,
|
||||
MilanLongInt *numGhostVerticesPtr,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &Counter,
|
||||
vector<MilanLongInt> &verGhostPtr,
|
||||
vector<MilanLongInt> &verGhostInd,
|
||||
vector<MilanLongInt> &tempCounter,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Message,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
MilanLongInt *&candidateMate,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void clean(MilanLongInt NLVer,
|
||||
MilanInt myRank,
|
||||
MilanLongInt MessageIndex,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus,
|
||||
MilanInt BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
MilanLongInt msgActual,
|
||||
MilanLongInt *msgActualSent,
|
||||
MilanLongInt msgInd,
|
||||
MilanLongInt *msgIndSent,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanReal *msgPercent);
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_B(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *candidateMate);
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
MilanLongInt *SPtr);
|
||||
|
||||
void processMatchedVertices(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void processMatchedVerticesAndSendMessages(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message);
|
||||
|
||||
void sendBundledMessages(MilanLongInt *numGhostEdgesPtr,
|
||||
MilanInt *BufferSizePtr,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanLongInt *MessageIndexPtr,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus);
|
||||
|
||||
void processMessages(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *SPtr,
|
||||
vector<MilanLongInt> &U);
|
||||
|
||||
void extractUChunk(
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU);
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
#endif
|
||||
|
||||
void salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
|
||||
|
||||
#ifndef OMP
|
||||
//Function of find the owner of a ghost vertex using binary search:
|
||||
inline MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs);
|
||||
#endif
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
|
||||
+10
-10
@@ -1303,16 +1303,16 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(
|
||||
// SINGLE PRECISION VERSION
|
||||
|
||||
void salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* verLocPtr, MilanLongInt* verLocInd,
|
||||
MilanFloat* edgeLocWeight,
|
||||
MilanLongInt* verDistance,
|
||||
MilanLongInt* Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt* msgIndSent, MilanLongInt* msgActualSent,
|
||||
MilanReal* msgPercent,
|
||||
MilanReal* ph0_time, MilanReal* ph1_time, MilanReal* ph2_time,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card ) {
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* verLocPtr, MilanLongInt* verLocInd,
|
||||
MilanFloat* edgeLocWeight,
|
||||
MilanLongInt* verDistance,
|
||||
MilanLongInt* Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt* msgIndSent, MilanLongInt* msgActualSent,
|
||||
MilanReal* msgPercent,
|
||||
MilanReal* ph0_time, MilanReal* ph1_time, MilanReal* ph2_time,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card ) {
|
||||
#if !defined(SERIAL_MPI)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout<<"\n("<<myRank<<")Within algoEdgeApproxDominatingEdgesLinearSearchMessageBundling()"; fflush(stdout);
|
||||
|
||||
+502
-14
@@ -1,5 +1,4 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
// ***********************************************************************
|
||||
//
|
||||
// MatchboxP: A C++ library for approximate weighted matching
|
||||
@@ -126,8 +125,10 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
MilanLongInt StartIndex = verDistance[myRank]; // The starting vertex owned by the current rank
|
||||
MilanLongInt EndIndex = verDistance[myRank + 1] - 1; // The ending vertex owned by the current rank
|
||||
// The starting vertex owned by the current rank
|
||||
MilanLongInt StartIndex = verDistance[myRank];
|
||||
// The ending vertex owned by the current rank
|
||||
MilanLongInt EndIndex = verDistance[myRank + 1] - 1;
|
||||
|
||||
MPI_Status computeStatus;
|
||||
|
||||
@@ -145,7 +146,8 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
// only one message will be sent in the initialization phase -
|
||||
// one of: REQUEST/FAILURE/SUCCESS
|
||||
vector<MilanLongInt> QLocalVtx, QGhostVtx, QMsgType;
|
||||
vector<MilanInt> QOwner; // Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
// Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
vector<MilanInt> QOwner;
|
||||
|
||||
MilanLongInt *PCounter = new MilanLongInt[numProcs];
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
@@ -153,7 +155,8 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
|
||||
MilanLongInt NumMessagesBundled = 0;
|
||||
// TODO when the last computational section will be refactored this could be eliminated
|
||||
MilanInt ghostOwner = 0; // Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
// Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
MilanInt ghostOwner = 0;
|
||||
MilanLongInt *candidateMate = nullptr;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")NV: " << NLVer << " Edges: " << NLEdge;
|
||||
@@ -168,9 +171,12 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt myCard = 0;
|
||||
|
||||
// Build the Ghost Vertex Set: Vg
|
||||
map<MilanLongInt, MilanLongInt> Ghost2LocalMap; // Map each ghost vertex to a local vertex
|
||||
vector<MilanLongInt> Counter; // Store the edge count for each ghost vertex
|
||||
MilanLongInt numGhostVertices = 0, numGhostEdges = 0; // Number of Ghost vertices
|
||||
// Map each ghost vertex to a local vertex
|
||||
map<MilanLongInt, MilanLongInt> Ghost2LocalMap;
|
||||
// Store the edge count for each ghost vertex
|
||||
vector<MilanLongInt> Counter;
|
||||
// Number of Ghost vertices
|
||||
MilanLongInt numGhostVertices = 0, numGhostEdges = 0;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")About to compute Ghost Vertices...";
|
||||
@@ -237,7 +243,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
* PARALLEL_COMPUTE_CANDIDATE_MATE_B is now totally parallel.
|
||||
*/
|
||||
|
||||
PARALLEL_COMPUTE_CANDIDATE_MATE_B(NLVer,
|
||||
PARALLEL_COMPUTE_CANDIDATE_MATE_BD(NLVer,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
myRank,
|
||||
@@ -262,7 +268,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
* TODO: Test when it's actually more efficient to execute this code
|
||||
* in parallel.
|
||||
*/
|
||||
PARALLEL_PROCESS_EXPOSED_VERTEX_B(NLVer,
|
||||
PARALLEL_PROCESS_EXPOSED_VERTEX_BD(NLVer,
|
||||
candidateMate,
|
||||
verLocInd,
|
||||
verLocPtr,
|
||||
@@ -314,7 +320,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
vector<MilanLongInt> UChunkBeingProcessed;
|
||||
UChunkBeingProcessed.reserve(UCHUNK);
|
||||
|
||||
processMatchedVertices(NLVer,
|
||||
processMatchedVerticesD(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
@@ -423,7 +429,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
/////////////////////////// PROCESS MATCHED VERTICES //////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMatchedVerticesAndSendMessages(NLVer,
|
||||
processMatchedVerticesAndSendMessagesD(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
@@ -483,8 +489,8 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MESSAGES //////////////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMessages(NLVer,
|
||||
//startTime = MPI_Wtime();
|
||||
processMessagesD(NLVer,
|
||||
Mate,
|
||||
candidateMate,
|
||||
Ghost2LocalMap,
|
||||
@@ -549,6 +555,488 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
*ph2_card = myCard; // Cardinality at the end of Phase-2
|
||||
}
|
||||
// End of algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMate
|
||||
|
||||
void salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent,
|
||||
MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card)
|
||||
{
|
||||
|
||||
/*
|
||||
* verDistance: it's a vector long as the number of processors.
|
||||
* verDistance[i] contains the first node index of the i-th processor
|
||||
* verDistance[i + 1] contains the last node index of the i-th processor
|
||||
* NLVer: number of elements in the LocPtr
|
||||
* NLEdge: number of edges assigned to the current processor
|
||||
*
|
||||
* Contains the portion of matrix assigned to the processor in
|
||||
* Yale notation
|
||||
* verLocInd: contains the positions on row of the matrix
|
||||
* verLocPtr: i-th value is the position of the first element on the i-th row and
|
||||
* i+1-th value is the position of the first element on the i+1-th row
|
||||
*/
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Within algoEdgeApproxDominatingEdgesLinearSearchMessageBundling()";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ") verDistance [" ;
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
cout << verDistance[i] << "," << verDistance[i+1];
|
||||
cout << "]\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0) {
|
||||
cout << "\n(" << myRank << ") verDistance [" ;
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
cout << verDistance[i] << "," ;
|
||||
cout << verDistance[numProcs]<< "]\n";
|
||||
}
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// The starting vertex owned by the current rank
|
||||
MilanLongInt StartIndex = verDistance[myRank];
|
||||
// The ending vertex owned by the current rank
|
||||
MilanLongInt EndIndex = verDistance[myRank + 1] - 1;
|
||||
|
||||
MPI_Status computeStatus;
|
||||
|
||||
MilanLongInt msgActual = 0, msgInd = 0;
|
||||
MilanFloat heaviestEdgeWt = 0.0f; // Assumes positive weight
|
||||
MilanReal startTime, finishTime;
|
||||
|
||||
startTime = MPI_Wtime();
|
||||
|
||||
// Data structures for sending and receiving messages:
|
||||
vector<MilanLongInt> Message; // [ u, v, message_type ]
|
||||
Message.resize(3, -1);
|
||||
// Data structures for Message Bundling:
|
||||
// Although up to two messages can be sent along any cross edge,
|
||||
// only one message will be sent in the initialization phase -
|
||||
// one of: REQUEST/FAILURE/SUCCESS
|
||||
vector<MilanLongInt> QLocalVtx, QGhostVtx, QMsgType;
|
||||
// Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
vector<MilanInt> QOwner;
|
||||
|
||||
MilanLongInt *PCounter = new MilanLongInt[numProcs];
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
PCounter[i] = 0;
|
||||
|
||||
MilanLongInt NumMessagesBundled = 0;
|
||||
// TODO when the last computational section will be refactored this could be eliminated
|
||||
// Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
MilanInt ghostOwner = 0;
|
||||
MilanLongInt *candidateMate = nullptr;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")NV: " << NLVer << " Edges: " << NLEdge;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")StartIndex: " << StartIndex << " EndIndex: " << EndIndex;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Other Variables:
|
||||
MilanLongInt u = -1, v = -1, w = -1, i = 0;
|
||||
MilanLongInt k = -1, adj1 = -1, adj2 = -1;
|
||||
MilanLongInt k1 = -1, adj11 = -1, adj12 = -1;
|
||||
MilanLongInt myCard = 0;
|
||||
|
||||
// Build the Ghost Vertex Set: Vg
|
||||
// Map each ghost vertex to a local vertex
|
||||
map<MilanLongInt, MilanLongInt> Ghost2LocalMap;
|
||||
// Store the edge count for each ghost vertex
|
||||
vector<MilanLongInt> Counter;
|
||||
// Number of Ghost vertices
|
||||
MilanLongInt numGhostVertices = 0, numGhostEdges = 0;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")About to compute Ghost Vertices...";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0)
|
||||
cout << "\n(" << myRank << ")About to compute Ghost Vertices...";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Define Adjacency Lists for Ghost Vertices:
|
||||
// cout<<"Building Ghost data structures ... \n\n";
|
||||
vector<MilanLongInt> verGhostPtr, verGhostInd, tempCounter;
|
||||
// Mate array for ghost vertices:
|
||||
vector<MilanLongInt> GMate; // Proportional to the number of ghost vertices
|
||||
MilanLongInt S;
|
||||
MilanLongInt privateMyCard = 0;
|
||||
vector<MilanLongInt> PCumulative, PMessageBundle, PSizeInfoMessages;
|
||||
vector<MPI_Request> SRequest; // Requests that are used for each send message
|
||||
vector<MPI_Status> SStatus; // Status of sent messages, used in MPI_Wait
|
||||
MilanLongInt MessageIndex = 0; // Pointer for current message
|
||||
MilanInt BufferSize;
|
||||
MilanLongInt *Buffer;
|
||||
|
||||
vector<MilanLongInt> privateQLocalVtx, privateQGhostVtx, privateQMsgType;
|
||||
vector<MilanInt> privateQOwner;
|
||||
vector<MilanLongInt> U, privateU;
|
||||
|
||||
|
||||
initialize(NLVer, NLEdge, StartIndex,
|
||||
EndIndex, &numGhostEdges,
|
||||
&numGhostVertices, &S,
|
||||
verLocInd, verLocPtr,
|
||||
Ghost2LocalMap, Counter,
|
||||
verGhostPtr, verGhostInd,
|
||||
tempCounter, GMate,
|
||||
Message, QLocalVtx,
|
||||
QGhostVtx, QMsgType, QOwner,
|
||||
candidateMate, U,
|
||||
privateU,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph0_time = finishTime - startTime; // Time taken for Phase-0: Initialization
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished initialization" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
startTime = MPI_Wtime();
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
//////////////////////////////////// INITIALIZATION /////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
// Compute the Initial Matching Set:
|
||||
|
||||
/*
|
||||
* OMP PARALLEL_COMPUTE_CANDIDATE_MATE_B has been splitted from
|
||||
* PARALLEL_PROCESS_EXPOSED_VERTEX_B in order to better parallelize
|
||||
* the two.
|
||||
* PARALLEL_COMPUTE_CANDIDATE_MATE_B is now totally parallel.
|
||||
*/
|
||||
|
||||
PARALLEL_COMPUTE_CANDIDATE_MATE_BS(NLVer,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
myRank,
|
||||
edgeLocWeight,
|
||||
candidateMate);
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Exposed Vertex" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " candidateMate after parallelCompute " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << candidateMate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
/*
|
||||
* PARALLEL_PROCESS_EXPOSED_VERTEX_B
|
||||
* TODO: write comment
|
||||
*
|
||||
* TODO: Test when it's actually more efficient to execute this code
|
||||
* in parallel.
|
||||
*/
|
||||
PARALLEL_PROCESS_EXPOSED_VERTEX_BS(NLVer,
|
||||
candidateMate,
|
||||
verLocInd,
|
||||
verLocPtr,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
Mate,
|
||||
GMate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
U,
|
||||
privateU,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
tempCounter.clear(); // Do not need this any more
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Exposed Vertex" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " Mate after Exposed Vertices " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MATCHED VERTICES //////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
// TODO what would be the optimal UCHUNK
|
||||
vector<MilanLongInt> UChunkBeingProcessed;
|
||||
UChunkBeingProcessed.reserve(UCHUNK);
|
||||
|
||||
processMatchedVerticesS(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
candidateMate,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Process Vertices" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " Mate after Matched Vertices " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
///////////////////////////// SEND BUNDLED MESSAGES /////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
sendBundledMessages(&numGhostEdges,
|
||||
&BufferSize,
|
||||
Buffer,
|
||||
PCumulative,
|
||||
PMessageBundle,
|
||||
PSizeInfoMessages,
|
||||
PCounter,
|
||||
NumMessagesBundled,
|
||||
&msgActual,
|
||||
&MessageIndex,
|
||||
numProcs,
|
||||
myRank,
|
||||
comm,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
SRequest,
|
||||
SStatus);
|
||||
|
||||
///////////////////////// END OF SEND BUNDLED MESSAGES //////////////////////////////////
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph1_time = finishTime - startTime; // Time taken for Phase-1
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished sendBundles" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
*ph1_card = myCard; // Cardinality at the end of Phase-1
|
||||
startTime = MPI_Wtime();
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
//////////////////////////////////////// MAIN LOOP //////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
// Main While Loop:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Entering While(true) loop..";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
while (true) {
|
||||
#ifdef DEBUG_HANG_
|
||||
//if (myRank == 0)
|
||||
cout << "\n(" << myRank << ") Main loop" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MATCHED VERTICES //////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMatchedVerticesAndSendMessagesS(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
candidateMate,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner,
|
||||
comm,
|
||||
&msgActual,
|
||||
Message);
|
||||
|
||||
///////////////////////// END OF PROCESS MATCHED VERTICES /////////////////////////
|
||||
|
||||
//// BREAK IF NO MESSAGES EXPECTED /////////
|
||||
#ifdef DEBUG_HANG_
|
||||
#if 0
|
||||
cout << myRank << " Mate after ProcessMatchedAndSend phase "<<S <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Deciding whether to break: S= " << S << endl;
|
||||
#endif
|
||||
|
||||
if (S == 0) {
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << "\n(" << myRank << ") Breaking out" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MESSAGES //////////////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMessagesS(NLVer,
|
||||
Mate,
|
||||
candidateMate,
|
||||
Ghost2LocalMap,
|
||||
GMate,
|
||||
Counter,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&msgActual,
|
||||
edgeLocWeight,
|
||||
verDistance,
|
||||
verLocPtr,
|
||||
k,
|
||||
verLocInd,
|
||||
numProcs,
|
||||
myRank,
|
||||
comm,
|
||||
Message,
|
||||
numGhostEdges,
|
||||
u,
|
||||
v,
|
||||
&S,
|
||||
U);
|
||||
|
||||
///////////////////////// END OF PROCESS MESSAGES /////////////////////////////////
|
||||
#ifdef DEBUG_HANG_
|
||||
#if 0
|
||||
cout << myRank << " Mate after ProcessMessages phase "<<S <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Finished Message processing phase: S= " << S;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")** SENT : ACTUAL= " << msgActual;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")** SENT : INDIVIDUAL= " << msgInd << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of while (true)
|
||||
|
||||
clean(NLVer,
|
||||
myRank,
|
||||
MessageIndex,
|
||||
SRequest,
|
||||
SStatus,
|
||||
BufferSize,
|
||||
Buffer,
|
||||
msgActual,
|
||||
msgActualSent,
|
||||
msgInd,
|
||||
msgIndSent,
|
||||
NumMessagesBundled,
|
||||
msgPercent);
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph2_time = finishTime - startTime; // Time taken for Phase-2
|
||||
*ph2_card = myCard; // Cardinality at the end of Phase-2
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
@@ -177,23 +177,24 @@ subroutine amg_c_dec_aggregator_mat_bld(ag,parms,a,desc_a,ilaggr,nlaggr,&
|
||||
select case (parms%aggr_prol)
|
||||
case (amg_no_smooth_)
|
||||
|
||||
call amg_caggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,&
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_caggrmat_nosmth_bld(parms%aggr_prol,a,desc_a,ilaggr,&
|
||||
nlaggr,parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case(amg_smooth_prol_)
|
||||
case(amg_smooth_prol_,amg_l1_smooth_prol_)
|
||||
|
||||
call amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_caggrmat_smth_bld(parms%aggr_prol,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,&
|
||||
op_restr,t_prol,info)
|
||||
|
||||
!!$ case(amg_biz_prol_)
|
||||
!!$
|
||||
!!$ call amg_caggrmat_biz_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
!!$ & parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
|
||||
case(amg_min_energy_)
|
||||
|
||||
call amg_caggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_caggrmat_minnrg_bld(parms%aggr_prol,a,desc_a,ilaggr,&
|
||||
nlaggr,parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
|
||||
@@ -216,6 +216,7 @@ subroutine amg_c_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
@@ -420,6 +421,7 @@ subroutine amg_c_lc_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
@@ -629,6 +631,7 @@ subroutine amg_lc_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
|
||||
@@ -142,6 +142,7 @@ subroutine amg_c_rap(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
|
||||
@@ -250,7 +250,7 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! we will not reset.
|
||||
if (j>nr) cycle step1
|
||||
if (ilaggr(j) > 0) cycle step1
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
if ((abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))).and.(diag(i).ne.czero)) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
@@ -275,7 +275,8 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
itmp = (bnds(kk)-1+locnaggr(kk)) !be careful about overflow
|
||||
itmp = itmp*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
@@ -356,7 +357,7 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
if ((1<=j).and.(j<=nr)) then
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
if ((abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))).and.(diag(i).ne.czero)) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
@@ -544,4 +545,3 @@ subroutine amg_c_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
return
|
||||
|
||||
end subroutine amg_c_soc1_map_bld
|
||||
|
||||
|
||||
@@ -309,7 +309,8 @@ subroutine amg_c_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
itmp = (bnds(kk)-1+locnaggr(kk)) !be careful about overflow
|
||||
itmp = itmp*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
|
||||
@@ -69,6 +69,7 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smoothing - fictitious integer argument, it is not used inside
|
||||
! a - type(psb_cspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -104,8 +105,8 @@
|
||||
! Error code.
|
||||
!
|
||||
!
|
||||
subroutine amg_caggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_caggrmat_minnrg_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_c_inner_mod, amg_protect_name => amg_caggrmat_minnrg_bld
|
||||
@@ -113,6 +114,7 @@ subroutine amg_caggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_cspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -171,6 +173,13 @@ subroutine amg_caggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_)
|
||||
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
!NEEDS TO BE REWORKED !!
|
||||
|
||||
! naggr: number of local aggregates
|
||||
@@ -300,7 +309,7 @@ subroutine amg_caggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ endif
|
||||
!!$ enddo
|
||||
!!$ if (jd == -1) then
|
||||
!!$ write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
!!$ write(0,*) name,': Warning: there is no diagonal element', i
|
||||
!!$ else
|
||||
!!$ acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
!!$ end if
|
||||
|
||||
@@ -94,10 +94,11 @@
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
! dol1smoothing - optional, this is here just for interfacing reasons. It is not used by the
|
||||
! code
|
||||
!
|
||||
!
|
||||
subroutine amg_caggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_caggrmat_nosmth_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_c_inner_mod, amg_protect_name => amg_caggrmat_nosmth_bld
|
||||
@@ -105,6 +106,7 @@ subroutine amg_caggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_cspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -137,6 +139,12 @@ subroutine amg_caggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt, me, np)
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
@@ -69,6 +69,8 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smooth - Integer taking the type of smoother that has to be used
|
||||
! on the tentative prolongator
|
||||
! a - type(psb_cspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -102,16 +104,18 @@
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_caggrmat_smth_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_c_inner_mod, amg_protect_name => amg_caggrmat_smth_bld
|
||||
use amg_c_base_aggregator_mod
|
||||
! use, intrinsic :: ieee_arithmetic
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_cspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -132,7 +136,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
type(psb_c_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
type(psb_c_csr_sparse_mat) :: acsr1, acsrf, csr_prol, acsr
|
||||
complex(psb_spk_), allocatable :: adiag(:)
|
||||
real(psb_spk_), allocatable :: arwsum(:)
|
||||
real(psb_spk_), allocatable :: arwsum(:),l1rwsum(:)
|
||||
integer(psb_ipk_) :: ierr(5)
|
||||
logical :: filter_mat
|
||||
integer(psb_ipk_) :: debug_level, debug_unit, err_act
|
||||
@@ -141,6 +145,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
logical :: do_l1correction=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
@@ -173,6 +178,9 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if ((do_timings).and.(idx_ptap==-1)) &
|
||||
& idx_ptap = psb_get_timer_idx("DEC_SMTH_BLD: ptap_bld ")
|
||||
|
||||
! check if we have to use Jacobi or l1-Jacobi to smooth the tentative prolongator
|
||||
if (dol1smoothing.eq.amg_l1_smooth_prol_) do_l1correction=.true.
|
||||
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
@@ -185,7 +193,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
naggrm1 = sum(nlaggr(1:me))
|
||||
naggrp1 = sum(nlaggr(1:me+1))
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_)
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_).or.(parms%aggr_filter == amg_filter_prow_mat_)
|
||||
|
||||
!
|
||||
! naggr: number of local aggregates
|
||||
@@ -200,6 +208,24 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(adiag,desc_a,info)
|
||||
if (info == psb_success_) call a%cp_to(acsr)
|
||||
!
|
||||
! Do the l1-correction on the diagonal if it is requested
|
||||
!
|
||||
if (do_l1correction) then
|
||||
allocate(l1rwsum(nrow))
|
||||
call acsr%arwsum(l1rwsum)
|
||||
if (info == psb_success_) &
|
||||
& call psb_realloc(ncol,l1rwsum,info)
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(l1rwsum,desc_a,info)
|
||||
! \tilde{D}_{i,i} = \sum_{j \ne i} |a_{i,j}|
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
adiag(i) = adiag(i) + l1rwsum(i) - abs(adiag(i))
|
||||
end do
|
||||
!$OMP end parallel do
|
||||
end if
|
||||
|
||||
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='sp_getdiag')
|
||||
@@ -230,9 +256,15 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
! if (.not.do_l1correction)
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
else if (parms%aggr_filter == amg_filter_mat_) then
|
||||
! We perform filtering in the standard way assuming that A is an M-matrix
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
else if (parms%aggr_filter == amg_filter_prow_mat_) then
|
||||
! We are probably doing l1-correction, hence we want to preserve the
|
||||
! row sum of the matrix: note the change in sign
|
||||
acsrf%val(jd)=acsrf%val(jd)+tmp
|
||||
end if
|
||||
enddo
|
||||
!$OMP end parallel do
|
||||
@@ -240,7 +272,6 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= czero) then
|
||||
@@ -252,14 +283,17 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!$OMP end parallel do
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
if ( (parms%aggr_filter == amg_filter_prow_mat_).and.(do_l1correction) ) then
|
||||
! For l1-Jacobi this can be estimated with 1:
|
||||
! this makes sense only if we are preserving the row-sum!
|
||||
parms%aggr_omega_val = done
|
||||
else if (parms%aggr_eig == amg_max_norm_) then
|
||||
allocate(arwsum(nrow))
|
||||
call acsr%arwsum(arwsum)
|
||||
anorm = maxval(abs(adiag(1:nrow)*arwsum(1:nrow)))
|
||||
call psb_amx(ctxt,anorm)
|
||||
omega = 4.d0/(3.d0*anorm)
|
||||
parms%aggr_omega_val = omega
|
||||
|
||||
else
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_eig_')
|
||||
@@ -322,6 +356,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
if (allocated(l1rwsum)) deallocate(l1rwsum)
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
|
||||
@@ -177,23 +177,24 @@ subroutine amg_d_dec_aggregator_mat_bld(ag,parms,a,desc_a,ilaggr,nlaggr,&
|
||||
select case (parms%aggr_prol)
|
||||
case (amg_no_smooth_)
|
||||
|
||||
call amg_daggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,&
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_daggrmat_nosmth_bld(parms%aggr_prol,a,desc_a,ilaggr,&
|
||||
nlaggr,parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case(amg_smooth_prol_)
|
||||
case(amg_smooth_prol_,amg_l1_smooth_prol_)
|
||||
|
||||
call amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_daggrmat_smth_bld(parms%aggr_prol,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,&
|
||||
op_restr,t_prol,info)
|
||||
|
||||
!!$ case(amg_biz_prol_)
|
||||
!!$
|
||||
!!$ call amg_daggrmat_biz_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
!!$ & parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
|
||||
case(amg_min_energy_)
|
||||
|
||||
call amg_daggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_daggrmat_minnrg_bld(parms%aggr_prol,a,desc_a,ilaggr,&
|
||||
nlaggr,parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
|
||||
@@ -184,20 +184,23 @@ subroutine amg_d_parmatch_aggregator_mat_bld(ag,parms,a,desc_a,ilaggr,nlaggr,&
|
||||
!
|
||||
select case (parms%aggr_prol)
|
||||
case (amg_no_smooth_)
|
||||
call amg_d_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_d_parmatch_unsmth_bld(parms%aggr_prol,ag,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,op_restr,&
|
||||
t_prol,info)
|
||||
|
||||
case(amg_smooth_prol_)
|
||||
call amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
case(amg_smooth_prol_,amg_l1_smooth_prol_)
|
||||
call amg_d_parmatch_smth_bld(parms%aggr_prol,ag,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,op_restr,&
|
||||
t_prol,info)
|
||||
|
||||
!!$ case(amg_biz_prol_)
|
||||
!!$ call amg_daggrmat_biz_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
!!$ & parms,ac,desc_ac,op_prol,op_restr,info)
|
||||
|
||||
case(amg_min_energy_)
|
||||
call amg_daggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_daggrmat_minnrg_bld(parms%aggr_prol,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,op_restr,&
|
||||
t_prol,info)
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
|
||||
@@ -69,6 +69,8 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smoothing - Select between l1-Jacobi and Jacobi as smoother for the
|
||||
! tentative prolongator
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -102,8 +104,8 @@
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_d_parmatch_smth_bld(dol1smoothing,ag,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_inner_mod
|
||||
@@ -116,6 +118,7 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
class(amg_d_parmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
@@ -137,7 +140,7 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
type(psb_d_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
type(psb_d_csr_sparse_mat) :: acsrf, csr_prol, acsr, tcsr
|
||||
real(psb_dpk_), allocatable :: adiag(:)
|
||||
real(psb_dpk_), allocatable :: arwsum(:)
|
||||
real(psb_dpk_), allocatable :: arwsum(:),l1rwsum(:)
|
||||
logical :: filter_mat
|
||||
integer(psb_ipk_) :: debug_level, debug_unit, err_act
|
||||
integer(psb_ipk_), parameter :: ncmax=16
|
||||
@@ -145,6 +148,7 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
logical, parameter :: debug_new=.false., dump_r=.false., dump_p=.false., debug=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
logical :: do_l1correction=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1, idx_phase3=-1
|
||||
integer(psb_ipk_), save :: idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
@@ -166,6 +170,10 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
theta = parms%aggr_thresh
|
||||
! Check if we have to perform l1-Jacobi or Jacobi as smoother
|
||||
if(dol1smoothing.eq.amg_l1_smooth_prol_) do_l1correction=.true.
|
||||
|
||||
|
||||
!write(0,*) me,' ',trim(name),' Start ',idx_spspmm
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("PMC_SMTH_BLD: par_spspmm")
|
||||
@@ -217,6 +225,19 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(adiag,desc_a,info)
|
||||
if (info == psb_success_) call a%cp_to(acsr)
|
||||
! Get the l1-diagonal of D
|
||||
if (do_l1correction) then
|
||||
allocate(l1rwsum(nrow))
|
||||
call acsr%arwsum(l1rwsum)
|
||||
if (info == psb_success_) &
|
||||
& call psb_realloc(ncol,l1rwsum,info)
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(l1rwsum,desc_a,info)
|
||||
! \tilde{D}_{i,i} = \sum_{j \ne i} |a_{i,j}|
|
||||
do i=1,size(adiag)
|
||||
adiag(i) = adiag(i) + l1rwsum(i) - abs(adiag(i))
|
||||
end do
|
||||
end if
|
||||
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='sp_getdiag')
|
||||
@@ -246,7 +267,7 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
write(0,*) name,': Warning: there is no diagonal element', i
|
||||
else
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
@@ -267,7 +288,10 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
if (do_l1correction) then
|
||||
! For l1-Jacobi this can be estimated with 1
|
||||
parms%aggr_omega_val = done
|
||||
else if (parms%aggr_eig == amg_max_norm_) then
|
||||
allocate(arwsum(nrow))
|
||||
call acsr%arwsum(arwsum)
|
||||
anorm = maxval(abs(adiag(1:nrow)*arwsum(1:nrow)))
|
||||
@@ -373,6 +397,7 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
end block
|
||||
end if
|
||||
if (allocated(l1rwsum)) deallocate(l1rwsum)
|
||||
if (do_timings) call psb_toc(idx_phase2)
|
||||
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
|
||||
@@ -68,6 +68,8 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smoothing - this not actually used inside unsmoothed aggregation, it
|
||||
! is used just to perform a check
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -101,8 +103,8 @@
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_d_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_d_parmatch_unsmth_bld(dol1smoothing,ag,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_inner_mod
|
||||
@@ -115,6 +117,7 @@ subroutine amg_d_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
class(amg_d_parmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
@@ -159,6 +162,11 @@ subroutine amg_d_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
ictxt = desc_a%get_context()
|
||||
|
||||
call psb_info(ictxt, me, np)
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
nglob = desc_a%get_global_rows()
|
||||
|
||||
@@ -216,6 +216,7 @@ subroutine amg_d_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
@@ -420,6 +421,7 @@ subroutine amg_d_ld_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
@@ -629,6 +631,7 @@ subroutine amg_ld_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
|
||||
@@ -142,6 +142,7 @@ subroutine amg_d_rap(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
|
||||
@@ -250,7 +250,7 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! we will not reset.
|
||||
if (j>nr) cycle step1
|
||||
if (ilaggr(j) > 0) cycle step1
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
if ((abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))).and.(diag(i).ne.dzero)) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
@@ -275,7 +275,8 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
itmp = (bnds(kk)-1+locnaggr(kk)) !be careful about overflow
|
||||
itmp = itmp*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
@@ -356,7 +357,7 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
if ((1<=j).and.(j<=nr)) then
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
if ((abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))).and.(diag(i).ne.dzero)) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
@@ -544,4 +545,3 @@ subroutine amg_d_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
return
|
||||
|
||||
end subroutine amg_d_soc1_map_bld
|
||||
|
||||
|
||||
@@ -309,7 +309,8 @@ subroutine amg_d_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
itmp = (bnds(kk)-1+locnaggr(kk)) !be careful about overflow
|
||||
itmp = itmp*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
|
||||
@@ -69,6 +69,7 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smoothing - fictitious integer argument, it is not used inside
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -104,8 +105,8 @@
|
||||
! Error code.
|
||||
!
|
||||
!
|
||||
subroutine amg_daggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_daggrmat_minnrg_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_inner_mod, amg_protect_name => amg_daggrmat_minnrg_bld
|
||||
@@ -113,6 +114,7 @@ subroutine amg_daggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -171,6 +173,13 @@ subroutine amg_daggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_)
|
||||
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
!NEEDS TO BE REWORKED !!
|
||||
|
||||
! naggr: number of local aggregates
|
||||
@@ -300,7 +309,7 @@ subroutine amg_daggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ endif
|
||||
!!$ enddo
|
||||
!!$ if (jd == -1) then
|
||||
!!$ write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
!!$ write(0,*) name,': Warning: there is no diagonal element', i
|
||||
!!$ else
|
||||
!!$ acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
!!$ end if
|
||||
|
||||
@@ -94,10 +94,11 @@
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
! dol1smoothing - optional, this is here just for interfacing reasons. It is not used by the
|
||||
! code
|
||||
!
|
||||
!
|
||||
subroutine amg_daggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_daggrmat_nosmth_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_inner_mod, amg_protect_name => amg_daggrmat_nosmth_bld
|
||||
@@ -105,6 +106,7 @@ subroutine amg_daggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -137,6 +139,12 @@ subroutine amg_daggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt, me, np)
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
@@ -69,6 +69,8 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smooth - Integer taking the type of smoother that has to be used
|
||||
! on the tentative prolongator
|
||||
! a - type(psb_dspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -102,16 +104,18 @@
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_daggrmat_smth_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_d_inner_mod, amg_protect_name => amg_daggrmat_smth_bld
|
||||
use amg_d_base_aggregator_mod
|
||||
! use, intrinsic :: ieee_arithmetic
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -132,7 +136,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
type(psb_d_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
type(psb_d_csr_sparse_mat) :: acsr1, acsrf, csr_prol, acsr
|
||||
real(psb_dpk_), allocatable :: adiag(:)
|
||||
real(psb_dpk_), allocatable :: arwsum(:)
|
||||
real(psb_dpk_), allocatable :: arwsum(:),l1rwsum(:)
|
||||
integer(psb_ipk_) :: ierr(5)
|
||||
logical :: filter_mat
|
||||
integer(psb_ipk_) :: debug_level, debug_unit, err_act
|
||||
@@ -141,6 +145,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
logical :: do_l1correction=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
@@ -173,6 +178,9 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if ((do_timings).and.(idx_ptap==-1)) &
|
||||
& idx_ptap = psb_get_timer_idx("DEC_SMTH_BLD: ptap_bld ")
|
||||
|
||||
! check if we have to use Jacobi or l1-Jacobi to smooth the tentative prolongator
|
||||
if (dol1smoothing.eq.amg_l1_smooth_prol_) do_l1correction=.true.
|
||||
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
@@ -185,7 +193,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
naggrm1 = sum(nlaggr(1:me))
|
||||
naggrp1 = sum(nlaggr(1:me+1))
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_)
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_).or.(parms%aggr_filter == amg_filter_prow_mat_)
|
||||
|
||||
!
|
||||
! naggr: number of local aggregates
|
||||
@@ -200,6 +208,24 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(adiag,desc_a,info)
|
||||
if (info == psb_success_) call a%cp_to(acsr)
|
||||
!
|
||||
! Do the l1-correction on the diagonal if it is requested
|
||||
!
|
||||
if (do_l1correction) then
|
||||
allocate(l1rwsum(nrow))
|
||||
call acsr%arwsum(l1rwsum)
|
||||
if (info == psb_success_) &
|
||||
& call psb_realloc(ncol,l1rwsum,info)
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(l1rwsum,desc_a,info)
|
||||
! \tilde{D}_{i,i} = \sum_{j \ne i} |a_{i,j}|
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
adiag(i) = adiag(i) + l1rwsum(i) - abs(adiag(i))
|
||||
end do
|
||||
!$OMP end parallel do
|
||||
end if
|
||||
|
||||
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='sp_getdiag')
|
||||
@@ -230,9 +256,15 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
! if (.not.do_l1correction)
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
else if (parms%aggr_filter == amg_filter_mat_) then
|
||||
! We perform filtering in the standard way assuming that A is an M-matrix
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
else if (parms%aggr_filter == amg_filter_prow_mat_) then
|
||||
! We are probably doing l1-correction, hence we want to preserve the
|
||||
! row sum of the matrix: note the change in sign
|
||||
acsrf%val(jd)=acsrf%val(jd)+tmp
|
||||
end if
|
||||
enddo
|
||||
!$OMP end parallel do
|
||||
@@ -240,7 +272,6 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= dzero) then
|
||||
@@ -252,14 +283,17 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!$OMP end parallel do
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
if ( (parms%aggr_filter == amg_filter_prow_mat_).and.(do_l1correction) ) then
|
||||
! For l1-Jacobi this can be estimated with 1:
|
||||
! this makes sense only if we are preserving the row-sum!
|
||||
parms%aggr_omega_val = done
|
||||
else if (parms%aggr_eig == amg_max_norm_) then
|
||||
allocate(arwsum(nrow))
|
||||
call acsr%arwsum(arwsum)
|
||||
anorm = maxval(abs(adiag(1:nrow)*arwsum(1:nrow)))
|
||||
call psb_amx(ctxt,anorm)
|
||||
omega = 4.d0/(3.d0*anorm)
|
||||
parms%aggr_omega_val = omega
|
||||
|
||||
else
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_eig_')
|
||||
@@ -322,6 +356,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
if (allocated(l1rwsum)) deallocate(l1rwsum)
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
|
||||
@@ -177,23 +177,24 @@ subroutine amg_s_dec_aggregator_mat_bld(ag,parms,a,desc_a,ilaggr,nlaggr,&
|
||||
select case (parms%aggr_prol)
|
||||
case (amg_no_smooth_)
|
||||
|
||||
call amg_saggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,&
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_saggrmat_nosmth_bld(parms%aggr_prol,a,desc_a,ilaggr,&
|
||||
nlaggr,parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case(amg_smooth_prol_)
|
||||
case(amg_smooth_prol_,amg_l1_smooth_prol_)
|
||||
|
||||
call amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_saggrmat_smth_bld(parms%aggr_prol,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,&
|
||||
op_restr,t_prol,info)
|
||||
|
||||
!!$ case(amg_biz_prol_)
|
||||
!!$
|
||||
!!$ call amg_saggrmat_biz_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
!!$ & parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
|
||||
case(amg_min_energy_)
|
||||
|
||||
call amg_saggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_saggrmat_minnrg_bld(parms%aggr_prol,a,desc_a,ilaggr,&
|
||||
nlaggr,parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
|
||||
@@ -184,20 +184,23 @@ subroutine amg_s_parmatch_aggregator_mat_bld(ag,parms,a,desc_a,ilaggr,nlaggr,&
|
||||
!
|
||||
select case (parms%aggr_prol)
|
||||
case (amg_no_smooth_)
|
||||
call amg_s_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_s_parmatch_unsmth_bld(parms%aggr_prol,ag,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,op_restr,&
|
||||
t_prol,info)
|
||||
|
||||
case(amg_smooth_prol_)
|
||||
call amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
case(amg_smooth_prol_,amg_l1_smooth_prol_)
|
||||
call amg_s_parmatch_smth_bld(parms%aggr_prol,ag,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,op_restr,&
|
||||
t_prol,info)
|
||||
|
||||
!!$ case(amg_biz_prol_)
|
||||
!!$ call amg_saggrmat_biz_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
!!$ & parms,ac,desc_ac,op_prol,op_restr,info)
|
||||
|
||||
case(amg_min_energy_)
|
||||
call amg_saggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_saggrmat_minnrg_bld(parms%aggr_prol,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,op_restr,&
|
||||
t_prol,info)
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
|
||||
@@ -69,6 +69,8 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smoothing - Select between l1-Jacobi and Jacobi as smoother for the
|
||||
! tentative prolongator
|
||||
! a - type(psb_sspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -102,8 +104,8 @@
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_s_parmatch_smth_bld(dol1smoothing,ag,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_s_inner_mod
|
||||
@@ -116,6 +118,7 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
class(amg_s_parmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(psb_sspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
@@ -137,7 +140,7 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
type(psb_s_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
type(psb_s_csr_sparse_mat) :: acsrf, csr_prol, acsr, tcsr
|
||||
real(psb_spk_), allocatable :: adiag(:)
|
||||
real(psb_spk_), allocatable :: arwsum(:)
|
||||
real(psb_spk_), allocatable :: arwsum(:),l1rwsum(:)
|
||||
logical :: filter_mat
|
||||
integer(psb_ipk_) :: debug_level, debug_unit, err_act
|
||||
integer(psb_ipk_), parameter :: ncmax=16
|
||||
@@ -145,6 +148,7 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
logical, parameter :: debug_new=.false., dump_r=.false., dump_p=.false., debug=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
logical :: do_l1correction=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1, idx_phase3=-1
|
||||
integer(psb_ipk_), save :: idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
@@ -166,6 +170,10 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
theta = parms%aggr_thresh
|
||||
! Check if we have to perform l1-Jacobi or Jacobi as smoother
|
||||
if(dol1smoothing.eq.amg_l1_smooth_prol_) do_l1correction=.true.
|
||||
|
||||
|
||||
!write(0,*) me,' ',trim(name),' Start ',idx_spspmm
|
||||
if ((do_timings).and.(idx_spspmm==-1)) &
|
||||
& idx_spspmm = psb_get_timer_idx("PMC_SMTH_BLD: par_spspmm")
|
||||
@@ -217,6 +225,19 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(adiag,desc_a,info)
|
||||
if (info == psb_success_) call a%cp_to(acsr)
|
||||
! Get the l1-diagonal of D
|
||||
if (do_l1correction) then
|
||||
allocate(l1rwsum(nrow))
|
||||
call acsr%arwsum(l1rwsum)
|
||||
if (info == psb_success_) &
|
||||
& call psb_realloc(ncol,l1rwsum,info)
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(l1rwsum,desc_a,info)
|
||||
! \tilde{D}_{i,i} = \sum_{j \ne i} |a_{i,j}|
|
||||
do i=1,size(adiag)
|
||||
adiag(i) = adiag(i) + l1rwsum(i) - abs(adiag(i))
|
||||
end do
|
||||
end if
|
||||
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='sp_getdiag')
|
||||
@@ -246,7 +267,7 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
write(0,*) name,': Warning: there is no diagonal element', i
|
||||
else
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
@@ -267,7 +288,10 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
if (do_l1correction) then
|
||||
! For l1-Jacobi this can be estimated with 1
|
||||
parms%aggr_omega_val = done
|
||||
else if (parms%aggr_eig == amg_max_norm_) then
|
||||
allocate(arwsum(nrow))
|
||||
call acsr%arwsum(arwsum)
|
||||
anorm = maxval(abs(adiag(1:nrow)*arwsum(1:nrow)))
|
||||
@@ -373,6 +397,7 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
end block
|
||||
end if
|
||||
if (allocated(l1rwsum)) deallocate(l1rwsum)
|
||||
if (do_timings) call psb_toc(idx_phase2)
|
||||
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
|
||||
@@ -68,6 +68,8 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smoothing - this not actually used inside unsmoothed aggregation, it
|
||||
! is used just to perform a check
|
||||
! a - type(psb_sspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -101,8 +103,8 @@
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_s_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_s_parmatch_unsmth_bld(dol1smoothing,ag,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_s_inner_mod
|
||||
@@ -115,6 +117,7 @@ subroutine amg_s_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
class(amg_s_parmatch_aggregator_type), target, intent(inout) :: ag
|
||||
type(psb_sspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
@@ -159,6 +162,11 @@ subroutine amg_s_parmatch_unsmth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
ictxt = desc_a%get_context()
|
||||
|
||||
call psb_info(ictxt, me, np)
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
nglob = desc_a%get_global_rows()
|
||||
|
||||
@@ -216,6 +216,7 @@ subroutine amg_s_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
@@ -420,6 +421,7 @@ subroutine amg_s_ls_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
@@ -629,6 +631,7 @@ subroutine amg_ls_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
|
||||
@@ -142,6 +142,7 @@ subroutine amg_s_rap(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
|
||||
@@ -250,7 +250,7 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! we will not reset.
|
||||
if (j>nr) cycle step1
|
||||
if (ilaggr(j) > 0) cycle step1
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
if ((abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))).and.(diag(i).ne.szero)) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
@@ -275,7 +275,8 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
itmp = (bnds(kk)-1+locnaggr(kk)) !be careful about overflow
|
||||
itmp = itmp*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
@@ -356,7 +357,7 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
if ((1<=j).and.(j<=nr)) then
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
if ((abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))).and.(diag(i).ne.szero)) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
@@ -544,4 +545,3 @@ subroutine amg_s_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
return
|
||||
|
||||
end subroutine amg_s_soc1_map_bld
|
||||
|
||||
|
||||
@@ -309,7 +309,8 @@ subroutine amg_s_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
itmp = (bnds(kk)-1+locnaggr(kk)) !be careful about overflow
|
||||
itmp = itmp*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
|
||||
@@ -69,6 +69,7 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smoothing - fictitious integer argument, it is not used inside
|
||||
! a - type(psb_sspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -104,8 +105,8 @@
|
||||
! Error code.
|
||||
!
|
||||
!
|
||||
subroutine amg_saggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_saggrmat_minnrg_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_s_inner_mod, amg_protect_name => amg_saggrmat_minnrg_bld
|
||||
@@ -113,6 +114,7 @@ subroutine amg_saggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_sspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -171,6 +173,13 @@ subroutine amg_saggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_)
|
||||
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
!NEEDS TO BE REWORKED !!
|
||||
|
||||
! naggr: number of local aggregates
|
||||
@@ -300,7 +309,7 @@ subroutine amg_saggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ endif
|
||||
!!$ enddo
|
||||
!!$ if (jd == -1) then
|
||||
!!$ write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
!!$ write(0,*) name,': Warning: there is no diagonal element', i
|
||||
!!$ else
|
||||
!!$ acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
!!$ end if
|
||||
|
||||
@@ -94,10 +94,11 @@
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
! dol1smoothing - optional, this is here just for interfacing reasons. It is not used by the
|
||||
! code
|
||||
!
|
||||
!
|
||||
subroutine amg_saggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_saggrmat_nosmth_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_s_inner_mod, amg_protect_name => amg_saggrmat_nosmth_bld
|
||||
@@ -105,6 +106,7 @@ subroutine amg_saggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_sspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -137,6 +139,12 @@ subroutine amg_saggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt, me, np)
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
@@ -69,6 +69,8 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smooth - Integer taking the type of smoother that has to be used
|
||||
! on the tentative prolongator
|
||||
! a - type(psb_sspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -102,16 +104,18 @@
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_saggrmat_smth_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_s_inner_mod, amg_protect_name => amg_saggrmat_smth_bld
|
||||
use amg_s_base_aggregator_mod
|
||||
! use, intrinsic :: ieee_arithmetic
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_sspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -132,7 +136,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
type(psb_s_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
type(psb_s_csr_sparse_mat) :: acsr1, acsrf, csr_prol, acsr
|
||||
real(psb_spk_), allocatable :: adiag(:)
|
||||
real(psb_spk_), allocatable :: arwsum(:)
|
||||
real(psb_spk_), allocatable :: arwsum(:),l1rwsum(:)
|
||||
integer(psb_ipk_) :: ierr(5)
|
||||
logical :: filter_mat
|
||||
integer(psb_ipk_) :: debug_level, debug_unit, err_act
|
||||
@@ -141,6 +145,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
logical :: do_l1correction=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
@@ -173,6 +178,9 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if ((do_timings).and.(idx_ptap==-1)) &
|
||||
& idx_ptap = psb_get_timer_idx("DEC_SMTH_BLD: ptap_bld ")
|
||||
|
||||
! check if we have to use Jacobi or l1-Jacobi to smooth the tentative prolongator
|
||||
if (dol1smoothing.eq.amg_l1_smooth_prol_) do_l1correction=.true.
|
||||
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
@@ -185,7 +193,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
naggrm1 = sum(nlaggr(1:me))
|
||||
naggrp1 = sum(nlaggr(1:me+1))
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_)
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_).or.(parms%aggr_filter == amg_filter_prow_mat_)
|
||||
|
||||
!
|
||||
! naggr: number of local aggregates
|
||||
@@ -200,6 +208,24 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(adiag,desc_a,info)
|
||||
if (info == psb_success_) call a%cp_to(acsr)
|
||||
!
|
||||
! Do the l1-correction on the diagonal if it is requested
|
||||
!
|
||||
if (do_l1correction) then
|
||||
allocate(l1rwsum(nrow))
|
||||
call acsr%arwsum(l1rwsum)
|
||||
if (info == psb_success_) &
|
||||
& call psb_realloc(ncol,l1rwsum,info)
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(l1rwsum,desc_a,info)
|
||||
! \tilde{D}_{i,i} = \sum_{j \ne i} |a_{i,j}|
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
adiag(i) = adiag(i) + l1rwsum(i) - abs(adiag(i))
|
||||
end do
|
||||
!$OMP end parallel do
|
||||
end if
|
||||
|
||||
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='sp_getdiag')
|
||||
@@ -230,9 +256,15 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
! if (.not.do_l1correction)
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
else if (parms%aggr_filter == amg_filter_mat_) then
|
||||
! We perform filtering in the standard way assuming that A is an M-matrix
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
else if (parms%aggr_filter == amg_filter_prow_mat_) then
|
||||
! We are probably doing l1-correction, hence we want to preserve the
|
||||
! row sum of the matrix: note the change in sign
|
||||
acsrf%val(jd)=acsrf%val(jd)+tmp
|
||||
end if
|
||||
enddo
|
||||
!$OMP end parallel do
|
||||
@@ -240,7 +272,6 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= szero) then
|
||||
@@ -252,14 +283,17 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!$OMP end parallel do
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
if ( (parms%aggr_filter == amg_filter_prow_mat_).and.(do_l1correction) ) then
|
||||
! For l1-Jacobi this can be estimated with 1:
|
||||
! this makes sense only if we are preserving the row-sum!
|
||||
parms%aggr_omega_val = done
|
||||
else if (parms%aggr_eig == amg_max_norm_) then
|
||||
allocate(arwsum(nrow))
|
||||
call acsr%arwsum(arwsum)
|
||||
anorm = maxval(abs(adiag(1:nrow)*arwsum(1:nrow)))
|
||||
call psb_amx(ctxt,anorm)
|
||||
omega = 4.d0/(3.d0*anorm)
|
||||
parms%aggr_omega_val = omega
|
||||
|
||||
else
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_eig_')
|
||||
@@ -322,6 +356,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
if (allocated(l1rwsum)) deallocate(l1rwsum)
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
|
||||
@@ -177,23 +177,24 @@ subroutine amg_z_dec_aggregator_mat_bld(ag,parms,a,desc_a,ilaggr,nlaggr,&
|
||||
select case (parms%aggr_prol)
|
||||
case (amg_no_smooth_)
|
||||
|
||||
call amg_zaggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,&
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_zaggrmat_nosmth_bld(parms%aggr_prol,a,desc_a,ilaggr,&
|
||||
nlaggr,parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case(amg_smooth_prol_)
|
||||
case(amg_smooth_prol_,amg_l1_smooth_prol_)
|
||||
|
||||
call amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_zaggrmat_smth_bld(parms%aggr_prol,a,desc_a,&
|
||||
ilaggr,nlaggr,parms,ac,desc_ac,op_prol,&
|
||||
op_restr,t_prol,info)
|
||||
|
||||
!!$ case(amg_biz_prol_)
|
||||
!!$
|
||||
!!$ call amg_zaggrmat_biz_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
!!$ & parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
|
||||
case(amg_min_energy_)
|
||||
|
||||
call amg_zaggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr, &
|
||||
& parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
call amg_zaggrmat_minnrg_bld(parms%aggr_prol,a,desc_a,ilaggr,&
|
||||
nlaggr,parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
|
||||
@@ -216,6 +216,7 @@ subroutine amg_z_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
@@ -420,6 +421,7 @@ subroutine amg_z_lz_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
@@ -629,6 +631,7 @@ subroutine amg_lz_ptap_bld(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
|
||||
@@ -142,6 +142,7 @@ subroutine amg_z_rap(a_csr,desc_a,nlaggr,parms,ac,&
|
||||
|
||||
call ac_csr%set_nrows(desc_ac%get_local_rows())
|
||||
call ac_csr%set_ncols(desc_ac%get_local_cols())
|
||||
call ac_csr%clean_zeros(info)
|
||||
call ac%mv_from(ac_csr)
|
||||
call ac%set_asb()
|
||||
|
||||
|
||||
@@ -250,7 +250,7 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
! we will not reset.
|
||||
if (j>nr) cycle step1
|
||||
if (ilaggr(j) > 0) cycle step1
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
if ((abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))).and.(diag(i).ne.zzero)) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
@@ -275,7 +275,8 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
disjoint = all(ilaggr(icol(1:ip)) == -(nr+1)).or.(ip==0)
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
itmp = (bnds(kk)-1+locnaggr(kk)) !be careful about overflow
|
||||
itmp = itmp*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
@@ -356,7 +357,7 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
do k=1, nz
|
||||
j = icol(k)
|
||||
if ((1<=j).and.(j<=nr)) then
|
||||
if (abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))) then
|
||||
if ((abs(val(k)) > theta*sqrt(abs(diag(i)*diag(j)))).and.(diag(i).ne.zzero)) then
|
||||
ip = ip + 1
|
||||
icol(ip) = icol(k)
|
||||
end if
|
||||
@@ -544,4 +545,3 @@ subroutine amg_z_soc1_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
return
|
||||
|
||||
end subroutine amg_z_soc1_map_bld
|
||||
|
||||
|
||||
@@ -309,7 +309,8 @@ subroutine amg_z_soc2_map_bld(iorder,theta,clean_zeros,a,desc_a,nlaggr,ilaggr,in
|
||||
!
|
||||
if (disjoint) then
|
||||
locnaggr(kk) = locnaggr(kk) + 1
|
||||
itmp = (bnds(kk)-1+locnaggr(kk))*nths+kk
|
||||
itmp = (bnds(kk)-1+locnaggr(kk)) !be careful about overflow
|
||||
itmp = itmp*nths+kk
|
||||
if (itmp < (bnds(kk)-1+locnaggr(kk))) then
|
||||
!$omp atomic update
|
||||
info = max(12345678,info)
|
||||
|
||||
@@ -69,6 +69,7 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smoothing - fictitious integer argument, it is not used inside
|
||||
! a - type(psb_zspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -104,8 +105,8 @@
|
||||
! Error code.
|
||||
!
|
||||
!
|
||||
subroutine amg_zaggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_zaggrmat_minnrg_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_z_inner_mod, amg_protect_name => amg_zaggrmat_minnrg_bld
|
||||
@@ -113,6 +114,7 @@ subroutine amg_zaggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_zspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -171,6 +173,13 @@ subroutine amg_zaggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_)
|
||||
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
|
||||
!NEEDS TO BE REWORKED !!
|
||||
|
||||
! naggr: number of local aggregates
|
||||
@@ -300,7 +309,7 @@ subroutine amg_zaggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ endif
|
||||
!!$ enddo
|
||||
!!$ if (jd == -1) then
|
||||
!!$ write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
!!$ write(0,*) name,': Warning: there is no diagonal element', i
|
||||
!!$ else
|
||||
!!$ acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
!!$ end if
|
||||
|
||||
@@ -94,10 +94,11 @@
|
||||
!
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
! dol1smoothing - optional, this is here just for interfacing reasons. It is not used by the
|
||||
! code
|
||||
!
|
||||
!
|
||||
subroutine amg_zaggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_zaggrmat_nosmth_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_z_inner_mod, amg_protect_name => amg_zaggrmat_nosmth_bld
|
||||
@@ -105,6 +106,7 @@ subroutine amg_zaggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_zspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -137,6 +139,12 @@ subroutine amg_zaggrmat_nosmth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
ctxt = desc_a%get_context()
|
||||
call psb_info(ctxt, me, np)
|
||||
if (dol1smoothing.ne.amg_no_smooth_) then
|
||||
info=psb_err_fatal_;
|
||||
call psb_errpush(info,name,a_err='Are you trying to smooth an unsmoothed aggregation?')
|
||||
goto 9999
|
||||
end if
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
ncol = desc_a%get_local_cols()
|
||||
|
||||
@@ -69,6 +69,8 @@
|
||||
!
|
||||
!
|
||||
! Arguments:
|
||||
! dol1smooth - Integer taking the type of smoother that has to be used
|
||||
! on the tentative prolongator
|
||||
! a - type(psb_zspmat_type), input.
|
||||
! The sparse matrix structure containing the local part of
|
||||
! the fine-level matrix.
|
||||
@@ -102,16 +104,18 @@
|
||||
! info - integer, output.
|
||||
! Error code.
|
||||
!
|
||||
subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
subroutine amg_zaggrmat_smth_bld(dol1smoothing,a,desc_a,ilaggr,nlaggr,&
|
||||
parms,ac,desc_ac,op_prol,op_restr,t_prol,info)
|
||||
use psb_base_mod
|
||||
use amg_base_prec_type
|
||||
use amg_z_inner_mod, amg_protect_name => amg_zaggrmat_smth_bld
|
||||
use amg_z_base_aggregator_mod
|
||||
! use, intrinsic :: ieee_arithmetic
|
||||
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
integer(psb_ipk_), intent(in) :: dol1smoothing
|
||||
type(psb_zspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
@@ -132,7 +136,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
type(psb_z_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
type(psb_z_csr_sparse_mat) :: acsr1, acsrf, csr_prol, acsr
|
||||
complex(psb_dpk_), allocatable :: adiag(:)
|
||||
real(psb_dpk_), allocatable :: arwsum(:)
|
||||
real(psb_dpk_), allocatable :: arwsum(:),l1rwsum(:)
|
||||
integer(psb_ipk_) :: ierr(5)
|
||||
logical :: filter_mat
|
||||
integer(psb_ipk_) :: debug_level, debug_unit, err_act
|
||||
@@ -141,6 +145,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
logical :: do_l1correction=.false.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
@@ -173,6 +178,9 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if ((do_timings).and.(idx_ptap==-1)) &
|
||||
& idx_ptap = psb_get_timer_idx("DEC_SMTH_BLD: ptap_bld ")
|
||||
|
||||
! check if we have to use Jacobi or l1-Jacobi to smooth the tentative prolongator
|
||||
if (dol1smoothing.eq.amg_l1_smooth_prol_) do_l1correction=.true.
|
||||
|
||||
|
||||
nglob = desc_a%get_global_rows()
|
||||
nrow = desc_a%get_local_rows()
|
||||
@@ -185,7 +193,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
naggrm1 = sum(nlaggr(1:me))
|
||||
naggrp1 = sum(nlaggr(1:me+1))
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_)
|
||||
filter_mat = (parms%aggr_filter == amg_filter_mat_).or.(parms%aggr_filter == amg_filter_prow_mat_)
|
||||
|
||||
!
|
||||
! naggr: number of local aggregates
|
||||
@@ -200,6 +208,24 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(adiag,desc_a,info)
|
||||
if (info == psb_success_) call a%cp_to(acsr)
|
||||
!
|
||||
! Do the l1-correction on the diagonal if it is requested
|
||||
!
|
||||
if (do_l1correction) then
|
||||
allocate(l1rwsum(nrow))
|
||||
call acsr%arwsum(l1rwsum)
|
||||
if (info == psb_success_) &
|
||||
& call psb_realloc(ncol,l1rwsum,info)
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(l1rwsum,desc_a,info)
|
||||
! \tilde{D}_{i,i} = \sum_{j \ne i} |a_{i,j}|
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
adiag(i) = adiag(i) + l1rwsum(i) - abs(adiag(i))
|
||||
end do
|
||||
!$OMP end parallel do
|
||||
end if
|
||||
|
||||
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='sp_getdiag')
|
||||
@@ -230,9 +256,15 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
! if (.not.do_l1correction)
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
else if (parms%aggr_filter == amg_filter_mat_) then
|
||||
! We perform filtering in the standard way assuming that A is an M-matrix
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
else if (parms%aggr_filter == amg_filter_prow_mat_) then
|
||||
! We are probably doing l1-correction, hence we want to preserve the
|
||||
! row sum of the matrix: note the change in sign
|
||||
acsrf%val(jd)=acsrf%val(jd)+tmp
|
||||
end if
|
||||
enddo
|
||||
!$OMP end parallel do
|
||||
@@ -240,7 +272,6 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= zzero) then
|
||||
@@ -252,14 +283,17 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!$OMP end parallel do
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
if ( (parms%aggr_filter == amg_filter_prow_mat_).and.(do_l1correction) ) then
|
||||
! For l1-Jacobi this can be estimated with 1:
|
||||
! this makes sense only if we are preserving the row-sum!
|
||||
parms%aggr_omega_val = done
|
||||
else if (parms%aggr_eig == amg_max_norm_) then
|
||||
allocate(arwsum(nrow))
|
||||
call acsr%arwsum(arwsum)
|
||||
anorm = maxval(abs(adiag(1:nrow)*arwsum(1:nrow)))
|
||||
call psb_amx(ctxt,anorm)
|
||||
omega = 4.d0/(3.d0*anorm)
|
||||
parms%aggr_omega_val = omega
|
||||
|
||||
else
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_eig_')
|
||||
@@ -322,6 +356,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (debug_level >= psb_debug_outer_) &
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
if (allocated(l1rwsum)) deallocate(l1rwsum)
|
||||
call psb_erractionrestore(err_act)
|
||||
return
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
// TODO comment
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
void clean(MilanLongInt NLVer,
|
||||
MilanInt myRank,
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
/**
|
||||
* Execute the research fr the Candidate Mate without controlling if the vertices are already matched.
|
||||
* Returns the vertices with the highest weight
|
||||
@@ -9,7 +8,8 @@
|
||||
* @param edgeLocWeight
|
||||
* @return
|
||||
*/
|
||||
MilanLongInt firstComputeCandidateMate(MilanLongInt adj1,
|
||||
|
||||
MilanLongInt firstComputeCandidateMateD(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight)
|
||||
@@ -42,7 +42,7 @@ MilanLongInt firstComputeCandidateMate(MilanLongInt adj1,
|
||||
* @param Ghost2LocalMap
|
||||
* @return
|
||||
*/
|
||||
MilanLongInt computeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt computeCandidateMateD(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
@@ -71,4 +71,68 @@ MilanLongInt computeCandidateMate(MilanLongInt adj1,
|
||||
|
||||
return w;
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
MilanLongInt firstComputeCandidateMateS(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanFloat *edgeLocWeight)
|
||||
{
|
||||
MilanInt w = -1;
|
||||
MilanFloat heaviestEdgeWt = 0.0f; // Assign the smallest
|
||||
int finalK;
|
||||
for (int k = adj1; k < adj2; k++) {
|
||||
if ((edgeLocWeight[k] > heaviestEdgeWt) ||
|
||||
((edgeLocWeight[k] == heaviestEdgeWt) && (w < verLocInd[k]))) {
|
||||
heaviestEdgeWt = edgeLocWeight[k];
|
||||
w = verLocInd[k];
|
||||
finalK = k;
|
||||
}
|
||||
} // End of for loop
|
||||
return finalK;
|
||||
}
|
||||
|
||||
/**
|
||||
* //TODO documentation
|
||||
* @param adj1
|
||||
* @param adj2
|
||||
* @param edgeLocWeight
|
||||
* @param k
|
||||
* @param verLocInd
|
||||
* @param StartIndex
|
||||
* @param EndIndex
|
||||
* @param GMate
|
||||
* @param Mate
|
||||
* @param Ghost2LocalMap
|
||||
* @return
|
||||
*/
|
||||
MilanLongInt computeCandidateMateS(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap)
|
||||
{
|
||||
// Start: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
|
||||
MilanInt w = -1;
|
||||
MilanFloat heaviestEdgeWt = 0.0f; // Assign the smallest Value
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap))
|
||||
continue;
|
||||
|
||||
if ((edgeLocWeight[k] > heaviestEdgeWt) ||
|
||||
((edgeLocWeight[k] == heaviestEdgeWt) && (w < verLocInd[k]))) {
|
||||
heaviestEdgeWt = edgeLocWeight[k];
|
||||
w = verLocInd[k];
|
||||
}
|
||||
} // End of for loop
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
|
||||
return w;
|
||||
}
|
||||
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
void extractUChunk(
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
@@ -29,4 +28,3 @@ void extractUChunk(
|
||||
|
||||
} // End of critical U // End of critical U
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
/// Find the owner of a ghost node:
|
||||
MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs)
|
||||
@@ -27,4 +26,3 @@ MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
|
||||
return Current;
|
||||
} // End of findOwnerOfGhost()
|
||||
#endif
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt StartIndex, MilanLongInt EndIndex,
|
||||
MilanLongInt *numGhostEdges,
|
||||
@@ -302,4 +301,3 @@ void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
} // End of single region
|
||||
} // End of parallel region
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
/**
|
||||
* //TODO documentation
|
||||
* @param k
|
||||
@@ -44,4 +43,3 @@ bool isAlreadyMatched(MilanLongInt node,
|
||||
|
||||
return val >= 0; // Already matched
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_B(MilanLongInt NLVer,
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_BD(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
@@ -20,9 +20,37 @@ void PARALLEL_COMPUTE_CANDIDATE_MATE_B(MilanLongInt NLVer,
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Start: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
candidateMate[v] = firstComputeCandidateMate(verLocPtr[v], verLocPtr[v + 1], verLocInd, edgeLocWeight);
|
||||
candidateMate[v] = firstComputeCandidateMateD(verLocPtr[v], verLocPtr[v + 1],
|
||||
verLocInd, edgeLocWeight);
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_BS(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *candidateMate)
|
||||
{
|
||||
|
||||
MilanLongInt v = -1;
|
||||
|
||||
#pragma omp parallel private(v) default(shared) num_threads(NUM_THREAD)
|
||||
{
|
||||
|
||||
#pragma omp for schedule(static)
|
||||
for (v = 0; v < NLVer; v++) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing: " << v + StartIndex << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Start: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
candidateMate[v] = firstComputeCandidateMateS(verLocPtr[v], verLocPtr[v + 1],
|
||||
verLocInd, edgeLocWeight);
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
MilanLongInt *S)
|
||||
{
|
||||
@@ -22,4 +21,3 @@ void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
|
||||
// End: PARALLEL_PROCESS_CROSS_EDGE_B
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
#include "MatchBoxPC.h"
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_BD(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
@@ -31,15 +30,16 @@ void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
MilanLongInt v = -1, k = -1, w = -1, adj11 = 0, adj12 = 0, k1 = 0;
|
||||
MilanLongInt v = -1, k = -1, w = -1, adj11 = 0, adj12 = 0, k1 = 0;
|
||||
MilanInt ghostOwner = 0, option, igw;
|
||||
|
||||
#pragma omp parallel private(option, k, w, v, k1, adj11, adj12, ghostOwner) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx, privateQMsgType, privateQOwner) \
|
||||
default(shared) num_threads(NUM_THREAD)
|
||||
//#pragma omp parallel private(option, k, w, v, k1, adj11, adj12, ghostOwner) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, \
|
||||
privateQGhostVtx, privateQMsgType, privateQOwner) \
|
||||
default(shared) num_threads(NUM_THREAD)
|
||||
|
||||
{
|
||||
#pragma omp for reduction(+ \
|
||||
//#pragma omp for reduction(+ \
|
||||
: PCounter[:numProcs], myCard \
|
||||
[:1], msgInd \
|
||||
[:1], NumMessagesBundled \
|
||||
@@ -62,102 +62,89 @@ void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0)
|
||||
{
|
||||
|
||||
#pragma omp critical(processExposed)
|
||||
{
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap)) {
|
||||
w = computeCandidateMate(verLocPtr[v],
|
||||
verLocPtr[v + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
candidateMate[v] = w;
|
||||
}
|
||||
|
||||
if (w >= 0) {
|
||||
(*myCard)++;
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // w is a ghost vertex
|
||||
option = 2;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v + StartIndex) {
|
||||
option = 1;
|
||||
Mate[v] = w;
|
||||
GMate[Ghost2LocalMap[w]] = v + StartIndex; // w is a Ghost
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
|
||||
if (candidateMate[w - StartIndex] == (v + StartIndex)) {
|
||||
option = 3;
|
||||
Mate[v] = w; // v is local
|
||||
Mate[w - StartIndex] = v + StartIndex; // w is local
|
||||
|
||||
if (w >= 0) {
|
||||
#pragma omp critical(Matching)
|
||||
{
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap)) {
|
||||
w = computeCandidateMateD(verLocPtr[v], verLocPtr[v + 1], edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v] = w;
|
||||
}
|
||||
}
|
||||
if (w >= 0) {
|
||||
(*myCard)++;
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // w is a ghost vertex
|
||||
option = 2;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v + StartIndex) {
|
||||
option = 1;
|
||||
Mate[v] = w;
|
||||
GMate[Ghost2LocalMap[w]] = v + StartIndex; // w is a Ghost
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == (v + StartIndex)) {
|
||||
option = 3;
|
||||
Mate[v] = w; // v is local
|
||||
Mate[w - StartIndex] = v + StartIndex; // w is local
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
} // End of if ( candidateMate[w-StartIndex] == (v+StartIndex) )
|
||||
} // End of Else
|
||||
|
||||
} // End of second if
|
||||
|
||||
} // End critical processExposed
|
||||
|
||||
} // End of if ( candidateMate[w-StartIndex] == (v+StartIndex) )
|
||||
} // End of Else
|
||||
} // End of second if
|
||||
} // End of if(w >=0)
|
||||
else {
|
||||
// This piece of code is executed a really small amount of times
|
||||
adj11 = verLocPtr[v];
|
||||
adj12 = verLocPtr[v + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
//#pragma omp critical(adjuse)
|
||||
{
|
||||
// This piece of code is executed a really small number of times
|
||||
adj11 = verLocPtr[v];
|
||||
adj12 = verLocPtr[v + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
}
|
||||
}
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
{
|
||||
case -1:
|
||||
break;
|
||||
case 1:
|
||||
case 1:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ")";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
case 2:
|
||||
case 2:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message (291):";
|
||||
cout << "\n(" << myRank << ")Local is: " << v + StartIndex << " Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
@@ -168,29 +155,215 @@ void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
default:
|
||||
case 3:
|
||||
default:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
break;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
} // End of for ( v=0; v < NLVer; v++ )
|
||||
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
|
||||
} // End of parallel region
|
||||
}
|
||||
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_BS(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
MilanLongInt v = -1, k = -1, w = -1, adj11 = 0, adj12 = 0, k1 = 0;
|
||||
MilanInt ghostOwner = 0, option, igw;
|
||||
|
||||
//#pragma omp parallel private(option, k, w, v, k1, adj11, adj12, ghostOwner) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, \
|
||||
privateQGhostVtx, privateQMsgType, privateQOwner) \
|
||||
default(shared) num_threads(NUM_THREAD)
|
||||
|
||||
{
|
||||
//#pragma omp for reduction(+ \
|
||||
: PCounter[:numProcs], myCard \
|
||||
[:1], msgInd \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1]) \
|
||||
schedule(static)
|
||||
for (v = 0; v < NLVer; v++) {
|
||||
option = -1;
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
k = candidateMate[v];
|
||||
candidateMate[v] = verLocInd[k];
|
||||
w = candidateMate[v];
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing: " << v + StartIndex << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v + StartIndex << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
#pragma omp critical(Matching)
|
||||
{
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap)) {
|
||||
w = computeCandidateMateS(verLocPtr[v], verLocPtr[v + 1], edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v] = w;
|
||||
}
|
||||
if (w >= 0) {
|
||||
(*myCard)++;
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // w is a ghost vertex
|
||||
option = 2;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v + StartIndex) {
|
||||
option = 1;
|
||||
Mate[v] = w;
|
||||
GMate[Ghost2LocalMap[w]] = v + StartIndex; // w is a Ghost
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == (v + StartIndex)) {
|
||||
option = 3;
|
||||
Mate[v] = w; // v is local
|
||||
Mate[w - StartIndex] = v + StartIndex; // w is local
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if ( candidateMate[w-StartIndex] == (v+StartIndex) )
|
||||
} // End of Else
|
||||
} // End of second if
|
||||
}
|
||||
|
||||
} // End of if(w >=0)
|
||||
else {
|
||||
//#pragma omp critical(adjuse)
|
||||
{
|
||||
// This piece of code is executed a really small number of times
|
||||
adj11 = verLocPtr[v];
|
||||
adj12 = verLocPtr[v + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
}
|
||||
}
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
break;
|
||||
case 1:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ")";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
case 2:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message (291):";
|
||||
cout << "\n(" << myRank << ")Local is: " << v + StartIndex << " Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
default:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
break;
|
||||
}
|
||||
|
||||
} // End of for ( v=0; v < NLVer; v++ )
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
} // End of parallel region
|
||||
}
|
||||
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
void processMatchedVertices(
|
||||
void processMatchedVerticesD(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
@@ -91,57 +90,57 @@ void processMatchedVertices(
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
|
||||
candidateMate[v - StartIndex] = w;
|
||||
|
||||
w = computeCandidateMateD(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
}
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
@@ -179,6 +178,7 @@ void processMatchedVertices(
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
@@ -211,7 +211,7 @@ void processMatchedVertices(
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
@@ -248,7 +248,6 @@ void processMatchedVertices(
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
@@ -265,17 +264,15 @@ void processMatchedVertices(
|
||||
U.insert(U.end(), privateU.begin(), privateU.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
|
||||
#pragma omp critical(sendMessageTransfer)
|
||||
{
|
||||
|
||||
QLocalVtx.insert(QLocalVtx.end(), privateQLocalVtx.begin(), privateQLocalVtx.end());
|
||||
QGhostVtx.insert(QGhostVtx.end(), privateQGhostVtx.begin(), privateQGhostVtx.end());
|
||||
QMsgType.insert(QMsgType.end(), privateQMsgType.begin(), privateQMsgType.end());
|
||||
QOwner.insert(QOwner.end(), privateQOwner.begin(), privateQOwner.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
privateQLocalVtx.clear();
|
||||
privateQGhostVtx.clear();
|
||||
privateQMsgType.clear();
|
||||
@@ -292,4 +289,297 @@ void processMatchedVertices(
|
||||
#endif
|
||||
} // End of parallel region
|
||||
}
|
||||
|
||||
|
||||
|
||||
void processMatchedVerticesS(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
MilanLongInt adj1, adj2, adj11, adj12, k, k1, v = -1, w = -1, ghostOwner;
|
||||
int option;
|
||||
MilanLongInt mateVal;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
MilanLongInt localVertices = 0;
|
||||
#endif
|
||||
//#pragma omp parallel private(k, w, v, k1, adj1, adj2, adj11, adj12, ghostOwner, option) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx, \
|
||||
privateQMsgType, privateQOwner, UChunkBeingProcessed) \
|
||||
default(shared) num_threads(NUM_THREAD) \
|
||||
reduction(+ \
|
||||
: msgInd[:1], PCounter \
|
||||
[:numProcs], myCard \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1])
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
|
||||
extractUChunk(UChunkBeingProcessed, U, privateU);
|
||||
|
||||
for (MilanLongInt u : UChunkBeingProcessed) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")u: " << u;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((u >= StartIndex) && (u <= EndIndex)) { // Process Only the Local Vertices
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
option = -1;
|
||||
v = verLocInd[k];
|
||||
|
||||
if ((v >= StartIndex) && (v <= EndIndex)) { // If Local Vertex:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")v: " << v << " c(v)= " << candidateMate[v - StartIndex] << " Mate[v]: " << Mate[v];
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateS(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
}
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
else { // Neighbor is a ghost vertex
|
||||
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[v]] == u)
|
||||
candidateMate[NLVer + Ghost2LocalMap[v]] = -1;
|
||||
if (v != Mate[u - StartIndex])
|
||||
option = 5; // u is local
|
||||
} // End of critical
|
||||
} // End of Else //A Ghost Vertex
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
// No things to do
|
||||
break;
|
||||
case 1:
|
||||
// Found a dominating edge, it is a ghost and candidateMate[NLVer + Ghost2LocalMap[w]] == v
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], SPtr);
|
||||
case 2:
|
||||
|
||||
// Found a dominating edge, it is a ghost
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
|
||||
(*myCard)++;
|
||||
break;
|
||||
case 4:
|
||||
// Could not find a dominating vertex
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
break;
|
||||
case 5:
|
||||
default:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a success message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << v << " Owner is: " << findOwnerOfGhost(v, verDistance, myRank, numProcs) << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
|
||||
(*NumMessagesBundled)++;
|
||||
PCounter[ghostOwner]++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
#pragma omp critical(U)
|
||||
{
|
||||
U.insert(U.end(), privateU.begin(), privateU.end());
|
||||
}
|
||||
|
||||
#pragma omp critical(sendMessageTransfer)
|
||||
{
|
||||
QLocalVtx.insert(QLocalVtx.end(), privateQLocalVtx.begin(), privateQLocalVtx.end());
|
||||
QGhostVtx.insert(QGhostVtx.end(), privateQGhostVtx.begin(), privateQGhostVtx.end());
|
||||
QMsgType.insert(QMsgType.end(), privateQMsgType.begin(), privateQMsgType.end());
|
||||
QOwner.insert(QOwner.end(), privateQOwner.begin(), privateQOwner.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
privateQLocalVtx.clear();
|
||||
privateQGhostVtx.clear();
|
||||
privateQMsgType.clear();
|
||||
privateQOwner.clear();
|
||||
|
||||
} // End of while ( !U.empty() )
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices,
|
||||
omp_get_thread_num(),
|
||||
myRank);
|
||||
|
||||
#endif
|
||||
} // End of parallel region
|
||||
}
|
||||
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
//#define DEBUG_HANG_
|
||||
void processMatchedVerticesAndSendMessages(
|
||||
void processMatchedVerticesAndSendMessagesD(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
@@ -27,7 +26,7 @@ void processMatchedVerticesAndSendMessages(
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
@@ -52,15 +51,16 @@ void processMatchedVerticesAndSendMessages(
|
||||
MilanLongInt localVertices = 0;
|
||||
#endif
|
||||
//#pragma omp parallel private(k, w, v, k1, adj1, adj2, adj11, adj12, ghostOwner, option) \
|
||||
firstprivate(Message, privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx,\
|
||||
privateQMsgType, privateQOwner, UChunkBeingProcessed) default(shared) \
|
||||
num_threads(NUM_THREAD) \
|
||||
reduction(+ \
|
||||
: msgInd[:1], PCounter \
|
||||
[:numProcs], myCard \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1], msgActual \
|
||||
[:1])
|
||||
firstprivate(Message, privateU, StartIndex, EndIndex, privateQLocalVtx, \
|
||||
privateQGhostVtx, privateQMsgType, privateQOwner, UChunkBeingProcessed) \
|
||||
default(shared) \
|
||||
num_threads(NUM_THREAD) \
|
||||
reduction(+ \
|
||||
: msgInd[:1], PCounter \
|
||||
[:numProcs], myCard \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1], msgActual \
|
||||
[:1])
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
@@ -77,7 +77,6 @@ void processMatchedVerticesAndSendMessages(
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
@@ -97,58 +96,57 @@ void processMatchedVerticesAndSendMessages(
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
|
||||
candidateMate[v - StartIndex] = w;
|
||||
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateD(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
}
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
@@ -211,15 +209,12 @@ void processMatchedVerticesAndSendMessages(
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
@@ -229,7 +224,6 @@ void processMatchedVerticesAndSendMessages(
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
@@ -248,17 +242,14 @@ void processMatchedVerticesAndSendMessages(
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = u; // LOCAL
|
||||
// Message[1] = v; // GHOST
|
||||
// Message[2] = SUCCESS; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
@@ -281,10 +272,7 @@ void processMatchedVerticesAndSendMessages(
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices,
|
||||
omp_get_thread_num(),
|
||||
myRank);
|
||||
|
||||
localVertices, mp_get_thread_num(), myRank);
|
||||
#endif
|
||||
} // End of parallel region
|
||||
|
||||
@@ -293,12 +281,10 @@ void processMatchedVerticesAndSendMessages(
|
||||
cout << myRank<<" Sending: "<<QOwner.size()-initialSize<<" messages" <<endl;
|
||||
#endif
|
||||
for (int i = initialSize; i < QOwner.size(); i++) {
|
||||
|
||||
Message[0] = QLocalVtx[i];
|
||||
Message[1] = QGhostVtx[i];
|
||||
Message[2] = QMsgType[i];
|
||||
ghostOwner = QOwner[i];
|
||||
|
||||
//MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
@@ -307,4 +293,299 @@ void processMatchedVerticesAndSendMessages(
|
||||
cout << myRank<<" Done sending messages"<<endl;
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
|
||||
void processMatchedVerticesAndSendMessagesS(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message)
|
||||
{
|
||||
|
||||
MilanLongInt initialSize = QLocalVtx.size();
|
||||
MilanLongInt adj1, adj2, adj11, adj12, k, k1, v = -1, w = -1, ghostOwner;
|
||||
int option;
|
||||
MilanLongInt mateVal;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
MilanLongInt localVertices = 0;
|
||||
#endif
|
||||
//#pragma omp parallel private(k, w, v, k1, adj1, adj2, adj11, adj12, ghostOwner, option) \
|
||||
firstprivate(Message, privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx,\
|
||||
privateQMsgType, privateQOwner, UChunkBeingProcessed) default(shared) \
|
||||
num_threads(NUM_THREAD) \
|
||||
reduction(+ \
|
||||
: msgInd[:1], PCounter \
|
||||
[:numProcs], myCard \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1], msgActual \
|
||||
[:1])
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
|
||||
extractUChunk(UChunkBeingProcessed, U, privateU);
|
||||
|
||||
for (MilanLongInt u : UChunkBeingProcessed) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")u: " << u;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((u >= StartIndex) && (u <= EndIndex)) { // Process Only the Local Vertices
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
option = -1;
|
||||
v = verLocInd[k];
|
||||
|
||||
if ((v >= StartIndex) && (v <= EndIndex)) { // If Local Vertex:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")v: " << v << " c(v)= " << candidateMate[v - StartIndex] << " Mate[v]: " << Mate[v];
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateS(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
}
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
else { // Neighbor is a ghost vertex
|
||||
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[v]] == u)
|
||||
candidateMate[NLVer + Ghost2LocalMap[v]] = -1;
|
||||
if (v != Mate[u - StartIndex])
|
||||
option = 5; // u is local
|
||||
} // End of critical
|
||||
} // End of Else //A Ghost Vertex
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
// No things to do
|
||||
break;
|
||||
case 1:
|
||||
// Found a dominating edge, it is a ghost and candidateMate[NLVer + Ghost2LocalMap[w]] == v
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], SPtr);
|
||||
case 2:
|
||||
|
||||
// Found a dominating edge, it is a ghost
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
// Message[2] = REQUEST; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
(*myCard)++;
|
||||
break;
|
||||
case 4:
|
||||
// Could not find a dominating vertex
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
// Message[2] = FAILURE; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
break;
|
||||
case 5:
|
||||
default:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a success message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << v << " Owner is: " << findOwnerOfGhost(v, verDistance, myRank, numProcs) << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
// Build the Message Packet:
|
||||
// Message[0] = u; // LOCAL
|
||||
// Message[1] = v; // GHOST
|
||||
// Message[2] = SUCCESS; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
} // End of while ( !U.empty() )
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices, mp_get_thread_num(), myRank);
|
||||
#endif
|
||||
} // End of parallel region
|
||||
|
||||
// Send the messages
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank<<" Sending: "<<QOwner.size()-initialSize<<" messages" <<endl;
|
||||
#endif
|
||||
for (int i = initialSize; i < QOwner.size(); i++) {
|
||||
Message[0] = QLocalVtx[i];
|
||||
Message[1] = QGhostVtx[i];
|
||||
Message[2] = QMsgType[i];
|
||||
ghostOwner = QOwner[i];
|
||||
//MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
}
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank<<" Done sending messages"<<endl;
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
//#define DEBUG_HANG_
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
void processMessages(
|
||||
void processMessagesD(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
@@ -212,8 +212,324 @@ void processMessages(
|
||||
// Process only if not already matched ( v is local)
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex], verLocPtr[v - StartIndex + 1], edgeLocWeight, k,
|
||||
verLocInd, StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap);
|
||||
w = computeCandidateMateD(verLocPtr[v - StartIndex], verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, k,verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) {
|
||||
// w is a ghost
|
||||
// Build the Message Packet:
|
||||
Message[0] = v; // LOCAL
|
||||
Message[1] = w; // GHOST
|
||||
Message[2] = REQUEST; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
//assert(ghostOwner != -1);
|
||||
//assert(ghostOwner != myRank);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
(*msgInd)++;
|
||||
(*msgActual)++;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
Mate[v - StartIndex] = w; // v is local
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is ghost
|
||||
U.push_back(v);
|
||||
U.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
Mate[v - StartIndex] = w; // v is local
|
||||
Mate[w - StartIndex] = v; // w is local
|
||||
// Q.push_back(u);
|
||||
U.push_back(v);
|
||||
U.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else { // No dominant edge found
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) {
|
||||
// A ghost
|
||||
// Build the Message Packet:
|
||||
Message[0] = v; // LOCAL
|
||||
Message[1] = w; // GHOST
|
||||
Message[2] = FAILURE; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
//assert(ghostOwner != -1);
|
||||
//assert(ghostOwner != myRank);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
(*msgInd)++;
|
||||
(*msgActual)++;
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
} // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of if ( candidateMate[v-StartIndex] == u )
|
||||
} // End of if ( Mate[v] == -1 )
|
||||
} // End of if ( message_type == SUCCESS )
|
||||
else {
|
||||
// CASE III: FAILURE
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is FAILURE" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
GMate[Ghost2LocalMap[u]] = EndIndex + 1; // Set a Dummy Mate to make sure that we do not (u is a ghost) process this anymore
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S); // Decrease the counter
|
||||
} // End of else: CASE III
|
||||
} // End of else: CASE I
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
void processMessagesS(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *msgActual,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *S,
|
||||
vector<MilanLongInt> &U)
|
||||
{
|
||||
|
||||
//#define PRINT_DEBUG_INFO_
|
||||
|
||||
MilanInt Sender;
|
||||
MPI_Status computeStatus;
|
||||
MilanLongInt bundleSize, w;
|
||||
MilanLongInt adj11, adj12, k1;
|
||||
MilanLongInt ghostOwner;
|
||||
int error_codeC;
|
||||
error_codeC = MPI_Comm_set_errhandler(MPI_COMM_WORLD, MPI_ERRORS_RETURN);
|
||||
char error_message[MPI_MAX_ERROR_STRING];
|
||||
int message_length;
|
||||
MilanLongInt message_type = 0;
|
||||
|
||||
// Buffer to receive bundled messages
|
||||
// Maximum messages that can be received from any processor is
|
||||
// twice the edge cut: REQUEST; REQUEST+(FAILURE/SUCCESS)
|
||||
vector<MilanLongInt> ReceiveBuffer;
|
||||
try
|
||||
{
|
||||
ReceiveBuffer.reserve(numGhostEdges * 2 * 3); // Three integers per cross edge
|
||||
}
|
||||
catch (length_error)
|
||||
{
|
||||
cout << "Error in function algoDistEdgeApproxDominatingEdgesMessageBundling: \n";
|
||||
cout << "Not enough memory to allocate the internal variables \n";
|
||||
exit(1);
|
||||
}
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout
|
||||
<< "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")About to begin Message processing phase ... *S=" << *S << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// BLOCKING RECEIVE:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << " Waiting for blocking receive..." << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
//cout << myRank<<" Receiving ...";
|
||||
error_codeC = MPI_Recv(&Message[0], 3, TypeMap<MilanLongInt>(), MPI_ANY_SOURCE, ComputeTag, comm, &computeStatus);
|
||||
if (error_codeC != MPI_SUCCESS)
|
||||
{
|
||||
MPI_Error_string(error_codeC, error_message, &message_length);
|
||||
cout << "\n*Error in call to MPI_Receive on Slave: " << error_message << "\n";
|
||||
fflush(stdout);
|
||||
}
|
||||
Sender = computeStatus.MPI_SOURCE;
|
||||
//cout << " ...from "<<Sender << endl;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received message from Process " << Sender << " Type= " << Message[2] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
if (Message[2] == SIZEINFO) {
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received bundled message from Process " << Sender << " Size= " << Message[0] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
bundleSize = Message[0]; //#of integers in the message
|
||||
// Build the Message Buffer:
|
||||
if (!ReceiveBuffer.empty())
|
||||
ReceiveBuffer.clear(); // Empty it out first
|
||||
ReceiveBuffer.resize(bundleSize, -1); // Initialize
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message Bundle Before: " << endl;
|
||||
for (int i = 0; i < bundleSize; i++)
|
||||
cout << ReceiveBuffer[i] << ",";
|
||||
cout << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Receive the message
|
||||
//cout << myRank<<" Receiving from "<<Sender<<endl;
|
||||
error_codeC = MPI_Recv(&ReceiveBuffer[0], bundleSize, TypeMap<MilanLongInt>(), Sender, BundleTag, comm, &computeStatus);
|
||||
if (error_codeC != MPI_SUCCESS) {
|
||||
MPI_Error_string(error_codeC, error_message, &message_length);
|
||||
cout << "\n*Error in call to MPI_Receive on processor " << myRank << " Error: " << error_message << "\n";
|
||||
fflush(stdout);
|
||||
}
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message Bundle After: " << endl;
|
||||
for (int i = 0; i < bundleSize; i++)
|
||||
cout << ReceiveBuffer[i] << ",";
|
||||
cout << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} else { // Just a single message:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received regular message from Process " << Sender << " u= " << Message[0] << " v= " << Message[1] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Add the current message to Queue:
|
||||
bundleSize = 3; //#of integers in the message
|
||||
// Build the Message Buffer:
|
||||
if (!ReceiveBuffer.empty())
|
||||
ReceiveBuffer.clear(); // Empty it out first
|
||||
ReceiveBuffer.resize(bundleSize, -1); // Initialize
|
||||
|
||||
ReceiveBuffer[0] = Message[0]; // u
|
||||
ReceiveBuffer[1] = Message[1]; // v
|
||||
ReceiveBuffer[2] = Message[2]; // message_type
|
||||
}
|
||||
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < StartIndex) || (v > EndIndex)) {
|
||||
cout << "\n(" << myRank << ") From ReceiveBuffer: This should not happen: u= " << u << " v= " << v << " Type= " << message_type << " StartIndex " << StartIndex << " EndIndex " << EndIndex << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing message: u= " << u << " v= " << v << " Type= " << message_type << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Most of the time bundleSize == 3, thus, it's not worth parallelizing thi loop
|
||||
for (MilanLongInt bundleCounter = 3; bundleCounter < bundleSize + 3; bundleCounter += 3) {
|
||||
u = ReceiveBuffer[bundleCounter - 3]; // GHOST
|
||||
v = ReceiveBuffer[bundleCounter - 2]; // LOCAL
|
||||
message_type = ReceiveBuffer[bundleCounter - 1]; // TYPE
|
||||
|
||||
// CASE I: REQUEST
|
||||
if (message_type == REQUEST) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is REQUEST" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < 0) || (v < StartIndex) || ((v - StartIndex) > NLVer)) {
|
||||
cout << "\n(" << myRank << ") case 1 Bad address " << v << " " << StartIndex << " " << v - StartIndex << " " << NLVer << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
if (Mate[v - StartIndex] == -1) {
|
||||
// Process only if not already matched (v is local)
|
||||
candidateMate[NLVer + Ghost2LocalMap[u]] = v; // Set CandidateMate for the ghost
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
GMate[Ghost2LocalMap[u]] = v; // u is ghost
|
||||
Mate[v - StartIndex] = u; // v is local
|
||||
U.push_back(v);
|
||||
U.push_back(u);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << u << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S);
|
||||
} // End of if ( candidateMate[v-StartIndex] == u )e
|
||||
} // End of if ( Mate[v] == -1 )
|
||||
} // End of REQUEST
|
||||
else { // CASE II: SUCCESS
|
||||
if (message_type == SUCCESS) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is SUCCESS" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
GMate[Ghost2LocalMap[u]] = EndIndex + 1; // Set a Dummy Mate to make sure that we do not (u is a ghost) process it again
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S);
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < 0) || (v < StartIndex) || ((v - StartIndex) > NLVer)) {
|
||||
cout << "\n(" << myRank << ") case 2 Bad address " << v << " " << StartIndex << " " << v - StartIndex << " " << NLVer << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
#endif
|
||||
if (Mate[v - StartIndex] == -1) {
|
||||
// Process only if not already matched ( v is local)
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateS(verLocPtr[v - StartIndex], verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, k,verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w << endl;
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
void queuesTransfer(vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
@@ -15,23 +14,20 @@ void queuesTransfer(vector<MilanLongInt> &U,
|
||||
#pragma omp critical(U)
|
||||
{
|
||||
U.insert(U.end(), privateU.begin(), privateU.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
|
||||
#pragma omp critical(sendMessageTransfer)
|
||||
{
|
||||
|
||||
//#pragma omp critical(sendMessageTransfer)
|
||||
//{
|
||||
QLocalVtx.insert(QLocalVtx.end(), privateQLocalVtx.begin(), privateQLocalVtx.end());
|
||||
QGhostVtx.insert(QGhostVtx.end(), privateQGhostVtx.begin(), privateQGhostVtx.end());
|
||||
QMsgType.insert(QMsgType.end(), privateQMsgType.begin(), privateQMsgType.end());
|
||||
QOwner.insert(QOwner.end(), privateQOwner.begin(), privateQOwner.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
privateQLocalVtx.clear();
|
||||
privateQGhostVtx.clear();
|
||||
privateQMsgType.clear();
|
||||
privateQOwner.clear();
|
||||
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
@@ -1,24 +1,23 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef OMP
|
||||
void sendBundledMessages(MilanLongInt *numGhostEdges,
|
||||
MilanInt *BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActual,
|
||||
MilanLongInt *msgInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus)
|
||||
MilanInt *BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActual,
|
||||
MilanLongInt *msgInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus)
|
||||
{
|
||||
|
||||
MilanLongInt myIndex = 0, numMessagesToSend;
|
||||
@@ -207,4 +206,3 @@ void sendBundledMessages(MilanLongInt *numGhostEdges,
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -451,11 +451,16 @@ subroutine amg_c_hierarchy_bld(a,desc_a,prec,info)
|
||||
if (.not.associated(prec%precv(1)%base_desc,desc_a)) then
|
||||
prec%precv(1)%base_desc => prec%precv(1)%desc_ac
|
||||
end if
|
||||
do i=2, iszv
|
||||
do i=2, iszv
|
||||
prec%precv(i)%base_a => prec%precv(i)%ac
|
||||
prec%precv(i)%base_desc => prec%precv(i)%desc_ac
|
||||
prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
! This is needed when the linmap object has been built
|
||||
! reusing the base_desc descriptor through a pointer.
|
||||
! With PSBLAS 4 we will have a better solution
|
||||
if (associated(prec%precv(i)%linmap%p_desc_U)) &
|
||||
& prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
if (associated(prec%precv(i)%linmap%p_desc_V))&
|
||||
& prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
end do
|
||||
end if
|
||||
|
||||
@@ -507,6 +512,71 @@ subroutine amg_c_hierarchy_bld(a,desc_a,prec,info)
|
||||
return
|
||||
|
||||
contains
|
||||
|
||||
#if ( __GNUC__ == 13 && __GNUC_MINOR__ == 3)
|
||||
! gfortran 13.3.0 generates a strange error here with MOLD
|
||||
! moving to SOURCE but only for this version, since it's heavier
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_c_onelev_type), intent(inout) :: level
|
||||
class(amg_c_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(save1)) then
|
||||
call save1%free(info)
|
||||
if (info == 0) deallocate(save1,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
call save2%free(info)
|
||||
if (info == 0) deallocate(save2,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
allocate(save1, source=level%sm,stat=info)
|
||||
if (info == 0) call level%sm%clone_settings(save1,info)
|
||||
if ((info == 0).and.allocated(level%sm2a)) then
|
||||
allocate(save2, source=level%sm2a,stat=info)
|
||||
if (info == 0) call level%sm2a%clone_settings(save2,info)
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine save_smoothers
|
||||
|
||||
subroutine restore_smoothers(level,save1, save2,info)
|
||||
type(amg_c_onelev_type), intent(inout), target :: level
|
||||
class(amg_c_base_smoother_type), allocatable, intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
|
||||
if (allocated(level%sm)) then
|
||||
if (info == 0) call level%sm%free(info)
|
||||
if (info == 0) deallocate(level%sm,stat=info)
|
||||
end if
|
||||
if (allocated(save1)) then
|
||||
if (info == 0) allocate(level%sm,source=save1,stat=info)
|
||||
if (info == 0) call save1%clone_settings(level%sm,info)
|
||||
end if
|
||||
|
||||
if (info /= 0) return
|
||||
|
||||
if (allocated(level%sm2a)) then
|
||||
if (info == 0) call level%sm2a%free(info)
|
||||
if (info == 0) deallocate(level%sm2a,stat=info)
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
if (info == 0) allocate(level%sm2a,source=save2,stat=info)
|
||||
if (info == 0) call save2%clone_settings(level%sm2a,info)
|
||||
if (info == 0) level%sm2 => level%sm2a
|
||||
else
|
||||
if (allocated(level%sm)) level%sm2 => level%sm
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#else
|
||||
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_c_onelev_type), intent(inout) :: level
|
||||
class(amg_c_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
@@ -565,5 +635,5 @@ contains
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#endif
|
||||
end subroutine amg_c_hierarchy_bld
|
||||
@@ -451,11 +451,16 @@ subroutine amg_d_hierarchy_bld(a,desc_a,prec,info)
|
||||
if (.not.associated(prec%precv(1)%base_desc,desc_a)) then
|
||||
prec%precv(1)%base_desc => prec%precv(1)%desc_ac
|
||||
end if
|
||||
do i=2, iszv
|
||||
do i=2, iszv
|
||||
prec%precv(i)%base_a => prec%precv(i)%ac
|
||||
prec%precv(i)%base_desc => prec%precv(i)%desc_ac
|
||||
prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
! This is needed when the linmap object has been built
|
||||
! reusing the base_desc descriptor through a pointer.
|
||||
! With PSBLAS 4 we will have a better solution
|
||||
if (associated(prec%precv(i)%linmap%p_desc_U)) &
|
||||
& prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
if (associated(prec%precv(i)%linmap%p_desc_V))&
|
||||
& prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
end do
|
||||
end if
|
||||
|
||||
@@ -507,6 +512,71 @@ subroutine amg_d_hierarchy_bld(a,desc_a,prec,info)
|
||||
return
|
||||
|
||||
contains
|
||||
|
||||
#if ( __GNUC__ == 13 && __GNUC_MINOR__ == 3)
|
||||
! gfortran 13.3.0 generates a strange error here with MOLD
|
||||
! moving to SOURCE but only for this version, since it's heavier
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_d_onelev_type), intent(inout) :: level
|
||||
class(amg_d_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(save1)) then
|
||||
call save1%free(info)
|
||||
if (info == 0) deallocate(save1,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
call save2%free(info)
|
||||
if (info == 0) deallocate(save2,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
allocate(save1, source=level%sm,stat=info)
|
||||
if (info == 0) call level%sm%clone_settings(save1,info)
|
||||
if ((info == 0).and.allocated(level%sm2a)) then
|
||||
allocate(save2, source=level%sm2a,stat=info)
|
||||
if (info == 0) call level%sm2a%clone_settings(save2,info)
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine save_smoothers
|
||||
|
||||
subroutine restore_smoothers(level,save1, save2,info)
|
||||
type(amg_d_onelev_type), intent(inout), target :: level
|
||||
class(amg_d_base_smoother_type), allocatable, intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
|
||||
if (allocated(level%sm)) then
|
||||
if (info == 0) call level%sm%free(info)
|
||||
if (info == 0) deallocate(level%sm,stat=info)
|
||||
end if
|
||||
if (allocated(save1)) then
|
||||
if (info == 0) allocate(level%sm,source=save1,stat=info)
|
||||
if (info == 0) call save1%clone_settings(level%sm,info)
|
||||
end if
|
||||
|
||||
if (info /= 0) return
|
||||
|
||||
if (allocated(level%sm2a)) then
|
||||
if (info == 0) call level%sm2a%free(info)
|
||||
if (info == 0) deallocate(level%sm2a,stat=info)
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
if (info == 0) allocate(level%sm2a,source=save2,stat=info)
|
||||
if (info == 0) call save2%clone_settings(level%sm2a,info)
|
||||
if (info == 0) level%sm2 => level%sm2a
|
||||
else
|
||||
if (allocated(level%sm)) level%sm2 => level%sm
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#else
|
||||
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_d_onelev_type), intent(inout) :: level
|
||||
class(amg_d_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
@@ -565,5 +635,5 @@ contains
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#endif
|
||||
end subroutine amg_d_hierarchy_bld
|
||||
@@ -451,11 +451,16 @@ subroutine amg_s_hierarchy_bld(a,desc_a,prec,info)
|
||||
if (.not.associated(prec%precv(1)%base_desc,desc_a)) then
|
||||
prec%precv(1)%base_desc => prec%precv(1)%desc_ac
|
||||
end if
|
||||
do i=2, iszv
|
||||
do i=2, iszv
|
||||
prec%precv(i)%base_a => prec%precv(i)%ac
|
||||
prec%precv(i)%base_desc => prec%precv(i)%desc_ac
|
||||
prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
! This is needed when the linmap object has been built
|
||||
! reusing the base_desc descriptor through a pointer.
|
||||
! With PSBLAS 4 we will have a better solution
|
||||
if (associated(prec%precv(i)%linmap%p_desc_U)) &
|
||||
& prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
if (associated(prec%precv(i)%linmap%p_desc_V))&
|
||||
& prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
end do
|
||||
end if
|
||||
|
||||
@@ -507,6 +512,71 @@ subroutine amg_s_hierarchy_bld(a,desc_a,prec,info)
|
||||
return
|
||||
|
||||
contains
|
||||
|
||||
#if ( __GNUC__ == 13 && __GNUC_MINOR__ == 3)
|
||||
! gfortran 13.3.0 generates a strange error here with MOLD
|
||||
! moving to SOURCE but only for this version, since it's heavier
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_s_onelev_type), intent(inout) :: level
|
||||
class(amg_s_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(save1)) then
|
||||
call save1%free(info)
|
||||
if (info == 0) deallocate(save1,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
call save2%free(info)
|
||||
if (info == 0) deallocate(save2,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
allocate(save1, source=level%sm,stat=info)
|
||||
if (info == 0) call level%sm%clone_settings(save1,info)
|
||||
if ((info == 0).and.allocated(level%sm2a)) then
|
||||
allocate(save2, source=level%sm2a,stat=info)
|
||||
if (info == 0) call level%sm2a%clone_settings(save2,info)
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine save_smoothers
|
||||
|
||||
subroutine restore_smoothers(level,save1, save2,info)
|
||||
type(amg_s_onelev_type), intent(inout), target :: level
|
||||
class(amg_s_base_smoother_type), allocatable, intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
|
||||
if (allocated(level%sm)) then
|
||||
if (info == 0) call level%sm%free(info)
|
||||
if (info == 0) deallocate(level%sm,stat=info)
|
||||
end if
|
||||
if (allocated(save1)) then
|
||||
if (info == 0) allocate(level%sm,source=save1,stat=info)
|
||||
if (info == 0) call save1%clone_settings(level%sm,info)
|
||||
end if
|
||||
|
||||
if (info /= 0) return
|
||||
|
||||
if (allocated(level%sm2a)) then
|
||||
if (info == 0) call level%sm2a%free(info)
|
||||
if (info == 0) deallocate(level%sm2a,stat=info)
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
if (info == 0) allocate(level%sm2a,source=save2,stat=info)
|
||||
if (info == 0) call save2%clone_settings(level%sm2a,info)
|
||||
if (info == 0) level%sm2 => level%sm2a
|
||||
else
|
||||
if (allocated(level%sm)) level%sm2 => level%sm
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#else
|
||||
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_s_onelev_type), intent(inout) :: level
|
||||
class(amg_s_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
@@ -565,5 +635,5 @@ contains
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#endif
|
||||
end subroutine amg_s_hierarchy_bld
|
||||
@@ -451,11 +451,16 @@ subroutine amg_z_hierarchy_bld(a,desc_a,prec,info)
|
||||
if (.not.associated(prec%precv(1)%base_desc,desc_a)) then
|
||||
prec%precv(1)%base_desc => prec%precv(1)%desc_ac
|
||||
end if
|
||||
do i=2, iszv
|
||||
do i=2, iszv
|
||||
prec%precv(i)%base_a => prec%precv(i)%ac
|
||||
prec%precv(i)%base_desc => prec%precv(i)%desc_ac
|
||||
prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
! This is needed when the linmap object has been built
|
||||
! reusing the base_desc descriptor through a pointer.
|
||||
! With PSBLAS 4 we will have a better solution
|
||||
if (associated(prec%precv(i)%linmap%p_desc_U)) &
|
||||
& prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
if (associated(prec%precv(i)%linmap%p_desc_V))&
|
||||
& prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
end do
|
||||
end if
|
||||
|
||||
@@ -507,6 +512,71 @@ subroutine amg_z_hierarchy_bld(a,desc_a,prec,info)
|
||||
return
|
||||
|
||||
contains
|
||||
|
||||
#if ( __GNUC__ == 13 && __GNUC_MINOR__ == 3)
|
||||
! gfortran 13.3.0 generates a strange error here with MOLD
|
||||
! moving to SOURCE but only for this version, since it's heavier
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_z_onelev_type), intent(inout) :: level
|
||||
class(amg_z_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(save1)) then
|
||||
call save1%free(info)
|
||||
if (info == 0) deallocate(save1,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
call save2%free(info)
|
||||
if (info == 0) deallocate(save2,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
allocate(save1, source=level%sm,stat=info)
|
||||
if (info == 0) call level%sm%clone_settings(save1,info)
|
||||
if ((info == 0).and.allocated(level%sm2a)) then
|
||||
allocate(save2, source=level%sm2a,stat=info)
|
||||
if (info == 0) call level%sm2a%clone_settings(save2,info)
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine save_smoothers
|
||||
|
||||
subroutine restore_smoothers(level,save1, save2,info)
|
||||
type(amg_z_onelev_type), intent(inout), target :: level
|
||||
class(amg_z_base_smoother_type), allocatable, intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
|
||||
if (allocated(level%sm)) then
|
||||
if (info == 0) call level%sm%free(info)
|
||||
if (info == 0) deallocate(level%sm,stat=info)
|
||||
end if
|
||||
if (allocated(save1)) then
|
||||
if (info == 0) allocate(level%sm,source=save1,stat=info)
|
||||
if (info == 0) call save1%clone_settings(level%sm,info)
|
||||
end if
|
||||
|
||||
if (info /= 0) return
|
||||
|
||||
if (allocated(level%sm2a)) then
|
||||
if (info == 0) call level%sm2a%free(info)
|
||||
if (info == 0) deallocate(level%sm2a,stat=info)
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
if (info == 0) allocate(level%sm2a,source=save2,stat=info)
|
||||
if (info == 0) call save2%clone_settings(level%sm2a,info)
|
||||
if (info == 0) level%sm2 => level%sm2a
|
||||
else
|
||||
if (allocated(level%sm)) level%sm2 => level%sm
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#else
|
||||
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_z_onelev_type), intent(inout) :: level
|
||||
class(amg_z_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
@@ -565,5 +635,5 @@ contains
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#endif
|
||||
end subroutine amg_z_hierarchy_bld
|
||||
@@ -241,6 +241,8 @@ subroutine amg_c_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
if (info == 0) deallocate(lv%aggr,stat=info)
|
||||
if (info /= 0) then
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='aggregator deallocation?')
|
||||
goto 9999
|
||||
return
|
||||
end if
|
||||
end if
|
||||
@@ -250,8 +252,16 @@ subroutine amg_c_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
allocate(amg_c_dec_aggregator_type :: lv%aggr, stat=info)
|
||||
case('SYMDEC')
|
||||
allocate(amg_c_symdec_aggregator_type :: lv%aggr, stat=info)
|
||||
case default
|
||||
#if !defined(SERIAL_MPI)
|
||||
#endif
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
#if !defined(SERIAL_MPI)
|
||||
call psb_errpush(info,name,a_err='Unsupported PAR_AGGR_ALG')
|
||||
#else
|
||||
call psb_errpush(info,name,a_err='PAR_AGGR_ALG unsupported (SERIAL_MPI on)')
|
||||
#endif
|
||||
goto 9999
|
||||
end select
|
||||
if (info == psb_success_) call lv%aggr%default()
|
||||
|
||||
|
||||
@@ -129,7 +129,7 @@ subroutine amg_c_base_onelev_descr(lv,il,nl,ilmin,info,iout,verbosity,prefix)
|
||||
& ' avg:', &
|
||||
& lv%linmap%nagavg
|
||||
end if
|
||||
write(iout_,'(a,1xa,1x,f14.2)') trim(prefix_),&
|
||||
write(iout_,'(a,1x,a,1x,f14.2)') trim(prefix_),&
|
||||
& ' Aggregation ratio: ', &
|
||||
& lv%szratio
|
||||
end if
|
||||
|
||||
@@ -127,8 +127,7 @@ subroutine amg_c_base_onelev_dump(lv,level,info,prefix,head,ac,rp,&
|
||||
ivr = lv%linmap%p_desc_U%get_global_indices(owned=.false.)
|
||||
write(fname(lname+1:),'(a,i3.3,a)')'_l',level,'_tprol.mtx'
|
||||
!
|
||||
! This is not implemented yet.
|
||||
!call lv%tprol%print(fname,head=head,ivr=ivr)
|
||||
call lv%tprol%print(fname,head=head,ivr=ivr)
|
||||
end if
|
||||
end if
|
||||
else
|
||||
@@ -151,8 +150,7 @@ subroutine amg_c_base_onelev_dump(lv,level,info,prefix,head,ac,rp,&
|
||||
if (tprol_) then
|
||||
write(fname(lname+1:),'(a,i3.3,a)')'_l',level,'_tprol.mtx'
|
||||
!
|
||||
! This is not implemented yet.
|
||||
!call lv%tprol%print(fname,head=head)
|
||||
call lv%tprol%print(fname,head=head)
|
||||
end if
|
||||
end if
|
||||
end if
|
||||
|
||||
@@ -89,7 +89,7 @@ subroutine amg_c_base_onelev_memory_use(lv,il,nl,ilmin,info,iout,verbosity,prefi
|
||||
if (present(global)) then
|
||||
global_ = global
|
||||
else
|
||||
global_ = .false.
|
||||
global_ = .true.
|
||||
end if
|
||||
|
||||
if (present(prefix)) then
|
||||
@@ -98,8 +98,7 @@ subroutine amg_c_base_onelev_memory_use(lv,il,nl,ilmin,info,iout,verbosity,prefi
|
||||
prefix_ = ""
|
||||
end if
|
||||
|
||||
write(iout_,*) trim(prefix_)
|
||||
|
||||
if ((me == 0).or.(verbosity_>0)) write(iout_,*) trim(prefix_)
|
||||
|
||||
if (global_) then
|
||||
allocate(sz(6))
|
||||
|
||||
@@ -42,7 +42,9 @@ subroutine amg_d_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
use amg_d_base_aggregator_mod
|
||||
use amg_d_dec_aggregator_mod
|
||||
use amg_d_symdec_aggregator_mod
|
||||
#if !defined(SERIAL_MPI)
|
||||
use amg_d_parmatch_aggregator_mod
|
||||
#endif
|
||||
use amg_d_poly_smoother
|
||||
use amg_d_jac_smoother
|
||||
use amg_d_as_smoother
|
||||
@@ -267,6 +269,8 @@ subroutine amg_d_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
if (info == 0) deallocate(lv%aggr,stat=info)
|
||||
if (info /= 0) then
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='aggregator deallocation?')
|
||||
goto 9999
|
||||
return
|
||||
end if
|
||||
end if
|
||||
@@ -276,10 +280,18 @@ subroutine amg_d_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
allocate(amg_d_dec_aggregator_type :: lv%aggr, stat=info)
|
||||
case('SYMDEC')
|
||||
allocate(amg_d_symdec_aggregator_type :: lv%aggr, stat=info)
|
||||
#if !defined(SERIAL_MPI)
|
||||
case('COUP','COUPLED')
|
||||
allocate(amg_d_parmatch_aggregator_type :: lv%aggr, stat=info)
|
||||
case default
|
||||
#endif
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
#if !defined(SERIAL_MPI)
|
||||
call psb_errpush(info,name,a_err='Unsupported PAR_AGGR_ALG')
|
||||
#else
|
||||
call psb_errpush(info,name,a_err='PAR_AGGR_ALG unsupported (SERIAL_MPI on)')
|
||||
#endif
|
||||
goto 9999
|
||||
end select
|
||||
if (info == psb_success_) call lv%aggr%default()
|
||||
|
||||
|
||||
@@ -129,7 +129,7 @@ subroutine amg_d_base_onelev_descr(lv,il,nl,ilmin,info,iout,verbosity,prefix)
|
||||
& ' avg:', &
|
||||
& lv%linmap%nagavg
|
||||
end if
|
||||
write(iout_,'(a,1xa,1x,f14.2)') trim(prefix_),&
|
||||
write(iout_,'(a,1x,a,1x,f14.2)') trim(prefix_),&
|
||||
& ' Aggregation ratio: ', &
|
||||
& lv%szratio
|
||||
end if
|
||||
|
||||
@@ -127,8 +127,7 @@ subroutine amg_d_base_onelev_dump(lv,level,info,prefix,head,ac,rp,&
|
||||
ivr = lv%linmap%p_desc_U%get_global_indices(owned=.false.)
|
||||
write(fname(lname+1:),'(a,i3.3,a)')'_l',level,'_tprol.mtx'
|
||||
!
|
||||
! This is not implemented yet.
|
||||
!call lv%tprol%print(fname,head=head,ivr=ivr)
|
||||
call lv%tprol%print(fname,head=head,ivr=ivr)
|
||||
end if
|
||||
end if
|
||||
else
|
||||
@@ -151,8 +150,7 @@ subroutine amg_d_base_onelev_dump(lv,level,info,prefix,head,ac,rp,&
|
||||
if (tprol_) then
|
||||
write(fname(lname+1:),'(a,i3.3,a)')'_l',level,'_tprol.mtx'
|
||||
!
|
||||
! This is not implemented yet.
|
||||
!call lv%tprol%print(fname,head=head)
|
||||
call lv%tprol%print(fname,head=head)
|
||||
end if
|
||||
end if
|
||||
end if
|
||||
|
||||
@@ -89,7 +89,7 @@ subroutine amg_d_base_onelev_memory_use(lv,il,nl,ilmin,info,iout,verbosity,prefi
|
||||
if (present(global)) then
|
||||
global_ = global
|
||||
else
|
||||
global_ = .false.
|
||||
global_ = .true.
|
||||
end if
|
||||
|
||||
if (present(prefix)) then
|
||||
@@ -98,8 +98,7 @@ subroutine amg_d_base_onelev_memory_use(lv,il,nl,ilmin,info,iout,verbosity,prefi
|
||||
prefix_ = ""
|
||||
end if
|
||||
|
||||
write(iout_,*) trim(prefix_)
|
||||
|
||||
if ((me == 0).or.(verbosity_>0)) write(iout_,*) trim(prefix_)
|
||||
|
||||
if (global_) then
|
||||
allocate(sz(6))
|
||||
|
||||
@@ -42,7 +42,9 @@ subroutine amg_s_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
use amg_s_base_aggregator_mod
|
||||
use amg_s_dec_aggregator_mod
|
||||
use amg_s_symdec_aggregator_mod
|
||||
#if !defined(SERIAL_MPI)
|
||||
use amg_s_parmatch_aggregator_mod
|
||||
#endif
|
||||
use amg_s_poly_smoother
|
||||
use amg_s_jac_smoother
|
||||
use amg_s_as_smoother
|
||||
@@ -247,6 +249,8 @@ subroutine amg_s_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
if (info == 0) deallocate(lv%aggr,stat=info)
|
||||
if (info /= 0) then
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='aggregator deallocation?')
|
||||
goto 9999
|
||||
return
|
||||
end if
|
||||
end if
|
||||
@@ -256,10 +260,18 @@ subroutine amg_s_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
allocate(amg_s_dec_aggregator_type :: lv%aggr, stat=info)
|
||||
case('SYMDEC')
|
||||
allocate(amg_s_symdec_aggregator_type :: lv%aggr, stat=info)
|
||||
#if !defined(SERIAL_MPI)
|
||||
case('COUP','COUPLED')
|
||||
allocate(amg_s_parmatch_aggregator_type :: lv%aggr, stat=info)
|
||||
case default
|
||||
#endif
|
||||
case default
|
||||
info = psb_err_internal_error_
|
||||
#if !defined(SERIAL_MPI)
|
||||
call psb_errpush(info,name,a_err='Unsupported PAR_AGGR_ALG')
|
||||
#else
|
||||
call psb_errpush(info,name,a_err='PAR_AGGR_ALG unsupported (SERIAL_MPI on)')
|
||||
#endif
|
||||
goto 9999
|
||||
end select
|
||||
if (info == psb_success_) call lv%aggr%default()
|
||||
|
||||
|
||||
@@ -129,7 +129,7 @@ subroutine amg_s_base_onelev_descr(lv,il,nl,ilmin,info,iout,verbosity,prefix)
|
||||
& ' avg:', &
|
||||
& lv%linmap%nagavg
|
||||
end if
|
||||
write(iout_,'(a,1xa,1x,f14.2)') trim(prefix_),&
|
||||
write(iout_,'(a,1x,a,1x,f14.2)') trim(prefix_),&
|
||||
& ' Aggregation ratio: ', &
|
||||
& lv%szratio
|
||||
end if
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user