mirror of
https://github.com/sfilippone/amg4psblas.git
synced 2026-10-07 15:15:07 +00:00
Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
843ea99b17 | ||
|
|
8c00a7cabe | ||
|
|
b4c60ae409 | ||
|
|
5b17e1bbf1 | ||
|
|
0f3c3380cb | ||
|
|
15bb4ac101 | ||
|
|
12356f65f6 |
@@ -323,10 +323,10 @@ module amg_base_prec_type
|
||||
!
|
||||
! Legal values for entry: amg_poly_variant_
|
||||
!
|
||||
integer(psb_ipk_), parameter :: amg_cheb_4_ = 0
|
||||
integer(psb_ipk_), parameter :: amg_cheb_4_opt_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_cheb_1_opt_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_poly_dbg_ = 8
|
||||
integer(psb_ipk_), parameter :: amg_poly_lottes_ = 0
|
||||
integer(psb_ipk_), parameter :: amg_poly_lottes_beta_ = 1
|
||||
integer(psb_ipk_), parameter :: amg_poly_new_ = 2
|
||||
integer(psb_ipk_), parameter :: amg_poly_dbg_ = 8
|
||||
|
||||
integer(psb_ipk_), parameter :: amg_poly_rho_est_power_ = 0
|
||||
|
||||
@@ -570,12 +570,12 @@ contains
|
||||
val = amg_as_
|
||||
case('POLY')
|
||||
val = amg_poly_
|
||||
case('CHEB_4')
|
||||
val = amg_cheb_4_
|
||||
case('CHEB_4_OPT')
|
||||
val = amg_cheb_4_opt_
|
||||
case('CHEB_1_OPT')
|
||||
val = amg_cheb_1_opt_
|
||||
case('POLY_LOTTES')
|
||||
val = amg_poly_lottes_
|
||||
case('POLY_LOTTES_BETA')
|
||||
val = amg_poly_lottes_beta_
|
||||
case('POLY_NEW')
|
||||
val = amg_poly_new_
|
||||
case('POLY_DBG')
|
||||
val = amg_poly_dbg_
|
||||
case('POLY_RHO_EST_POWER')
|
||||
|
||||
@@ -322,7 +322,7 @@ contains
|
||||
!
|
||||
sm%pdegree = 1
|
||||
sm%rho_ba = -done
|
||||
sm%variant = amg_cheb_4_
|
||||
sm%variant = amg_poly_lottes_
|
||||
sm%rho_estimate = amg_poly_rho_est_power_
|
||||
sm%rho_estimate_iterations = 20
|
||||
if (allocated(sm%sv)) then
|
||||
|
||||
@@ -322,7 +322,7 @@ contains
|
||||
!
|
||||
sm%pdegree = 1
|
||||
sm%rho_ba = -sone
|
||||
sm%variant = amg_cheb_4_
|
||||
sm%variant = amg_poly_lottes_
|
||||
sm%rho_estimate = amg_poly_rho_est_power_
|
||||
sm%rho_estimate_iterations = 20
|
||||
if (allocated(sm%sv)) then
|
||||
|
||||
@@ -42,7 +42,6 @@
|
||||
#include <stdlib.h>
|
||||
#if !defined(SERIAL_MPI)
|
||||
#include <mpi.h>
|
||||
#endif
|
||||
|
||||
#include "MatchBoxPC.h"
|
||||
#ifdef __cplusplus
|
||||
@@ -67,13 +66,13 @@ void dMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
#endif
|
||||
|
||||
|
||||
#undef TIME_TRACKER
|
||||
#ifdef TIME_TRACKER
|
||||
double tmr = MPI_Wtime();
|
||||
#endif
|
||||
#define TIME_TRACKER
|
||||
#ifdef TIME_TRACKER
|
||||
double tmr = MPI_Wtime();
|
||||
#endif
|
||||
|
||||
#if defined(OPENMP)
|
||||
//fprintf(stderr,"Warning: using buggy OpenMP matching!\n");
|
||||
#define OMP
|
||||
#ifdef OMP
|
||||
dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(NLVer, NLEdge,
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
@@ -92,11 +91,11 @@ void dMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
#endif
|
||||
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
tmr = MPI_Wtime() - tmr;
|
||||
fprintf(stderr, "Elaboration time: %f for %ld nodes\n", tmr, NLVer);
|
||||
#endif
|
||||
|
||||
#ifdef TIME_TRACKER
|
||||
tmr = MPI_Wtime() - tmr;
|
||||
fprintf(stderr, "Elaboration time: %f for %ld nodes\n", tmr, NLVer);
|
||||
#endif
|
||||
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -114,27 +113,17 @@ void sMatchBoxPC(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
fprintf(stderr,"MatchBoxPC: rank %d nlver %ld nledge %ld [ %ld %ld ]\n",
|
||||
myRank,NLVer, NLEdge,verDistance[0],verDistance[1]);
|
||||
#endif
|
||||
#if defined(OPENMP)
|
||||
//fprintf(stderr,"Warning: using buggy OpenMP matching!\n");
|
||||
salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(NLVer, NLEdge,
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
myRank, numProcs, C_comm,
|
||||
msgIndSent, msgActualSent, msgPercent,
|
||||
ph0_time, ph1_time, ph2_time,
|
||||
ph1_card, ph2_card );
|
||||
#else
|
||||
salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(NLVer, NLEdge,
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
myRank, numProcs, C_comm,
|
||||
msgIndSent, msgActualSent, msgPercent,
|
||||
ph0_time, ph1_time, ph2_time,
|
||||
ph1_card, ph2_card );
|
||||
#endif
|
||||
verLocPtr, verLocInd, edgeLocWeight,
|
||||
verDistance, Mate,
|
||||
myRank, numProcs, C_comm,
|
||||
msgIndSent, msgActualSent, msgPercent,
|
||||
ph0_time, ph1_time, ph2_time,
|
||||
ph1_card, ph2_card );
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -59,11 +59,7 @@
|
||||
#include <assert.h>
|
||||
#include <map>
|
||||
#include <vector>
|
||||
#ifdef OPENMP
|
||||
// OpenMP is included and used if and only if the OpenMP version of the matching
|
||||
// is required
|
||||
#include "omp.h"
|
||||
#endif
|
||||
#include "primitiveDataTypeDefinitions.h"
|
||||
#include "dataStrStaticQueue.h"
|
||||
|
||||
@@ -181,416 +177,252 @@ extern "C"
|
||||
#define MilanRealMin MINUS_INFINITY
|
||||
#endif
|
||||
|
||||
/* These functions are only used in the experimental OMP implementation, if that
|
||||
is disabled there is no reason to actually compile or reference them. */
|
||||
// Function of find the owner of a ghost vertex using binary search:
|
||||
MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs);
|
||||
|
||||
// Function of find the owner of a ghost vertex using binary search:
|
||||
MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs);
|
||||
|
||||
MilanLongInt firstComputeCandidateMateD(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight);
|
||||
MilanLongInt firstComputeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight);
|
||||
|
||||
|
||||
void queuesTransfer(vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
bool isAlreadyMatched(MilanLongInt node,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
MilanLongInt computeCandidateMateD(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt StartIndex, MilanLongInt EndIndex,
|
||||
MilanLongInt *numGhostEdgesPtr,
|
||||
MilanLongInt *numGhostVerticesPtr,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &Counter,
|
||||
vector<MilanLongInt> &verGhostPtr,
|
||||
vector<MilanLongInt> &verGhostInd,
|
||||
vector<MilanLongInt> &tempCounter,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Message,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
MilanLongInt *&candidateMate,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void clean(MilanLongInt NLVer,
|
||||
MilanInt myRank,
|
||||
MilanLongInt MessageIndex,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus,
|
||||
MilanInt BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
MilanLongInt msgActual,
|
||||
MilanLongInt *msgActualSent,
|
||||
MilanLongInt msgInd,
|
||||
MilanLongInt *msgIndSent,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanReal *msgPercent);
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_BD(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *candidateMate);
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_BD(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
MilanLongInt *SPtr);
|
||||
|
||||
void processMatchedVerticesD(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void processMatchedVerticesAndSendMessagesD(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message);
|
||||
|
||||
void sendBundledMessages(MilanLongInt *numGhostEdgesPtr,
|
||||
MilanInt *BufferSizePtr,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanLongInt *MessageIndexPtr,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus);
|
||||
|
||||
void processMessagesD(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *SPtr,
|
||||
vector<MilanLongInt> &U);
|
||||
|
||||
void extractUChunk(
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU);
|
||||
void queuesTransfer(vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
MilanLongInt firstComputeCandidateMateS(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanFloat *edgeLocWeight);
|
||||
bool isAlreadyMatched(MilanLongInt node,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
MilanLongInt computeCandidateMateS(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_BS(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *candidateMate);
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_BS(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
void processMatchedVerticesS(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void processMatchedVerticesAndSendMessagesS(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message);
|
||||
|
||||
void processMessagesS(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *SPtr,
|
||||
vector<MilanLongInt> &U);
|
||||
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
MilanLongInt computeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap);
|
||||
|
||||
|
||||
void salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt StartIndex, MilanLongInt EndIndex,
|
||||
MilanLongInt *numGhostEdgesPtr,
|
||||
MilanLongInt *numGhostVerticesPtr,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &Counter,
|
||||
vector<MilanLongInt> &verGhostPtr,
|
||||
vector<MilanLongInt> &verGhostInd,
|
||||
vector<MilanLongInt> &tempCounter,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Message,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
MilanLongInt *&candidateMate,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void clean(MilanLongInt NLVer,
|
||||
MilanInt myRank,
|
||||
MilanLongInt MessageIndex,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus,
|
||||
MilanInt BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
MilanLongInt msgActual,
|
||||
MilanLongInt *msgActualSent,
|
||||
MilanLongInt msgInd,
|
||||
MilanLongInt *msgIndSent,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanReal *msgPercent);
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_B(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *candidateMate);
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
MilanLongInt *SPtr);
|
||||
|
||||
void processMatchedVertices(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner);
|
||||
|
||||
void processMatchedVerticesAndSendMessages(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *NumMessagesBundledPtr,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanReal *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message);
|
||||
|
||||
void sendBundledMessages(MilanLongInt *numGhostEdgesPtr,
|
||||
MilanInt *BufferSizePtr,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanLongInt *MessageIndexPtr,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus);
|
||||
|
||||
void processMessages(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCardPtr,
|
||||
MilanLongInt *msgIndPtr,
|
||||
MilanLongInt *msgActualPtr,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *SPtr,
|
||||
vector<MilanLongInt> &U);
|
||||
|
||||
void extractUChunk(
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU);
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd, MilanReal *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent, MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card);
|
||||
|
||||
void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
|
||||
+10
-10
@@ -1303,16 +1303,16 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(
|
||||
// SINGLE PRECISION VERSION
|
||||
|
||||
void salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateC(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* verLocPtr, MilanLongInt* verLocInd,
|
||||
MilanFloat* edgeLocWeight,
|
||||
MilanLongInt* verDistance,
|
||||
MilanLongInt* Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt* msgIndSent, MilanLongInt* msgActualSent,
|
||||
MilanReal* msgPercent,
|
||||
MilanReal* ph0_time, MilanReal* ph1_time, MilanReal* ph2_time,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card ) {
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt* verLocPtr, MilanLongInt* verLocInd,
|
||||
MilanFloat* edgeLocWeight,
|
||||
MilanLongInt* verDistance,
|
||||
MilanLongInt* Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt* msgIndSent, MilanLongInt* msgActualSent,
|
||||
MilanReal* msgPercent,
|
||||
MilanReal* ph0_time, MilanReal* ph1_time, MilanReal* ph2_time,
|
||||
MilanLongInt* ph1_card, MilanLongInt* ph2_card ) {
|
||||
#if !defined(SERIAL_MPI)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout<<"\n("<<myRank<<")Within algoEdgeApproxDominatingEdgesLinearSearchMessageBundling()"; fflush(stdout);
|
||||
|
||||
+18
-507
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
// ***********************************************************************
|
||||
//
|
||||
// MatchboxP: A C++ library for approximate weighted matching
|
||||
@@ -125,10 +126,8 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// The starting vertex owned by the current rank
|
||||
MilanLongInt StartIndex = verDistance[myRank];
|
||||
// The ending vertex owned by the current rank
|
||||
MilanLongInt EndIndex = verDistance[myRank + 1] - 1;
|
||||
MilanLongInt StartIndex = verDistance[myRank]; // The starting vertex owned by the current rank
|
||||
MilanLongInt EndIndex = verDistance[myRank + 1] - 1; // The ending vertex owned by the current rank
|
||||
|
||||
MPI_Status computeStatus;
|
||||
|
||||
@@ -146,8 +145,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
// only one message will be sent in the initialization phase -
|
||||
// one of: REQUEST/FAILURE/SUCCESS
|
||||
vector<MilanLongInt> QLocalVtx, QGhostVtx, QMsgType;
|
||||
// Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
vector<MilanInt> QOwner;
|
||||
vector<MilanInt> QOwner; // Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
|
||||
MilanLongInt *PCounter = new MilanLongInt[numProcs];
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
@@ -155,8 +153,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
|
||||
MilanLongInt NumMessagesBundled = 0;
|
||||
// TODO when the last computational section will be refactored this could be eliminated
|
||||
// Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
MilanInt ghostOwner = 0;
|
||||
MilanInt ghostOwner = 0; // Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
MilanLongInt *candidateMate = nullptr;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")NV: " << NLVer << " Edges: " << NLEdge;
|
||||
@@ -171,12 +168,9 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt myCard = 0;
|
||||
|
||||
// Build the Ghost Vertex Set: Vg
|
||||
// Map each ghost vertex to a local vertex
|
||||
map<MilanLongInt, MilanLongInt> Ghost2LocalMap;
|
||||
// Store the edge count for each ghost vertex
|
||||
vector<MilanLongInt> Counter;
|
||||
// Number of Ghost vertices
|
||||
MilanLongInt numGhostVertices = 0, numGhostEdges = 0;
|
||||
map<MilanLongInt, MilanLongInt> Ghost2LocalMap; // Map each ghost vertex to a local vertex
|
||||
vector<MilanLongInt> Counter; // Store the edge count for each ghost vertex
|
||||
MilanLongInt numGhostVertices = 0, numGhostEdges = 0; // Number of Ghost vertices
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")About to compute Ghost Vertices...";
|
||||
@@ -228,7 +222,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
cout << myRank << " Finished initialization" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
startTime = MPI_Wtime();
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
@@ -243,7 +237,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
* PARALLEL_COMPUTE_CANDIDATE_MATE_B is now totally parallel.
|
||||
*/
|
||||
|
||||
PARALLEL_COMPUTE_CANDIDATE_MATE_BD(NLVer,
|
||||
PARALLEL_COMPUTE_CANDIDATE_MATE_B(NLVer,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
myRank,
|
||||
@@ -268,7 +262,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
* TODO: Test when it's actually more efficient to execute this code
|
||||
* in parallel.
|
||||
*/
|
||||
PARALLEL_PROCESS_EXPOSED_VERTEX_BD(NLVer,
|
||||
PARALLEL_PROCESS_EXPOSED_VERTEX_B(NLVer,
|
||||
candidateMate,
|
||||
verLocInd,
|
||||
verLocPtr,
|
||||
@@ -320,7 +314,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
vector<MilanLongInt> UChunkBeingProcessed;
|
||||
UChunkBeingProcessed.reserve(UCHUNK);
|
||||
|
||||
processMatchedVerticesD(NLVer,
|
||||
processMatchedVertices(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
@@ -397,7 +391,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
cout << myRank << " Finished sendBundles" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
*ph1_card = myCard; // Cardinality at the end of Phase-1
|
||||
startTime = MPI_Wtime();
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
@@ -428,8 +422,8 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MATCHED VERTICES //////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMatchedVerticesAndSendMessagesD(NLVer,
|
||||
|
||||
processMatchedVerticesAndSendMessages(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
@@ -462,7 +456,7 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
comm,
|
||||
&msgActual,
|
||||
Message);
|
||||
|
||||
|
||||
///////////////////////// END OF PROCESS MATCHED VERTICES /////////////////////////
|
||||
|
||||
//// BREAK IF NO MESSAGES EXPECTED /////////
|
||||
@@ -489,8 +483,8 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MESSAGES //////////////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
//startTime = MPI_Wtime();
|
||||
processMessagesD(NLVer,
|
||||
|
||||
processMessages(NLVer,
|
||||
Mate,
|
||||
candidateMate,
|
||||
Ghost2LocalMap,
|
||||
@@ -555,489 +549,6 @@ void dalgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
*ph2_card = myCard; // Cardinality at the end of Phase-2
|
||||
}
|
||||
// End of algoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMate
|
||||
|
||||
void salgoDistEdgeApproxDomEdgesLinearSearchMesgBndlSmallMateCMP(
|
||||
MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt *verLocPtr, MilanLongInt *verLocInd,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *Mate,
|
||||
MilanInt myRank, MilanInt numProcs, MPI_Comm comm,
|
||||
MilanLongInt *msgIndSent, MilanLongInt *msgActualSent,
|
||||
MilanReal *msgPercent,
|
||||
MilanReal *ph0_time, MilanReal *ph1_time, MilanReal *ph2_time,
|
||||
MilanLongInt *ph1_card, MilanLongInt *ph2_card)
|
||||
{
|
||||
|
||||
/*
|
||||
* verDistance: it's a vector long as the number of processors.
|
||||
* verDistance[i] contains the first node index of the i-th processor
|
||||
* verDistance[i + 1] contains the last node index of the i-th processor
|
||||
* NLVer: number of elements in the LocPtr
|
||||
* NLEdge: number of edges assigned to the current processor
|
||||
*
|
||||
* Contains the portion of matrix assigned to the processor in
|
||||
* Yale notation
|
||||
* verLocInd: contains the positions on row of the matrix
|
||||
* verLocPtr: i-th value is the position of the first element on the i-th row and
|
||||
* i+1-th value is the position of the first element on the i+1-th row
|
||||
*/
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Within algoEdgeApproxDominatingEdgesLinearSearchMessageBundling()";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ") verDistance [" ;
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
cout << verDistance[i] << "," << verDistance[i+1];
|
||||
cout << "]\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0) {
|
||||
cout << "\n(" << myRank << ") verDistance [" ;
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
cout << verDistance[i] << "," ;
|
||||
cout << verDistance[numProcs]<< "]\n";
|
||||
}
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// The starting vertex owned by the current rank
|
||||
MilanLongInt StartIndex = verDistance[myRank];
|
||||
// The ending vertex owned by the current rank
|
||||
MilanLongInt EndIndex = verDistance[myRank + 1] - 1;
|
||||
|
||||
MPI_Status computeStatus;
|
||||
|
||||
MilanLongInt msgActual = 0, msgInd = 0;
|
||||
MilanFloat heaviestEdgeWt = 0.0f; // Assumes positive weight
|
||||
MilanReal startTime, finishTime;
|
||||
|
||||
startTime = MPI_Wtime();
|
||||
|
||||
// Data structures for sending and receiving messages:
|
||||
vector<MilanLongInt> Message; // [ u, v, message_type ]
|
||||
Message.resize(3, -1);
|
||||
// Data structures for Message Bundling:
|
||||
// Although up to two messages can be sent along any cross edge,
|
||||
// only one message will be sent in the initialization phase -
|
||||
// one of: REQUEST/FAILURE/SUCCESS
|
||||
vector<MilanLongInt> QLocalVtx, QGhostVtx, QMsgType;
|
||||
// Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
vector<MilanInt> QOwner;
|
||||
|
||||
MilanLongInt *PCounter = new MilanLongInt[numProcs];
|
||||
for (int i = 0; i < numProcs; i++)
|
||||
PCounter[i] = 0;
|
||||
|
||||
MilanLongInt NumMessagesBundled = 0;
|
||||
// TODO when the last computational section will be refactored this could be eliminated
|
||||
// Changed by Fabio to be an integer, addresses needs to be integers!
|
||||
MilanInt ghostOwner = 0;
|
||||
MilanLongInt *candidateMate = nullptr;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")NV: " << NLVer << " Edges: " << NLEdge;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")StartIndex: " << StartIndex << " EndIndex: " << EndIndex;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Other Variables:
|
||||
MilanLongInt u = -1, v = -1, w = -1, i = 0;
|
||||
MilanLongInt k = -1, adj1 = -1, adj2 = -1;
|
||||
MilanLongInt k1 = -1, adj11 = -1, adj12 = -1;
|
||||
MilanLongInt myCard = 0;
|
||||
|
||||
// Build the Ghost Vertex Set: Vg
|
||||
// Map each ghost vertex to a local vertex
|
||||
map<MilanLongInt, MilanLongInt> Ghost2LocalMap;
|
||||
// Store the edge count for each ghost vertex
|
||||
vector<MilanLongInt> Counter;
|
||||
// Number of Ghost vertices
|
||||
MilanLongInt numGhostVertices = 0, numGhostEdges = 0;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")About to compute Ghost Vertices...";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_HANG_
|
||||
if (myRank == 0)
|
||||
cout << "\n(" << myRank << ")About to compute Ghost Vertices...";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Define Adjacency Lists for Ghost Vertices:
|
||||
// cout<<"Building Ghost data structures ... \n\n";
|
||||
vector<MilanLongInt> verGhostPtr, verGhostInd, tempCounter;
|
||||
// Mate array for ghost vertices:
|
||||
vector<MilanLongInt> GMate; // Proportional to the number of ghost vertices
|
||||
MilanLongInt S;
|
||||
MilanLongInt privateMyCard = 0;
|
||||
vector<MilanLongInt> PCumulative, PMessageBundle, PSizeInfoMessages;
|
||||
vector<MPI_Request> SRequest; // Requests that are used for each send message
|
||||
vector<MPI_Status> SStatus; // Status of sent messages, used in MPI_Wait
|
||||
MilanLongInt MessageIndex = 0; // Pointer for current message
|
||||
MilanInt BufferSize;
|
||||
MilanLongInt *Buffer;
|
||||
|
||||
vector<MilanLongInt> privateQLocalVtx, privateQGhostVtx, privateQMsgType;
|
||||
vector<MilanInt> privateQOwner;
|
||||
vector<MilanLongInt> U, privateU;
|
||||
|
||||
|
||||
initialize(NLVer, NLEdge, StartIndex,
|
||||
EndIndex, &numGhostEdges,
|
||||
&numGhostVertices, &S,
|
||||
verLocInd, verLocPtr,
|
||||
Ghost2LocalMap, Counter,
|
||||
verGhostPtr, verGhostInd,
|
||||
tempCounter, GMate,
|
||||
Message, QLocalVtx,
|
||||
QGhostVtx, QMsgType, QOwner,
|
||||
candidateMate, U,
|
||||
privateU,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph0_time = finishTime - startTime; // Time taken for Phase-0: Initialization
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished initialization" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
startTime = MPI_Wtime();
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
//////////////////////////////////// INITIALIZATION /////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
// Compute the Initial Matching Set:
|
||||
|
||||
/*
|
||||
* OMP PARALLEL_COMPUTE_CANDIDATE_MATE_B has been splitted from
|
||||
* PARALLEL_PROCESS_EXPOSED_VERTEX_B in order to better parallelize
|
||||
* the two.
|
||||
* PARALLEL_COMPUTE_CANDIDATE_MATE_B is now totally parallel.
|
||||
*/
|
||||
|
||||
PARALLEL_COMPUTE_CANDIDATE_MATE_BS(NLVer,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
myRank,
|
||||
edgeLocWeight,
|
||||
candidateMate);
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Exposed Vertex" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " candidateMate after parallelCompute " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << candidateMate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
/*
|
||||
* PARALLEL_PROCESS_EXPOSED_VERTEX_B
|
||||
* TODO: write comment
|
||||
*
|
||||
* TODO: Test when it's actually more efficient to execute this code
|
||||
* in parallel.
|
||||
*/
|
||||
PARALLEL_PROCESS_EXPOSED_VERTEX_BS(NLVer,
|
||||
candidateMate,
|
||||
verLocInd,
|
||||
verLocPtr,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
Mate,
|
||||
GMate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
U,
|
||||
privateU,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
tempCounter.clear(); // Do not need this any more
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Exposed Vertex" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " Mate after Exposed Vertices " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MATCHED VERTICES //////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
// TODO what would be the optimal UCHUNK
|
||||
vector<MilanLongInt> UChunkBeingProcessed;
|
||||
UChunkBeingProcessed.reserve(UCHUNK);
|
||||
|
||||
processMatchedVerticesS(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
candidateMate,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished Process Vertices" << endl;
|
||||
fflush(stdout);
|
||||
#if 0
|
||||
cout << myRank << " Mate after Matched Vertices " <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
///////////////////////////// SEND BUNDLED MESSAGES /////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
sendBundledMessages(&numGhostEdges,
|
||||
&BufferSize,
|
||||
Buffer,
|
||||
PCumulative,
|
||||
PMessageBundle,
|
||||
PSizeInfoMessages,
|
||||
PCounter,
|
||||
NumMessagesBundled,
|
||||
&msgActual,
|
||||
&MessageIndex,
|
||||
numProcs,
|
||||
myRank,
|
||||
comm,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
SRequest,
|
||||
SStatus);
|
||||
|
||||
///////////////////////// END OF SEND BUNDLED MESSAGES //////////////////////////////////
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph1_time = finishTime - startTime; // Time taken for Phase-1
|
||||
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank << " Finished sendBundles" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
*ph1_card = myCard; // Cardinality at the end of Phase-1
|
||||
startTime = MPI_Wtime();
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
//////////////////////////////////////// MAIN LOOP //////////////////////////////////////
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
// Main While Loop:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Entering While(true) loop..";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
while (true) {
|
||||
#ifdef DEBUG_HANG_
|
||||
//if (myRank == 0)
|
||||
cout << "\n(" << myRank << ") Main loop" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MATCHED VERTICES //////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMatchedVerticesAndSendMessagesS(NLVer,
|
||||
UChunkBeingProcessed,
|
||||
U,
|
||||
privateU,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&NumMessagesBundled,
|
||||
&S,
|
||||
verLocPtr,
|
||||
verLocInd,
|
||||
verDistance,
|
||||
PCounter,
|
||||
Counter,
|
||||
myRank,
|
||||
numProcs,
|
||||
candidateMate,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap,
|
||||
edgeLocWeight,
|
||||
QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType,
|
||||
QOwner,
|
||||
privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner,
|
||||
comm,
|
||||
&msgActual,
|
||||
Message);
|
||||
|
||||
///////////////////////// END OF PROCESS MATCHED VERTICES /////////////////////////
|
||||
|
||||
//// BREAK IF NO MESSAGES EXPECTED /////////
|
||||
#ifdef DEBUG_HANG_
|
||||
#if 0
|
||||
cout << myRank << " Mate after ProcessMatchedAndSend phase "<<S <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Deciding whether to break: S= " << S << endl;
|
||||
#endif
|
||||
|
||||
if (S == 0) {
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << "\n(" << myRank << ") Breaking out" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
break;
|
||||
}
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
/////////////////////////// PROCESS MESSAGES //////////////////////////////////////
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
processMessagesS(NLVer,
|
||||
Mate,
|
||||
candidateMate,
|
||||
Ghost2LocalMap,
|
||||
GMate,
|
||||
Counter,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
&myCard,
|
||||
&msgInd,
|
||||
&msgActual,
|
||||
edgeLocWeight,
|
||||
verDistance,
|
||||
verLocPtr,
|
||||
k,
|
||||
verLocInd,
|
||||
numProcs,
|
||||
myRank,
|
||||
comm,
|
||||
Message,
|
||||
numGhostEdges,
|
||||
u,
|
||||
v,
|
||||
&S,
|
||||
U);
|
||||
|
||||
///////////////////////// END OF PROCESS MESSAGES /////////////////////////////////
|
||||
#ifdef DEBUG_HANG_
|
||||
#if 0
|
||||
cout << myRank << " Mate after ProcessMessages phase "<<S <<endl;
|
||||
for (int i=0; i<NLVer; i++) {
|
||||
cout << Mate[i] << " " ;
|
||||
}
|
||||
cout << endl;
|
||||
#endif
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Finished Message processing phase: S= " << S;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")** SENT : ACTUAL= " << msgActual;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")** SENT : INDIVIDUAL= " << msgInd << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of while (true)
|
||||
|
||||
clean(NLVer,
|
||||
myRank,
|
||||
MessageIndex,
|
||||
SRequest,
|
||||
SStatus,
|
||||
BufferSize,
|
||||
Buffer,
|
||||
msgActual,
|
||||
msgActualSent,
|
||||
msgInd,
|
||||
msgIndSent,
|
||||
NumMessagesBundled,
|
||||
msgPercent);
|
||||
|
||||
finishTime = MPI_Wtime();
|
||||
*ph2_time = finishTime - startTime; // Time taken for Phase-2
|
||||
*ph2_card = myCard; // Cardinality at the end of Phase-2
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -300,7 +300,7 @@ subroutine amg_caggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ endif
|
||||
!!$ enddo
|
||||
!!$ if (jd == -1) then
|
||||
!!$ write(0,*) name,': Warning: there is no diagonal element', i
|
||||
!!$ write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
!!$ else
|
||||
!!$ acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
!!$ end if
|
||||
|
||||
@@ -230,7 +230,7 @@ subroutine amg_caggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
write(0,*) name,': Warning: there is no diagonal element', i
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
|
||||
@@ -246,7 +246,7 @@ subroutine amg_d_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
write(0,*) name,': Warning: there is no diagonal element', i
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
|
||||
@@ -300,7 +300,7 @@ subroutine amg_daggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ endif
|
||||
!!$ enddo
|
||||
!!$ if (jd == -1) then
|
||||
!!$ write(0,*) name,': Warning: there is no diagonal element', i
|
||||
!!$ write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
!!$ else
|
||||
!!$ acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
!!$ end if
|
||||
|
||||
@@ -112,11 +112,11 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
implicit none
|
||||
|
||||
! Arguments
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_dspmat_type), intent(in) :: a
|
||||
type(psb_desc_type), intent(inout) :: desc_a
|
||||
integer(psb_lpk_), intent(inout) :: ilaggr(:), nlaggr(:)
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol,ac,op_restr
|
||||
type(amg_dml_parms), intent(inout) :: parms
|
||||
type(psb_dspmat_type), intent(inout) :: op_prol,ac,op_restr
|
||||
type(psb_ldspmat_type), intent(inout) :: t_prol
|
||||
type(psb_desc_type), intent(inout) :: desc_ac
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
@@ -132,7 +132,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
type(psb_d_coo_sparse_mat) :: coo_prol, coo_restr
|
||||
type(psb_d_csr_sparse_mat) :: acsr1, acsrf, csr_prol, acsr
|
||||
real(psb_dpk_), allocatable :: adiag(:)
|
||||
real(psb_dpk_), allocatable :: arwsum(:)
|
||||
real(psb_dpk_), allocatable :: arwsum(:),l1rwsum(:)
|
||||
integer(psb_ipk_) :: ierr(5)
|
||||
logical :: filter_mat
|
||||
integer(psb_ipk_) :: debug_level, debug_unit, err_act
|
||||
@@ -141,6 +141,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
logical, parameter :: debug_new=.false.
|
||||
character(len=80) :: filename
|
||||
logical, parameter :: do_timings=.false.
|
||||
logical, parameter :: do_l1correction=.true.
|
||||
integer(psb_ipk_), save :: idx_spspmm=-1, idx_phase1=-1, idx_gtrans=-1, idx_phase2=-1, idx_refine=-1
|
||||
integer(psb_ipk_), save :: idx_phase3=-1, idx_cdasb=-1, idx_ptap=-1
|
||||
|
||||
@@ -200,6 +201,21 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(adiag,desc_a,info)
|
||||
if (info == psb_success_) call a%cp_to(acsr)
|
||||
! Get the l1-diagonal of D
|
||||
if (do_l1correction) then
|
||||
allocate(l1rwsum(nrow))
|
||||
call acsr%arwsum(l1rwsum)
|
||||
if (info == psb_success_) &
|
||||
& call psb_realloc(ncol,l1rwsum,info)
|
||||
if (info == psb_success_) &
|
||||
& call psb_halo(l1rwsum,desc_a,info)
|
||||
! \tilde{D}_{i,i} = \sum_{j \ne i} |a_{i,j}|
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
adiag(i) = adiag(i) + l1rwsum(i) - abs(adiag(i))
|
||||
end do
|
||||
!$OMP end parallel do
|
||||
end if
|
||||
|
||||
if(info /= psb_success_) then
|
||||
call psb_errpush(psb_err_from_subroutine_,name,a_err='sp_getdiag')
|
||||
@@ -230,7 +246,7 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
write(0,*) name,': Warning: there is no diagonal element', i
|
||||
if (.not.do_l1correction) write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
@@ -240,7 +256,6 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call acsrf%clean_zeros(info)
|
||||
end if
|
||||
|
||||
|
||||
!$OMP parallel do private(i) schedule(static)
|
||||
do i=1,size(adiag)
|
||||
if (adiag(i) /= dzero) then
|
||||
@@ -249,7 +264,8 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
adiag(i) = done
|
||||
end if
|
||||
end do
|
||||
!$OMP end parallel do
|
||||
!$OMP end parallel do
|
||||
|
||||
if (parms%aggr_omega_alg == amg_eig_est_) then
|
||||
|
||||
if (parms%aggr_eig == amg_max_norm_) then
|
||||
@@ -259,7 +275,9 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
call psb_amx(ctxt,anorm)
|
||||
omega = 4.d0/(3.d0*anorm)
|
||||
parms%aggr_omega_val = omega
|
||||
|
||||
else if (do_l1correction) then
|
||||
! For l1-Jacobi this can be estimated with 1
|
||||
parms%aggr_omega_val = done
|
||||
else
|
||||
info = psb_err_internal_error_
|
||||
call psb_errpush(info,name,a_err='invalid amg_aggr_eig_')
|
||||
@@ -323,6 +341,9 @@ subroutine amg_daggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
& write(debug_unit,*) me,' ',trim(name),&
|
||||
& 'Done smooth_aggregate '
|
||||
call psb_erractionrestore(err_act)
|
||||
|
||||
if (allocated(l1rwsum)) deallocate(l1rwsum)
|
||||
if (allocated(arwsum)) deallocate(arwsum)
|
||||
return
|
||||
|
||||
9999 continue
|
||||
|
||||
@@ -246,7 +246,7 @@ subroutine amg_s_parmatch_smth_bld(ag,a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
write(0,*) name,': Warning: there is no diagonal element', i
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
|
||||
@@ -300,7 +300,7 @@ subroutine amg_saggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ endif
|
||||
!!$ enddo
|
||||
!!$ if (jd == -1) then
|
||||
!!$ write(0,*) name,': Warning: there is no diagonal element', i
|
||||
!!$ write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
!!$ else
|
||||
!!$ acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
!!$ end if
|
||||
|
||||
@@ -230,7 +230,7 @@ subroutine amg_saggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
write(0,*) name,': Warning: there is no diagonal element', i
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
|
||||
@@ -300,7 +300,7 @@ subroutine amg_zaggrmat_minnrg_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
!!$ endif
|
||||
!!$ enddo
|
||||
!!$ if (jd == -1) then
|
||||
!!$ write(0,*) name,': Warning: there is no diagonal element', i
|
||||
!!$ write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
!!$ else
|
||||
!!$ acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
!!$ end if
|
||||
|
||||
@@ -230,7 +230,7 @@ subroutine amg_zaggrmat_smth_bld(a,desc_a,ilaggr,nlaggr,parms,&
|
||||
|
||||
enddo
|
||||
if (jd == -1) then
|
||||
write(0,*) name,': Warning: there is no diagonal element', i
|
||||
write(0,*) 'Wrong input: we need the diagonal!!!!', i
|
||||
else
|
||||
acsrf%val(jd)=acsrf%val(jd)-tmp
|
||||
end if
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
// TODO comment
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
/**
|
||||
* Execute the research fr the Candidate Mate without controlling if the vertices are already matched.
|
||||
* Returns the vertices with the highest weight
|
||||
@@ -8,8 +9,9 @@
|
||||
* @param edgeLocWeight
|
||||
* @return
|
||||
*/
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
MilanLongInt firstComputeCandidateMateD(MilanLongInt adj1,
|
||||
MilanLongInt firstComputeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanReal *edgeLocWeight)
|
||||
@@ -42,7 +44,7 @@ MilanLongInt firstComputeCandidateMateD(MilanLongInt adj1,
|
||||
* @param Ghost2LocalMap
|
||||
* @return
|
||||
*/
|
||||
MilanLongInt computeCandidateMateD(MilanLongInt adj1,
|
||||
MilanLongInt computeCandidateMate(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanReal *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
@@ -60,7 +62,7 @@ MilanLongInt computeCandidateMateD(MilanLongInt adj1,
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap))
|
||||
continue;
|
||||
|
||||
|
||||
if ((edgeLocWeight[k] > heaviestEdgeWt) ||
|
||||
((edgeLocWeight[k] == heaviestEdgeWt) && (w < verLocInd[k]))) {
|
||||
heaviestEdgeWt = edgeLocWeight[k];
|
||||
@@ -68,71 +70,7 @@ MilanLongInt computeCandidateMateD(MilanLongInt adj1,
|
||||
}
|
||||
} // End of for loop
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
|
||||
|
||||
return w;
|
||||
}
|
||||
|
||||
|
||||
MilanLongInt firstComputeCandidateMateS(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanFloat *edgeLocWeight)
|
||||
{
|
||||
MilanInt w = -1;
|
||||
MilanFloat heaviestEdgeWt = 0.0f; // Assign the smallest
|
||||
int finalK;
|
||||
for (int k = adj1; k < adj2; k++) {
|
||||
if ((edgeLocWeight[k] > heaviestEdgeWt) ||
|
||||
((edgeLocWeight[k] == heaviestEdgeWt) && (w < verLocInd[k]))) {
|
||||
heaviestEdgeWt = edgeLocWeight[k];
|
||||
w = verLocInd[k];
|
||||
finalK = k;
|
||||
}
|
||||
} // End of for loop
|
||||
return finalK;
|
||||
}
|
||||
|
||||
/**
|
||||
* //TODO documentation
|
||||
* @param adj1
|
||||
* @param adj2
|
||||
* @param edgeLocWeight
|
||||
* @param k
|
||||
* @param verLocInd
|
||||
* @param StartIndex
|
||||
* @param EndIndex
|
||||
* @param GMate
|
||||
* @param Mate
|
||||
* @param Ghost2LocalMap
|
||||
* @return
|
||||
*/
|
||||
MilanLongInt computeCandidateMateS(MilanLongInt adj1,
|
||||
MilanLongInt adj2,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap)
|
||||
{
|
||||
// Start: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
|
||||
MilanInt w = -1;
|
||||
MilanFloat heaviestEdgeWt = 0.0f; // Assign the smallest Value
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap))
|
||||
continue;
|
||||
|
||||
if ((edgeLocWeight[k] > heaviestEdgeWt) ||
|
||||
((edgeLocWeight[k] == heaviestEdgeWt) && (w < verLocInd[k]))) {
|
||||
heaviestEdgeWt = edgeLocWeight[k];
|
||||
w = verLocInd[k];
|
||||
}
|
||||
} // End of for loop
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
|
||||
return w;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void extractUChunk(
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
@@ -27,4 +28,4 @@ void extractUChunk(
|
||||
}
|
||||
|
||||
} // End of critical U // End of critical U
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
/// Find the owner of a ghost node:
|
||||
MilanInt findOwnerOfGhost(MilanLongInt vtxIndex, MilanLongInt *mVerDistance,
|
||||
MilanInt myRank, MilanInt numProcs)
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
MilanLongInt StartIndex, MilanLongInt EndIndex,
|
||||
MilanLongInt *numGhostEdges,
|
||||
@@ -290,7 +291,7 @@ void initialize(MilanLongInt NLVer, MilanLongInt NLEdge,
|
||||
//new (&U) staticQueue(NLVer + (*numGhostVertices));
|
||||
U.reserve(NLVer + (*numGhostVertices));
|
||||
|
||||
// Initialize the private vectors
|
||||
// Initialize the private vectors
|
||||
privateQLocalVtx.reserve(*numGhostVertices);
|
||||
privateQGhostVtx.reserve(*numGhostVertices);
|
||||
privateQMsgType.reserve(*numGhostVertices);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
/**
|
||||
* //TODO documentation
|
||||
* @param k
|
||||
@@ -31,7 +32,7 @@ bool isAlreadyMatched(MilanLongInt node,
|
||||
*/
|
||||
MilanLongInt val;
|
||||
if ((node < StartIndex) || (node > EndIndex)) // if ghost vertex
|
||||
{
|
||||
{
|
||||
#pragma omp atomic read
|
||||
val = GMate[Ghost2LocalMap[node]];
|
||||
return val >= 0; // Already matched
|
||||
@@ -42,4 +43,4 @@ bool isAlreadyMatched(MilanLongInt node,
|
||||
val = Mate[node - StartIndex];
|
||||
|
||||
return val >= 0; // Already matched
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_BD(MilanLongInt NLVer,
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_B(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
@@ -20,37 +21,9 @@ void PARALLEL_COMPUTE_CANDIDATE_MATE_BD(MilanLongInt NLVer,
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Start: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
candidateMate[v] = firstComputeCandidateMateD(verLocPtr[v], verLocPtr[v + 1],
|
||||
verLocInd, edgeLocWeight);
|
||||
candidateMate[v] = firstComputeCandidateMate(verLocPtr[v], verLocPtr[v + 1], verLocInd, edgeLocWeight);
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void PARALLEL_COMPUTE_CANDIDATE_MATE_BS(MilanLongInt NLVer,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt myRank,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *candidateMate)
|
||||
{
|
||||
|
||||
MilanLongInt v = -1;
|
||||
|
||||
#pragma omp parallel private(v) default(shared) num_threads(NUM_THREAD)
|
||||
{
|
||||
|
||||
#pragma omp for schedule(static)
|
||||
for (v = 0; v < NLVer; v++) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing: " << v + StartIndex << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Start: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
candidateMate[v] = firstComputeCandidateMateS(verLocPtr[v], verLocPtr[v + 1],
|
||||
verLocInd, edgeLocWeight);
|
||||
// End: PARALLEL_COMPUTE_CANDIDATE_MATE_B(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
MilanLongInt *S)
|
||||
{
|
||||
@@ -20,4 +21,4 @@ void PROCESS_CROSS_EDGE(MilanLongInt *edge,
|
||||
#endif
|
||||
|
||||
// End: PARALLEL_PROCESS_CROSS_EDGE_B
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,7 @@
|
||||
#include "MatchBoxPC.h"
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_BD(MilanLongInt NLVer,
|
||||
#include "MatchBoxPC.h"
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_B(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
@@ -28,18 +30,17 @@ void PARALLEL_PROCESS_EXPOSED_VERTEX_BD(MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
{
|
||||
|
||||
MilanLongInt v = -1, k = -1, w = -1, adj11 = 0, adj12 = 0, k1 = 0;
|
||||
MilanLongInt v = -1, k = -1, w = -1, adj11 = 0, adj12 = 0, k1 = 0;
|
||||
MilanInt ghostOwner = 0, option, igw;
|
||||
|
||||
//#pragma omp parallel private(option, k, w, v, k1, adj11, adj12, ghostOwner) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, \
|
||||
privateQGhostVtx, privateQMsgType, privateQOwner) \
|
||||
default(shared) num_threads(NUM_THREAD)
|
||||
#pragma omp parallel private(option, k, w, v, k1, adj11, adj12, ghostOwner) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx, privateQMsgType, privateQOwner) \
|
||||
default(shared) num_threads(NUM_THREAD)
|
||||
|
||||
{
|
||||
//#pragma omp for reduction(+ \
|
||||
#pragma omp for reduction(+ \
|
||||
: PCounter[:numProcs], myCard \
|
||||
[:1], msgInd \
|
||||
[:1], NumMessagesBundled \
|
||||
@@ -62,89 +63,102 @@ void PARALLEL_PROCESS_EXPOSED_VERTEX_BD(MilanLongInt NLVer,
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
#pragma omp critical(Matching)
|
||||
{
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap)) {
|
||||
w = computeCandidateMateD(verLocPtr[v], verLocPtr[v + 1], edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v] = w;
|
||||
}
|
||||
}
|
||||
if (w >= 0) {
|
||||
(*myCard)++;
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // w is a ghost vertex
|
||||
option = 2;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v + StartIndex) {
|
||||
option = 1;
|
||||
Mate[v] = w;
|
||||
GMate[Ghost2LocalMap[w]] = v + StartIndex; // w is a Ghost
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == (v + StartIndex)) {
|
||||
option = 3;
|
||||
Mate[v] = w; // v is local
|
||||
Mate[w - StartIndex] = v + StartIndex; // w is local
|
||||
if (w >= 0)
|
||||
{
|
||||
|
||||
#pragma omp critical(processExposed)
|
||||
{
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap)) {
|
||||
w = computeCandidateMate(verLocPtr[v],
|
||||
verLocPtr[v + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
candidateMate[v] = w;
|
||||
}
|
||||
|
||||
if (w >= 0) {
|
||||
(*myCard)++;
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // w is a ghost vertex
|
||||
option = 2;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v + StartIndex) {
|
||||
option = 1;
|
||||
Mate[v] = w;
|
||||
GMate[Ghost2LocalMap[w]] = v + StartIndex; // w is a Ghost
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
|
||||
if (candidateMate[w - StartIndex] == (v + StartIndex)) {
|
||||
option = 3;
|
||||
Mate[v] = w; // v is local
|
||||
Mate[w - StartIndex] = v + StartIndex; // w is local
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if ( candidateMate[w-StartIndex] == (v+StartIndex) )
|
||||
} // End of Else
|
||||
} // End of second if
|
||||
|
||||
} // End of if ( candidateMate[w-StartIndex] == (v+StartIndex) )
|
||||
} // End of Else
|
||||
|
||||
} // End of second if
|
||||
|
||||
} // End critical processExposed
|
||||
|
||||
} // End of if(w >=0)
|
||||
else {
|
||||
//#pragma omp critical(adjuse)
|
||||
{
|
||||
// This piece of code is executed a really small number of times
|
||||
adj11 = verLocPtr[v];
|
||||
adj12 = verLocPtr[v + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
// This piece of code is executed a really small amount of times
|
||||
adj11 = verLocPtr[v];
|
||||
adj12 = verLocPtr[v + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
}
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
}
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
{
|
||||
case -1:
|
||||
break;
|
||||
case 1:
|
||||
case 1:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ")";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
case 2:
|
||||
case 2:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message (291):";
|
||||
cout << "\n(" << myRank << ")Local is: " << v + StartIndex << " Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
@@ -155,215 +169,29 @@ void PARALLEL_PROCESS_EXPOSED_VERTEX_BD(MilanLongInt NLVer,
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
default:
|
||||
case 3:
|
||||
default:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
break;
|
||||
}
|
||||
|
||||
} // End of for ( v=0; v < NLVer; v++ )
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
} // End of parallel region
|
||||
}
|
||||
|
||||
|
||||
void PARALLEL_PROCESS_EXPOSED_VERTEX_BS(MilanLongInt NLVer,
|
||||
MilanLongInt *candidateMate,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *Mate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *S,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
MilanLongInt v = -1, k = -1, w = -1, adj11 = 0, adj12 = 0, k1 = 0;
|
||||
MilanInt ghostOwner = 0, option, igw;
|
||||
|
||||
//#pragma omp parallel private(option, k, w, v, k1, adj11, adj12, ghostOwner) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, \
|
||||
privateQGhostVtx, privateQMsgType, privateQOwner) \
|
||||
default(shared) num_threads(NUM_THREAD)
|
||||
|
||||
{
|
||||
//#pragma omp for reduction(+ \
|
||||
: PCounter[:numProcs], myCard \
|
||||
[:1], msgInd \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1]) \
|
||||
schedule(static)
|
||||
for (v = 0; v < NLVer; v++) {
|
||||
option = -1;
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
k = candidateMate[v];
|
||||
candidateMate[v] = verLocInd[k];
|
||||
w = candidateMate[v];
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing: " << v + StartIndex << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v + StartIndex << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
#pragma omp critical(Matching)
|
||||
{
|
||||
if (isAlreadyMatched(verLocInd[k], StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap)) {
|
||||
w = computeCandidateMateS(verLocPtr[v], verLocPtr[v + 1], edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v] = w;
|
||||
}
|
||||
if (w >= 0) {
|
||||
(*myCard)++;
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // w is a ghost vertex
|
||||
option = 2;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v + StartIndex) {
|
||||
option = 1;
|
||||
Mate[v] = w;
|
||||
GMate[Ghost2LocalMap[w]] = v + StartIndex; // w is a Ghost
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == (v + StartIndex)) {
|
||||
option = 3;
|
||||
Mate[v] = w; // v is local
|
||||
Mate[w - StartIndex] = v + StartIndex; // w is local
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if ( candidateMate[w-StartIndex] == (v+StartIndex) )
|
||||
} // End of Else
|
||||
} // End of second if
|
||||
}
|
||||
|
||||
} // End of if(w >=0)
|
||||
else {
|
||||
//#pragma omp critical(adjuse)
|
||||
{
|
||||
// This piece of code is executed a really small number of times
|
||||
adj11 = verLocPtr[v];
|
||||
adj12 = verLocPtr[v + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
}
|
||||
}
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
break;
|
||||
case 1:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v + StartIndex << "," << w << ")";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
case 2:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message (291):";
|
||||
cout << "\n(" << myRank << ")Local is: " << v + StartIndex << " Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
(*msgInd)++;
|
||||
(*NumMessagesBundled)++;
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
|
||||
privateQLocalVtx.push_back(v + StartIndex);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
default:
|
||||
privateU.push_back(v + StartIndex);
|
||||
privateU.push_back(w);
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
} // End of for ( v=0; v < NLVer; v++ )
|
||||
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
|
||||
} // End of parallel region
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
#include "MatchBoxPC.h"
|
||||
void processMatchedVerticesD(
|
||||
|
||||
#if !defined(SERIAL_MPI)
|
||||
void processMatchedVertices(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
@@ -57,29 +59,29 @@ void processMatchedVerticesD(
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
|
||||
|
||||
extractUChunk(UChunkBeingProcessed, U, privateU);
|
||||
|
||||
|
||||
for (MilanLongInt u : UChunkBeingProcessed) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")u: " << u;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((u >= StartIndex) && (u <= EndIndex)) { // Process Only the Local Vertices
|
||||
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
|
||||
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
option = -1;
|
||||
v = verLocInd[k];
|
||||
|
||||
|
||||
if ((v >= StartIndex) && (v <= EndIndex)) { // If Local Vertex:
|
||||
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")v: " << v << " c(v)= " << candidateMate[v - StartIndex] << " Mate[v]: " << Mate[v];
|
||||
fflush(stdout);
|
||||
@@ -90,62 +92,62 @@ void processMatchedVerticesD(
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateD(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
|
||||
candidateMate[v - StartIndex] = w;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
}
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
else { // Neighbor is a ghost vertex
|
||||
|
||||
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[v]] == u)
|
||||
@@ -154,7 +156,7 @@ void processMatchedVerticesD(
|
||||
option = 5; // u is local
|
||||
} // End of critical
|
||||
} // End of Else //A Ghost Vertex
|
||||
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
@@ -164,7 +166,7 @@ void processMatchedVerticesD(
|
||||
// Found a dominating edge, it is a ghost and candidateMate[NLVer + Ghost2LocalMap[w]] == v
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
|
||||
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
@@ -173,16 +175,15 @@ void processMatchedVerticesD(
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], SPtr);
|
||||
case 2:
|
||||
|
||||
|
||||
// Found a dominating edge, it is a ghost
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
@@ -191,7 +192,7 @@ void processMatchedVerticesD(
|
||||
case 3:
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
|
||||
|
||||
(*myCard)++;
|
||||
break;
|
||||
case 4:
|
||||
@@ -201,385 +202,95 @@ void processMatchedVerticesD(
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
break;
|
||||
case 5:
|
||||
default:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a success message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << v << " Owner is: " << findOwnerOfGhost(v, verDistance, myRank, numProcs) << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
|
||||
(*NumMessagesBundled)++;
|
||||
PCounter[ghostOwner]++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
#pragma omp critical(U)
|
||||
{
|
||||
U.insert(U.end(), privateU.begin(), privateU.end());
|
||||
}
|
||||
|
||||
#pragma omp critical(sendMessageTransfer)
|
||||
{
|
||||
QLocalVtx.insert(QLocalVtx.end(), privateQLocalVtx.begin(), privateQLocalVtx.end());
|
||||
QGhostVtx.insert(QGhostVtx.end(), privateQGhostVtx.begin(), privateQGhostVtx.end());
|
||||
QMsgType.insert(QMsgType.end(), privateQMsgType.begin(), privateQMsgType.end());
|
||||
QOwner.insert(QOwner.end(), privateQOwner.begin(), privateQOwner.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
privateQLocalVtx.clear();
|
||||
privateQGhostVtx.clear();
|
||||
privateQMsgType.clear();
|
||||
privateQOwner.clear();
|
||||
|
||||
} // End of while ( !U.empty() )
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices,
|
||||
omp_get_thread_num(),
|
||||
myRank);
|
||||
|
||||
#endif
|
||||
} // End of parallel region
|
||||
}
|
||||
|
||||
|
||||
|
||||
void processMatchedVerticesS(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner)
|
||||
{
|
||||
|
||||
MilanLongInt adj1, adj2, adj11, adj12, k, k1, v = -1, w = -1, ghostOwner;
|
||||
int option;
|
||||
MilanLongInt mateVal;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
MilanLongInt localVertices = 0;
|
||||
#endif
|
||||
//#pragma omp parallel private(k, w, v, k1, adj1, adj2, adj11, adj12, ghostOwner, option) \
|
||||
firstprivate(privateU, StartIndex, EndIndex, privateQLocalVtx, privateQGhostVtx, \
|
||||
privateQMsgType, privateQOwner, UChunkBeingProcessed) \
|
||||
default(shared) num_threads(NUM_THREAD) \
|
||||
reduction(+ \
|
||||
: msgInd[:1], PCounter \
|
||||
[:numProcs], myCard \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1])
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
|
||||
extractUChunk(UChunkBeingProcessed, U, privateU);
|
||||
|
||||
for (MilanLongInt u : UChunkBeingProcessed) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")u: " << u;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((u >= StartIndex) && (u <= EndIndex)) { // Process Only the Local Vertices
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
option = -1;
|
||||
v = verLocInd[k];
|
||||
|
||||
if ((v >= StartIndex) && (v <= EndIndex)) { // If Local Vertex:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")v: " << v << " c(v)= " << candidateMate[v - StartIndex] << " Mate[v]: " << Mate[v];
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateS(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
}
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
else { // Neighbor is a ghost vertex
|
||||
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[v]] == u)
|
||||
candidateMate[NLVer + Ghost2LocalMap[v]] = -1;
|
||||
if (v != Mate[u - StartIndex])
|
||||
option = 5; // u is local
|
||||
} // End of critical
|
||||
} // End of Else //A Ghost Vertex
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
// No things to do
|
||||
break;
|
||||
case 1:
|
||||
// Found a dominating edge, it is a ghost and candidateMate[NLVer + Ghost2LocalMap[w]] == v
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], SPtr);
|
||||
case 2:
|
||||
|
||||
// Found a dominating edge, it is a ghost
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
|
||||
(*myCard)++;
|
||||
break;
|
||||
case 4:
|
||||
// Could not find a dominating vertex
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
#pragma omp atomic
|
||||
|
||||
PCounter[ghostOwner]++;
|
||||
(*NumMessagesBundled)++;
|
||||
(*msgInd)++;
|
||||
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
break;
|
||||
case 5:
|
||||
default:
|
||||
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a success message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << v << " Owner is: " << findOwnerOfGhost(v, verDistance, myRank, numProcs) << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
// assert(ghostOwner != -1);
|
||||
// assert(ghostOwner != myRank);
|
||||
|
||||
|
||||
(*NumMessagesBundled)++;
|
||||
PCounter[ghostOwner]++;
|
||||
(*msgInd)++;
|
||||
|
||||
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
|
||||
#pragma omp critical(U)
|
||||
{
|
||||
U.insert(U.end(), privateU.begin(), privateU.end());
|
||||
}
|
||||
|
||||
|
||||
privateU.clear();
|
||||
|
||||
#pragma omp critical(sendMessageTransfer)
|
||||
{
|
||||
|
||||
QLocalVtx.insert(QLocalVtx.end(), privateQLocalVtx.begin(), privateQLocalVtx.end());
|
||||
QGhostVtx.insert(QGhostVtx.end(), privateQGhostVtx.begin(), privateQGhostVtx.end());
|
||||
QMsgType.insert(QMsgType.end(), privateQMsgType.begin(), privateQMsgType.end());
|
||||
QOwner.insert(QOwner.end(), privateQOwner.begin(), privateQOwner.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
|
||||
privateQLocalVtx.clear();
|
||||
privateQGhostVtx.clear();
|
||||
privateQMsgType.clear();
|
||||
privateQOwner.clear();
|
||||
|
||||
|
||||
} // End of while ( !U.empty() )
|
||||
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices,
|
||||
omp_get_thread_num(),
|
||||
myRank);
|
||||
|
||||
|
||||
#endif
|
||||
} // End of parallel region
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "MatchBoxPC.h"
|
||||
//#define DEBUG_HANG_
|
||||
void processMatchedVerticesAndSendMessagesD(
|
||||
#if !defined(SERIAL_MPI)
|
||||
void processMatchedVerticesAndSendMessages(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
@@ -26,302 +27,6 @@ void processMatchedVerticesAndSendMessagesD(
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
vector<MilanLongInt> &privateQMsgType,
|
||||
vector<MilanInt> &privateQOwner,
|
||||
MPI_Comm comm,
|
||||
MilanLongInt *msgActual,
|
||||
vector<MilanLongInt> &Message)
|
||||
{
|
||||
|
||||
MilanLongInt initialSize = QLocalVtx.size();
|
||||
MilanLongInt adj1, adj2, adj11, adj12, k, k1, v = -1, w = -1, ghostOwner;
|
||||
int option;
|
||||
MilanLongInt mateVal;
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
MilanLongInt localVertices = 0;
|
||||
#endif
|
||||
//#pragma omp parallel private(k, w, v, k1, adj1, adj2, adj11, adj12, ghostOwner, option) \
|
||||
firstprivate(Message, privateU, StartIndex, EndIndex, privateQLocalVtx, \
|
||||
privateQGhostVtx, privateQMsgType, privateQOwner, UChunkBeingProcessed) \
|
||||
default(shared) \
|
||||
num_threads(NUM_THREAD) \
|
||||
reduction(+ \
|
||||
: msgInd[:1], PCounter \
|
||||
[:numProcs], myCard \
|
||||
[:1], NumMessagesBundled \
|
||||
[:1], msgActual \
|
||||
[:1])
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
|
||||
extractUChunk(UChunkBeingProcessed, U, privateU);
|
||||
|
||||
for (MilanLongInt u : UChunkBeingProcessed) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")u: " << u;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((u >= StartIndex) && (u <= EndIndex)) { // Process Only the Local Vertices
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
option = -1;
|
||||
v = verLocInd[k];
|
||||
|
||||
if ((v >= StartIndex) && (v <= EndIndex)) { // If Local Vertex:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")v: " << v << " c(v)= " << candidateMate[v - StartIndex] << " Mate[v]: " << Mate[v];
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateD(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
}
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
else { // Neighbor is a ghost vertex
|
||||
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[v]] == u)
|
||||
candidateMate[NLVer + Ghost2LocalMap[v]] = -1;
|
||||
if (v != Mate[u - StartIndex])
|
||||
option = 5; // u is local
|
||||
} // End of critical
|
||||
} // End of Else //A Ghost Vertex
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
// No things to do
|
||||
break;
|
||||
case 1:
|
||||
// Found a dominating edge, it is a ghost and candidateMate[NLVer + Ghost2LocalMap[w]] == v
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], SPtr);
|
||||
case 2:
|
||||
|
||||
// Found a dominating edge, it is a ghost
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
// Message[2] = REQUEST; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
break;
|
||||
case 3:
|
||||
privateU.push_back(v);
|
||||
privateU.push_back(w);
|
||||
(*myCard)++;
|
||||
break;
|
||||
case 4:
|
||||
// Could not find a dominating vertex
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
// Message[2] = FAILURE; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
break;
|
||||
case 5:
|
||||
default:
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a success message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << v << " Owner is: " << findOwnerOfGhost(v, verDistance, myRank, numProcs) << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
// Build the Message Packet:
|
||||
// Message[0] = u; // LOCAL
|
||||
// Message[1] = v; // GHOST
|
||||
// Message[2] = SUCCESS; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
} // End of while ( !U.empty() )
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices, mp_get_thread_num(), myRank);
|
||||
#endif
|
||||
} // End of parallel region
|
||||
|
||||
// Send the messages
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank<<" Sending: "<<QOwner.size()-initialSize<<" messages" <<endl;
|
||||
#endif
|
||||
for (int i = initialSize; i < QOwner.size(); i++) {
|
||||
Message[0] = QLocalVtx[i];
|
||||
Message[1] = QGhostVtx[i];
|
||||
Message[2] = QMsgType[i];
|
||||
ghostOwner = QOwner[i];
|
||||
//MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
}
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank<<" Done sending messages"<<endl;
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
|
||||
void processMatchedVerticesAndSendMessagesS(
|
||||
MilanLongInt NLVer,
|
||||
vector<MilanLongInt> &UChunkBeingProcessed,
|
||||
vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *NumMessagesBundled,
|
||||
MilanLongInt *SPtr,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *PCounter,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanInt myRank,
|
||||
MilanInt numProcs,
|
||||
MilanLongInt *candidateMate,
|
||||
vector<MilanLongInt> &GMate,
|
||||
MilanLongInt *Mate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
MilanFloat *edgeLocWeight,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MilanLongInt> &privateQLocalVtx,
|
||||
vector<MilanLongInt> &privateQGhostVtx,
|
||||
@@ -359,28 +64,29 @@ void processMatchedVerticesAndSendMessagesS(
|
||||
{
|
||||
|
||||
while (!U.empty()) {
|
||||
|
||||
|
||||
extractUChunk(UChunkBeingProcessed, U, privateU);
|
||||
|
||||
|
||||
for (MilanLongInt u : UChunkBeingProcessed) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")u: " << u;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
if ((u >= StartIndex) && (u <= EndIndex)) { // Process Only the Local Vertices
|
||||
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
localVertices++;
|
||||
#endif
|
||||
|
||||
// Get the Adjacency list for u
|
||||
adj1 = verLocPtr[u - StartIndex]; // Pointer
|
||||
adj2 = verLocPtr[u - StartIndex + 1];
|
||||
for (k = adj1; k < adj2; k++) {
|
||||
option = -1;
|
||||
v = verLocInd[k];
|
||||
|
||||
|
||||
if ((v >= StartIndex) && (v <= EndIndex)) { // If Local Vertex:
|
||||
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")v: " << v << " c(v)= " << candidateMate[v - StartIndex] << " Mate[v]: " << Mate[v];
|
||||
fflush(stdout);
|
||||
@@ -391,62 +97,63 @@ void processMatchedVerticesAndSendMessagesS(
|
||||
if (mateVal < 0) {
|
||||
#pragma omp critical
|
||||
{
|
||||
#pragma omp atomic read
|
||||
mateVal = Mate[v - StartIndex];
|
||||
// If the current vertex is pointing to a matched vertex and is not matched
|
||||
if (mateVal < 0) {
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd,
|
||||
StartIndex,
|
||||
EndIndex,
|
||||
GMate,
|
||||
Mate,
|
||||
Ghost2LocalMap);
|
||||
|
||||
candidateMate[v - StartIndex] = w;
|
||||
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateS(verLocPtr[v - StartIndex],
|
||||
verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, 0,
|
||||
verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message:";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
#endif
|
||||
option = 2;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
option = 1;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is a ghost vertex
|
||||
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
option = 3;
|
||||
Mate[v - StartIndex] = w; // v is a local vertex
|
||||
Mate[w - StartIndex] = v; // w is a local vertex
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") ";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
}
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else
|
||||
option = 4; // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of If (candidateMate[v-StartIndex] == u
|
||||
} // End of task
|
||||
} // mateval < 0
|
||||
} // End of if ( (v >= StartIndex) && (v <= EndIndex) ) //If Local Vertex:
|
||||
else { // Neighbor is a ghost vertex
|
||||
|
||||
|
||||
#pragma omp critical
|
||||
{
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[v]] == u)
|
||||
@@ -455,7 +162,7 @@ void processMatchedVerticesAndSendMessagesS(
|
||||
option = 5; // u is local
|
||||
} // End of critical
|
||||
} // End of Else //A Ghost Vertex
|
||||
|
||||
|
||||
switch (option)
|
||||
{
|
||||
case -1:
|
||||
@@ -473,20 +180,20 @@ void processMatchedVerticesAndSendMessagesS(
|
||||
// Decrement the counter:
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], SPtr);
|
||||
case 2:
|
||||
|
||||
|
||||
// Found a dominating edge, it is a ghost
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
// Message[2] = REQUEST; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(REQUEST);
|
||||
@@ -504,82 +211,94 @@ void processMatchedVerticesAndSendMessagesS(
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) { // A ghost
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = v; // LOCAL
|
||||
// Message[1] = w; // GHOST
|
||||
// Message[2] = FAILURE; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(v);
|
||||
privateQGhostVtx.push_back(w);
|
||||
privateQMsgType.push_back(FAILURE);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
break;
|
||||
case 5:
|
||||
default:
|
||||
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a success message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << v << " Owner is: " << findOwnerOfGhost(v, verDistance, myRank, numProcs) << "\n";
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
ghostOwner = findOwnerOfGhost(v, verDistance, myRank, numProcs);
|
||||
|
||||
// Build the Message Packet:
|
||||
// Message[0] = u; // LOCAL
|
||||
// Message[1] = v; // GHOST
|
||||
// Message[2] = SUCCESS; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
// MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
|
||||
(*msgActual)++;
|
||||
(*msgInd)++;
|
||||
|
||||
privateQLocalVtx.push_back(u);
|
||||
privateQGhostVtx.push_back(v);
|
||||
privateQMsgType.push_back(SUCCESS);
|
||||
privateQOwner.push_back(ghostOwner);
|
||||
|
||||
|
||||
break;
|
||||
} // End of switch
|
||||
} // End of inner for
|
||||
}
|
||||
} // End of outer for
|
||||
|
||||
|
||||
queuesTransfer(U, privateU, QLocalVtx,
|
||||
QGhostVtx,
|
||||
QMsgType, QOwner, privateQLocalVtx,
|
||||
privateQGhostVtx,
|
||||
privateQMsgType,
|
||||
privateQOwner);
|
||||
|
||||
|
||||
} // End of while ( !U.empty() )
|
||||
|
||||
|
||||
#ifdef COUNT_LOCAL_VERTEX
|
||||
printf("Count local vertexes: %ld for thread %d of processor %d\n",
|
||||
localVertices, mp_get_thread_num(), myRank);
|
||||
localVertices,
|
||||
omp_get_thread_num(),
|
||||
myRank);
|
||||
|
||||
#endif
|
||||
} // End of parallel region
|
||||
|
||||
|
||||
// Send the messages
|
||||
#ifdef DEBUG_HANG_
|
||||
cout << myRank<<" Sending: "<<QOwner.size()-initialSize<<" messages" <<endl;
|
||||
#endif
|
||||
for (int i = initialSize; i < QOwner.size(); i++) {
|
||||
|
||||
Message[0] = QLocalVtx[i];
|
||||
Message[1] = QGhostVtx[i];
|
||||
Message[2] = QMsgType[i];
|
||||
ghostOwner = QOwner[i];
|
||||
|
||||
//MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
@@ -588,4 +307,4 @@ void processMatchedVerticesAndSendMessagesS(
|
||||
cout << myRank<<" Done sending messages"<<endl;
|
||||
#endif
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
//#define DEBUG_HANG_
|
||||
#if !defined(SERIAL_MPI)
|
||||
|
||||
void processMessagesD(
|
||||
void processMessages(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
@@ -139,12 +139,12 @@ void processMessagesD(
|
||||
if (!ReceiveBuffer.empty())
|
||||
ReceiveBuffer.clear(); // Empty it out first
|
||||
ReceiveBuffer.resize(bundleSize, -1); // Initialize
|
||||
|
||||
|
||||
ReceiveBuffer[0] = Message[0]; // u
|
||||
ReceiveBuffer[1] = Message[1]; // v
|
||||
ReceiveBuffer[2] = Message[2]; // message_type
|
||||
}
|
||||
|
||||
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < StartIndex) || (v > EndIndex)) {
|
||||
cout << "\n(" << myRank << ") From ReceiveBuffer: This should not happen: u= " << u << " v= " << v << " Type= " << message_type << " StartIndex " << StartIndex << " EndIndex " << EndIndex << endl;
|
||||
@@ -161,7 +161,7 @@ void processMessagesD(
|
||||
u = ReceiveBuffer[bundleCounter - 3]; // GHOST
|
||||
v = ReceiveBuffer[bundleCounter - 2]; // LOCAL
|
||||
message_type = ReceiveBuffer[bundleCounter - 1]; // TYPE
|
||||
|
||||
|
||||
// CASE I: REQUEST
|
||||
if (message_type == REQUEST) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
@@ -189,7 +189,7 @@ void processMessagesD(
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << u << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S);
|
||||
} // End of if ( candidateMate[v-StartIndex] == u )e
|
||||
} // End of if ( Mate[v] == -1 )
|
||||
@@ -212,9 +212,8 @@ void processMessagesD(
|
||||
// Process only if not already matched ( v is local)
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateD(verLocPtr[v - StartIndex], verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, k,verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
w = computeCandidateMate(verLocPtr[v - StartIndex], verLocPtr[v - StartIndex + 1], edgeLocWeight, k,
|
||||
verLocInd, StartIndex, EndIndex, GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w << endl;
|
||||
@@ -251,7 +250,7 @@ void processMessagesD(
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
@@ -312,322 +311,7 @@ void processMessagesD(
|
||||
} // End of else: CASE III
|
||||
} // End of else: CASE I
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
void processMessagesS(
|
||||
MilanLongInt NLVer,
|
||||
MilanLongInt *Mate,
|
||||
MilanLongInt *candidateMate,
|
||||
map<MilanLongInt, MilanLongInt> &Ghost2LocalMap,
|
||||
vector<MilanLongInt> &GMate,
|
||||
vector<MilanLongInt> &Counter,
|
||||
MilanLongInt StartIndex,
|
||||
MilanLongInt EndIndex,
|
||||
MilanLongInt *myCard,
|
||||
MilanLongInt *msgInd,
|
||||
MilanLongInt *msgActual,
|
||||
MilanFloat *edgeLocWeight,
|
||||
MilanLongInt *verDistance,
|
||||
MilanLongInt *verLocPtr,
|
||||
MilanLongInt k,
|
||||
MilanLongInt *verLocInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &Message,
|
||||
MilanLongInt numGhostEdges,
|
||||
MilanLongInt u,
|
||||
MilanLongInt v,
|
||||
MilanLongInt *S,
|
||||
vector<MilanLongInt> &U)
|
||||
{
|
||||
|
||||
//#define PRINT_DEBUG_INFO_
|
||||
|
||||
MilanInt Sender;
|
||||
MPI_Status computeStatus;
|
||||
MilanLongInt bundleSize, w;
|
||||
MilanLongInt adj11, adj12, k1;
|
||||
MilanLongInt ghostOwner;
|
||||
int error_codeC;
|
||||
error_codeC = MPI_Comm_set_errhandler(MPI_COMM_WORLD, MPI_ERRORS_RETURN);
|
||||
char error_message[MPI_MAX_ERROR_STRING];
|
||||
int message_length;
|
||||
MilanLongInt message_type = 0;
|
||||
|
||||
// Buffer to receive bundled messages
|
||||
// Maximum messages that can be received from any processor is
|
||||
// twice the edge cut: REQUEST; REQUEST+(FAILURE/SUCCESS)
|
||||
vector<MilanLongInt> ReceiveBuffer;
|
||||
try
|
||||
{
|
||||
ReceiveBuffer.reserve(numGhostEdges * 2 * 3); // Three integers per cross edge
|
||||
}
|
||||
catch (length_error)
|
||||
{
|
||||
cout << "Error in function algoDistEdgeApproxDominatingEdgesMessageBundling: \n";
|
||||
cout << "Not enough memory to allocate the internal variables \n";
|
||||
exit(1);
|
||||
}
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout
|
||||
<< "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")About to begin Message processing phase ... *S=" << *S << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << "=========================************===============================" << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// BLOCKING RECEIVE:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << " Waiting for blocking receive..." << endl;
|
||||
fflush(stdout);
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
//cout << myRank<<" Receiving ...";
|
||||
error_codeC = MPI_Recv(&Message[0], 3, TypeMap<MilanLongInt>(), MPI_ANY_SOURCE, ComputeTag, comm, &computeStatus);
|
||||
if (error_codeC != MPI_SUCCESS)
|
||||
{
|
||||
MPI_Error_string(error_codeC, error_message, &message_length);
|
||||
cout << "\n*Error in call to MPI_Receive on Slave: " << error_message << "\n";
|
||||
fflush(stdout);
|
||||
}
|
||||
Sender = computeStatus.MPI_SOURCE;
|
||||
//cout << " ...from "<<Sender << endl;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received message from Process " << Sender << " Type= " << Message[2] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
if (Message[2] == SIZEINFO) {
|
||||
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received bundled message from Process " << Sender << " Size= " << Message[0] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
bundleSize = Message[0]; //#of integers in the message
|
||||
// Build the Message Buffer:
|
||||
if (!ReceiveBuffer.empty())
|
||||
ReceiveBuffer.clear(); // Empty it out first
|
||||
ReceiveBuffer.resize(bundleSize, -1); // Initialize
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message Bundle Before: " << endl;
|
||||
for (int i = 0; i < bundleSize; i++)
|
||||
cout << ReceiveBuffer[i] << ",";
|
||||
cout << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Receive the message
|
||||
//cout << myRank<<" Receiving from "<<Sender<<endl;
|
||||
error_codeC = MPI_Recv(&ReceiveBuffer[0], bundleSize, TypeMap<MilanLongInt>(), Sender, BundleTag, comm, &computeStatus);
|
||||
if (error_codeC != MPI_SUCCESS) {
|
||||
MPI_Error_string(error_codeC, error_message, &message_length);
|
||||
cout << "\n*Error in call to MPI_Receive on processor " << myRank << " Error: " << error_message << "\n";
|
||||
fflush(stdout);
|
||||
}
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message Bundle After: " << endl;
|
||||
for (int i = 0; i < bundleSize; i++)
|
||||
cout << ReceiveBuffer[i] << ",";
|
||||
cout << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} else { // Just a single message:
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Received regular message from Process " << Sender << " u= " << Message[0] << " v= " << Message[1] << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// Add the current message to Queue:
|
||||
bundleSize = 3; //#of integers in the message
|
||||
// Build the Message Buffer:
|
||||
if (!ReceiveBuffer.empty())
|
||||
ReceiveBuffer.clear(); // Empty it out first
|
||||
ReceiveBuffer.resize(bundleSize, -1); // Initialize
|
||||
|
||||
ReceiveBuffer[0] = Message[0]; // u
|
||||
ReceiveBuffer[1] = Message[1]; // v
|
||||
ReceiveBuffer[2] = Message[2]; // message_type
|
||||
}
|
||||
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < StartIndex) || (v > EndIndex)) {
|
||||
cout << "\n(" << myRank << ") From ReceiveBuffer: This should not happen: u= " << u << " v= " << v << " Type= " << message_type << " StartIndex " << StartIndex << " EndIndex " << EndIndex << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
#endif
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Processing message: u= " << u << " v= " << v << " Type= " << message_type << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
// Most of the time bundleSize == 3, thus, it's not worth parallelizing thi loop
|
||||
for (MilanLongInt bundleCounter = 3; bundleCounter < bundleSize + 3; bundleCounter += 3) {
|
||||
u = ReceiveBuffer[bundleCounter - 3]; // GHOST
|
||||
v = ReceiveBuffer[bundleCounter - 2]; // LOCAL
|
||||
message_type = ReceiveBuffer[bundleCounter - 1]; // TYPE
|
||||
|
||||
// CASE I: REQUEST
|
||||
if (message_type == REQUEST) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is REQUEST" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < 0) || (v < StartIndex) || ((v - StartIndex) > NLVer)) {
|
||||
cout << "\n(" << myRank << ") case 1 Bad address " << v << " " << StartIndex << " " << v - StartIndex << " " << NLVer << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
if (Mate[v - StartIndex] == -1) {
|
||||
// Process only if not already matched (v is local)
|
||||
candidateMate[NLVer + Ghost2LocalMap[u]] = v; // Set CandidateMate for the ghost
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
GMate[Ghost2LocalMap[u]] = v; // u is ghost
|
||||
Mate[v - StartIndex] = u; // v is local
|
||||
U.push_back(v);
|
||||
U.push_back(u);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << u << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S);
|
||||
} // End of if ( candidateMate[v-StartIndex] == u )e
|
||||
} // End of if ( Mate[v] == -1 )
|
||||
} // End of REQUEST
|
||||
else { // CASE II: SUCCESS
|
||||
if (message_type == SUCCESS) {
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is SUCCESS" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
GMate[Ghost2LocalMap[u]] = EndIndex + 1; // Set a Dummy Mate to make sure that we do not (u is a ghost) process it again
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S);
|
||||
#ifdef DEBUG_GHOST_
|
||||
if ((v < 0) || (v < StartIndex) || ((v - StartIndex) > NLVer)) {
|
||||
cout << "\n(" << myRank << ") case 2 Bad address " << v << " " << StartIndex << " " << v - StartIndex << " " << NLVer << endl;
|
||||
fflush(stdout);
|
||||
}
|
||||
#endif
|
||||
if (Mate[v - StartIndex] == -1) {
|
||||
// Process only if not already matched ( v is local)
|
||||
if (candidateMate[v - StartIndex] == u) {
|
||||
// Start: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
w = computeCandidateMateS(verLocPtr[v - StartIndex], verLocPtr[v - StartIndex + 1],
|
||||
edgeLocWeight, k,verLocInd, StartIndex, EndIndex,
|
||||
GMate, Mate, Ghost2LocalMap);
|
||||
candidateMate[v - StartIndex] = w;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")" << v << " Points to: " << w << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
// If found a dominating edge:
|
||||
if (w >= 0) {
|
||||
if ((w < StartIndex) || (w > EndIndex)) {
|
||||
// w is a ghost
|
||||
// Build the Message Packet:
|
||||
Message[0] = v; // LOCAL
|
||||
Message[1] = w; // GHOST
|
||||
Message[2] = REQUEST; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a request message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
//assert(ghostOwner != -1);
|
||||
//assert(ghostOwner != myRank);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
(*msgInd)++;
|
||||
(*msgActual)++;
|
||||
if (candidateMate[NLVer + Ghost2LocalMap[w]] == v) {
|
||||
Mate[v - StartIndex] = w; // v is local
|
||||
GMate[Ghost2LocalMap[w]] = v; // w is ghost
|
||||
U.push_back(v);
|
||||
U.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[w]], S);
|
||||
} // End of if CandidateMate[w] = v
|
||||
} // End of if a Ghost Vertex
|
||||
else { // w is a local vertex
|
||||
if (candidateMate[w - StartIndex] == v) {
|
||||
Mate[v - StartIndex] = w; // v is local
|
||||
Mate[w - StartIndex] = v; // w is local
|
||||
// Q.push_back(u);
|
||||
U.push_back(v);
|
||||
U.push_back(w);
|
||||
(*myCard)++;
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")MATCH: (" << v << "," << w << ") " << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
} // End of if(CandidateMate(w) = v
|
||||
} // End of Else
|
||||
} // End of if(w >=0)
|
||||
else { // No dominant edge found
|
||||
adj11 = verLocPtr[v - StartIndex];
|
||||
adj12 = verLocPtr[v - StartIndex + 1];
|
||||
for (k1 = adj11; k1 < adj12; k1++) {
|
||||
w = verLocInd[k1];
|
||||
if ((w < StartIndex) || (w > EndIndex)) {
|
||||
// A ghost
|
||||
// Build the Message Packet:
|
||||
Message[0] = v; // LOCAL
|
||||
Message[1] = w; // GHOST
|
||||
Message[2] = FAILURE; // TYPE
|
||||
// Send a Request (Asynchronous)
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Sending a failure message: ";
|
||||
cout << "\n(" << myRank << ")Ghost is " << w << " Owner is: " << findOwnerOfGhost(w, verDistance, myRank, numProcs) << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
ghostOwner = findOwnerOfGhost(w, verDistance, myRank, numProcs);
|
||||
//assert(ghostOwner != -1);
|
||||
//assert(ghostOwner != myRank);
|
||||
//cout << myRank<<" Sending to "<<ghostOwner<<endl;
|
||||
MPI_Bsend(&Message[0], 3, TypeMap<MilanLongInt>(), ghostOwner, ComputeTag, comm);
|
||||
(*msgInd)++;
|
||||
(*msgActual)++;
|
||||
} // End of if(GHOST)
|
||||
} // End of for loop
|
||||
} // End of Else: w == -1
|
||||
// End: PARALLEL_PROCESS_EXPOSED_VERTEX_B(v)
|
||||
} // End of if ( candidateMate[v-StartIndex] == u )
|
||||
} // End of if ( Mate[v] == -1 )
|
||||
} // End of if ( message_type == SUCCESS )
|
||||
else {
|
||||
// CASE III: FAILURE
|
||||
#ifdef PRINT_DEBUG_INFO_
|
||||
cout << "\n(" << myRank << ")Message type is FAILURE" << endl;
|
||||
fflush(stdout);
|
||||
#endif
|
||||
GMate[Ghost2LocalMap[u]] = EndIndex + 1; // Set a Dummy Mate to make sure that we do not (u is a ghost) process this anymore
|
||||
PROCESS_CROSS_EDGE(&Counter[Ghost2LocalMap[u]], S); // Decrease the counter
|
||||
} // End of else: CASE III
|
||||
} // End of else: CASE I
|
||||
}
|
||||
|
||||
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "MatchBoxPC.h"
|
||||
|
||||
void queuesTransfer(vector<MilanLongInt> &U,
|
||||
vector<MilanLongInt> &privateU,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
@@ -14,20 +15,22 @@ void queuesTransfer(vector<MilanLongInt> &U,
|
||||
#pragma omp critical(U)
|
||||
{
|
||||
U.insert(U.end(), privateU.begin(), privateU.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
|
||||
#pragma omp critical(sendMessageTransfer)
|
||||
{
|
||||
|
||||
//#pragma omp critical(sendMessageTransfer)
|
||||
//{
|
||||
QLocalVtx.insert(QLocalVtx.end(), privateQLocalVtx.begin(), privateQLocalVtx.end());
|
||||
QGhostVtx.insert(QGhostVtx.end(), privateQGhostVtx.begin(), privateQGhostVtx.end());
|
||||
QMsgType.insert(QMsgType.end(), privateQMsgType.begin(), privateQMsgType.end());
|
||||
QOwner.insert(QOwner.end(), privateQOwner.begin(), privateQOwner.end());
|
||||
}
|
||||
|
||||
privateU.clear();
|
||||
privateQLocalVtx.clear();
|
||||
privateQGhostVtx.clear();
|
||||
privateQMsgType.clear();
|
||||
privateQOwner.clear();
|
||||
|
||||
}
|
||||
|
||||
|
||||
@@ -1,23 +1,24 @@
|
||||
#include "MatchBoxPC.h"
|
||||
#if !defined(SERIAL_MPI)
|
||||
void sendBundledMessages(MilanLongInt *numGhostEdges,
|
||||
MilanInt *BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActual,
|
||||
MilanLongInt *msgInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus)
|
||||
MilanInt *BufferSize,
|
||||
MilanLongInt *Buffer,
|
||||
vector<MilanLongInt> &PCumulative,
|
||||
vector<MilanLongInt> &PMessageBundle,
|
||||
vector<MilanLongInt> &PSizeInfoMessages,
|
||||
MilanLongInt *PCounter,
|
||||
MilanLongInt NumMessagesBundled,
|
||||
MilanLongInt *msgActual,
|
||||
MilanLongInt *msgInd,
|
||||
MilanInt numProcs,
|
||||
MilanInt myRank,
|
||||
MPI_Comm comm,
|
||||
vector<MilanLongInt> &QLocalVtx,
|
||||
vector<MilanLongInt> &QGhostVtx,
|
||||
vector<MilanLongInt> &QMsgType,
|
||||
vector<MilanInt> &QOwner,
|
||||
vector<MPI_Request> &SRequest,
|
||||
vector<MPI_Status> &SStatus)
|
||||
{
|
||||
|
||||
MilanLongInt myIndex = 0, numMessagesToSend;
|
||||
@@ -61,7 +62,7 @@ void sendBundledMessages(MilanLongInt *numGhostEdges,
|
||||
for (i = 0; i < numProcs; i++)
|
||||
PCumulative[i + 1] = PCumulative[i] + PCounter[i];
|
||||
}
|
||||
|
||||
|
||||
#pragma omp task depend(inout \
|
||||
: PCounter)
|
||||
{
|
||||
@@ -83,7 +84,7 @@ void sendBundledMessages(MilanLongInt *numGhostEdges,
|
||||
PCounter[QOwner[i]]++;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Send the Bundled Messages: Use ISend
|
||||
#pragma omp task depend(out \
|
||||
: SRequest, SStatus)
|
||||
@@ -100,7 +101,7 @@ void sendBundledMessages(MilanLongInt *numGhostEdges,
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Send the Messages
|
||||
#pragma omp task depend(inout \
|
||||
: SRequest, PSizeInfoMessages, PCumulative) depend(out \
|
||||
@@ -206,3 +207,4 @@ void sendBundledMessages(MilanLongInt *numGhostEdges,
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -451,16 +451,11 @@ subroutine amg_c_hierarchy_bld(a,desc_a,prec,info)
|
||||
if (.not.associated(prec%precv(1)%base_desc,desc_a)) then
|
||||
prec%precv(1)%base_desc => prec%precv(1)%desc_ac
|
||||
end if
|
||||
do i=2, iszv
|
||||
do i=2, iszv
|
||||
prec%precv(i)%base_a => prec%precv(i)%ac
|
||||
prec%precv(i)%base_desc => prec%precv(i)%desc_ac
|
||||
! This is needed when the linmap object has been built
|
||||
! reusing the base_desc descriptor through a pointer.
|
||||
! With PSBLAS 4 we will have a better solution
|
||||
if (associated(prec%precv(i)%linmap%p_desc_U)) &
|
||||
& prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
if (associated(prec%precv(i)%linmap%p_desc_V))&
|
||||
& prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
end do
|
||||
end if
|
||||
|
||||
@@ -512,71 +507,6 @@ subroutine amg_c_hierarchy_bld(a,desc_a,prec,info)
|
||||
return
|
||||
|
||||
contains
|
||||
|
||||
#if ( __GNUC__ == 13 && __GNUC_MINOR__ == 3)
|
||||
! gfortran 13.3.0 generates a strange error here with MOLD
|
||||
! moving to SOURCE but only for this version, since it's heavier
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_c_onelev_type), intent(inout) :: level
|
||||
class(amg_c_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(save1)) then
|
||||
call save1%free(info)
|
||||
if (info == 0) deallocate(save1,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
call save2%free(info)
|
||||
if (info == 0) deallocate(save2,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
allocate(save1, source=level%sm,stat=info)
|
||||
if (info == 0) call level%sm%clone_settings(save1,info)
|
||||
if ((info == 0).and.allocated(level%sm2a)) then
|
||||
allocate(save2, source=level%sm2a,stat=info)
|
||||
if (info == 0) call level%sm2a%clone_settings(save2,info)
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine save_smoothers
|
||||
|
||||
subroutine restore_smoothers(level,save1, save2,info)
|
||||
type(amg_c_onelev_type), intent(inout), target :: level
|
||||
class(amg_c_base_smoother_type), allocatable, intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
|
||||
if (allocated(level%sm)) then
|
||||
if (info == 0) call level%sm%free(info)
|
||||
if (info == 0) deallocate(level%sm,stat=info)
|
||||
end if
|
||||
if (allocated(save1)) then
|
||||
if (info == 0) allocate(level%sm,source=save1,stat=info)
|
||||
if (info == 0) call save1%clone_settings(level%sm,info)
|
||||
end if
|
||||
|
||||
if (info /= 0) return
|
||||
|
||||
if (allocated(level%sm2a)) then
|
||||
if (info == 0) call level%sm2a%free(info)
|
||||
if (info == 0) deallocate(level%sm2a,stat=info)
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
if (info == 0) allocate(level%sm2a,source=save2,stat=info)
|
||||
if (info == 0) call save2%clone_settings(level%sm2a,info)
|
||||
if (info == 0) level%sm2 => level%sm2a
|
||||
else
|
||||
if (allocated(level%sm)) level%sm2 => level%sm
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#else
|
||||
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_c_onelev_type), intent(inout) :: level
|
||||
class(amg_c_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
@@ -635,5 +565,5 @@ contains
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
#endif
|
||||
|
||||
end subroutine amg_c_hierarchy_bld
|
||||
@@ -451,16 +451,11 @@ subroutine amg_d_hierarchy_bld(a,desc_a,prec,info)
|
||||
if (.not.associated(prec%precv(1)%base_desc,desc_a)) then
|
||||
prec%precv(1)%base_desc => prec%precv(1)%desc_ac
|
||||
end if
|
||||
do i=2, iszv
|
||||
do i=2, iszv
|
||||
prec%precv(i)%base_a => prec%precv(i)%ac
|
||||
prec%precv(i)%base_desc => prec%precv(i)%desc_ac
|
||||
! This is needed when the linmap object has been built
|
||||
! reusing the base_desc descriptor through a pointer.
|
||||
! With PSBLAS 4 we will have a better solution
|
||||
if (associated(prec%precv(i)%linmap%p_desc_U)) &
|
||||
& prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
if (associated(prec%precv(i)%linmap%p_desc_V))&
|
||||
& prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
end do
|
||||
end if
|
||||
|
||||
@@ -512,71 +507,6 @@ subroutine amg_d_hierarchy_bld(a,desc_a,prec,info)
|
||||
return
|
||||
|
||||
contains
|
||||
|
||||
#if ( __GNUC__ == 13 && __GNUC_MINOR__ == 3)
|
||||
! gfortran 13.3.0 generates a strange error here with MOLD
|
||||
! moving to SOURCE but only for this version, since it's heavier
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_d_onelev_type), intent(inout) :: level
|
||||
class(amg_d_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(save1)) then
|
||||
call save1%free(info)
|
||||
if (info == 0) deallocate(save1,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
call save2%free(info)
|
||||
if (info == 0) deallocate(save2,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
allocate(save1, source=level%sm,stat=info)
|
||||
if (info == 0) call level%sm%clone_settings(save1,info)
|
||||
if ((info == 0).and.allocated(level%sm2a)) then
|
||||
allocate(save2, source=level%sm2a,stat=info)
|
||||
if (info == 0) call level%sm2a%clone_settings(save2,info)
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine save_smoothers
|
||||
|
||||
subroutine restore_smoothers(level,save1, save2,info)
|
||||
type(amg_d_onelev_type), intent(inout), target :: level
|
||||
class(amg_d_base_smoother_type), allocatable, intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
|
||||
if (allocated(level%sm)) then
|
||||
if (info == 0) call level%sm%free(info)
|
||||
if (info == 0) deallocate(level%sm,stat=info)
|
||||
end if
|
||||
if (allocated(save1)) then
|
||||
if (info == 0) allocate(level%sm,source=save1,stat=info)
|
||||
if (info == 0) call save1%clone_settings(level%sm,info)
|
||||
end if
|
||||
|
||||
if (info /= 0) return
|
||||
|
||||
if (allocated(level%sm2a)) then
|
||||
if (info == 0) call level%sm2a%free(info)
|
||||
if (info == 0) deallocate(level%sm2a,stat=info)
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
if (info == 0) allocate(level%sm2a,source=save2,stat=info)
|
||||
if (info == 0) call save2%clone_settings(level%sm2a,info)
|
||||
if (info == 0) level%sm2 => level%sm2a
|
||||
else
|
||||
if (allocated(level%sm)) level%sm2 => level%sm
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#else
|
||||
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_d_onelev_type), intent(inout) :: level
|
||||
class(amg_d_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
@@ -635,5 +565,5 @@ contains
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
#endif
|
||||
|
||||
end subroutine amg_d_hierarchy_bld
|
||||
@@ -451,16 +451,11 @@ subroutine amg_s_hierarchy_bld(a,desc_a,prec,info)
|
||||
if (.not.associated(prec%precv(1)%base_desc,desc_a)) then
|
||||
prec%precv(1)%base_desc => prec%precv(1)%desc_ac
|
||||
end if
|
||||
do i=2, iszv
|
||||
do i=2, iszv
|
||||
prec%precv(i)%base_a => prec%precv(i)%ac
|
||||
prec%precv(i)%base_desc => prec%precv(i)%desc_ac
|
||||
! This is needed when the linmap object has been built
|
||||
! reusing the base_desc descriptor through a pointer.
|
||||
! With PSBLAS 4 we will have a better solution
|
||||
if (associated(prec%precv(i)%linmap%p_desc_U)) &
|
||||
& prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
if (associated(prec%precv(i)%linmap%p_desc_V))&
|
||||
& prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
end do
|
||||
end if
|
||||
|
||||
@@ -512,71 +507,6 @@ subroutine amg_s_hierarchy_bld(a,desc_a,prec,info)
|
||||
return
|
||||
|
||||
contains
|
||||
|
||||
#if ( __GNUC__ == 13 && __GNUC_MINOR__ == 3)
|
||||
! gfortran 13.3.0 generates a strange error here with MOLD
|
||||
! moving to SOURCE but only for this version, since it's heavier
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_s_onelev_type), intent(inout) :: level
|
||||
class(amg_s_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(save1)) then
|
||||
call save1%free(info)
|
||||
if (info == 0) deallocate(save1,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
call save2%free(info)
|
||||
if (info == 0) deallocate(save2,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
allocate(save1, source=level%sm,stat=info)
|
||||
if (info == 0) call level%sm%clone_settings(save1,info)
|
||||
if ((info == 0).and.allocated(level%sm2a)) then
|
||||
allocate(save2, source=level%sm2a,stat=info)
|
||||
if (info == 0) call level%sm2a%clone_settings(save2,info)
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine save_smoothers
|
||||
|
||||
subroutine restore_smoothers(level,save1, save2,info)
|
||||
type(amg_s_onelev_type), intent(inout), target :: level
|
||||
class(amg_s_base_smoother_type), allocatable, intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
|
||||
if (allocated(level%sm)) then
|
||||
if (info == 0) call level%sm%free(info)
|
||||
if (info == 0) deallocate(level%sm,stat=info)
|
||||
end if
|
||||
if (allocated(save1)) then
|
||||
if (info == 0) allocate(level%sm,source=save1,stat=info)
|
||||
if (info == 0) call save1%clone_settings(level%sm,info)
|
||||
end if
|
||||
|
||||
if (info /= 0) return
|
||||
|
||||
if (allocated(level%sm2a)) then
|
||||
if (info == 0) call level%sm2a%free(info)
|
||||
if (info == 0) deallocate(level%sm2a,stat=info)
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
if (info == 0) allocate(level%sm2a,source=save2,stat=info)
|
||||
if (info == 0) call save2%clone_settings(level%sm2a,info)
|
||||
if (info == 0) level%sm2 => level%sm2a
|
||||
else
|
||||
if (allocated(level%sm)) level%sm2 => level%sm
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#else
|
||||
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_s_onelev_type), intent(inout) :: level
|
||||
class(amg_s_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
@@ -635,5 +565,5 @@ contains
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
#endif
|
||||
|
||||
end subroutine amg_s_hierarchy_bld
|
||||
@@ -451,16 +451,11 @@ subroutine amg_z_hierarchy_bld(a,desc_a,prec,info)
|
||||
if (.not.associated(prec%precv(1)%base_desc,desc_a)) then
|
||||
prec%precv(1)%base_desc => prec%precv(1)%desc_ac
|
||||
end if
|
||||
do i=2, iszv
|
||||
do i=2, iszv
|
||||
prec%precv(i)%base_a => prec%precv(i)%ac
|
||||
prec%precv(i)%base_desc => prec%precv(i)%desc_ac
|
||||
! This is needed when the linmap object has been built
|
||||
! reusing the base_desc descriptor through a pointer.
|
||||
! With PSBLAS 4 we will have a better solution
|
||||
if (associated(prec%precv(i)%linmap%p_desc_U)) &
|
||||
& prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
if (associated(prec%precv(i)%linmap%p_desc_V))&
|
||||
& prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_U => prec%precv(i-1)%base_desc
|
||||
prec%precv(i)%linmap%p_desc_V => prec%precv(i)%base_desc
|
||||
end do
|
||||
end if
|
||||
|
||||
@@ -512,71 +507,6 @@ subroutine amg_z_hierarchy_bld(a,desc_a,prec,info)
|
||||
return
|
||||
|
||||
contains
|
||||
|
||||
#if ( __GNUC__ == 13 && __GNUC_MINOR__ == 3)
|
||||
! gfortran 13.3.0 generates a strange error here with MOLD
|
||||
! moving to SOURCE but only for this version, since it's heavier
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_z_onelev_type), intent(inout) :: level
|
||||
class(amg_z_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
if (allocated(save1)) then
|
||||
call save1%free(info)
|
||||
if (info == 0) deallocate(save1,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
call save2%free(info)
|
||||
if (info == 0) deallocate(save2,stat=info)
|
||||
if (info /= 0) return
|
||||
end if
|
||||
allocate(save1, source=level%sm,stat=info)
|
||||
if (info == 0) call level%sm%clone_settings(save1,info)
|
||||
if ((info == 0).and.allocated(level%sm2a)) then
|
||||
allocate(save2, source=level%sm2a,stat=info)
|
||||
if (info == 0) call level%sm2a%clone_settings(save2,info)
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine save_smoothers
|
||||
|
||||
subroutine restore_smoothers(level,save1, save2,info)
|
||||
type(amg_z_onelev_type), intent(inout), target :: level
|
||||
class(amg_z_base_smoother_type), allocatable, intent(inout) :: save1, save2
|
||||
integer(psb_ipk_), intent(out) :: info
|
||||
|
||||
info = 0
|
||||
|
||||
if (allocated(level%sm)) then
|
||||
if (info == 0) call level%sm%free(info)
|
||||
if (info == 0) deallocate(level%sm,stat=info)
|
||||
end if
|
||||
if (allocated(save1)) then
|
||||
if (info == 0) allocate(level%sm,source=save1,stat=info)
|
||||
if (info == 0) call save1%clone_settings(level%sm,info)
|
||||
end if
|
||||
|
||||
if (info /= 0) return
|
||||
|
||||
if (allocated(level%sm2a)) then
|
||||
if (info == 0) call level%sm2a%free(info)
|
||||
if (info == 0) deallocate(level%sm2a,stat=info)
|
||||
end if
|
||||
if (allocated(save2)) then
|
||||
if (info == 0) allocate(level%sm2a,source=save2,stat=info)
|
||||
if (info == 0) call save2%clone_settings(level%sm2a,info)
|
||||
if (info == 0) level%sm2 => level%sm2a
|
||||
else
|
||||
if (allocated(level%sm)) level%sm2 => level%sm
|
||||
end if
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
|
||||
#else
|
||||
|
||||
subroutine save_smoothers(level,save1, save2,info)
|
||||
type(amg_z_onelev_type), intent(inout) :: level
|
||||
class(amg_z_base_smoother_type), allocatable , intent(inout) :: save1, save2
|
||||
@@ -635,5 +565,5 @@ contains
|
||||
|
||||
return
|
||||
end subroutine restore_smoothers
|
||||
#endif
|
||||
|
||||
end subroutine amg_z_hierarchy_bld
|
||||
@@ -42,6 +42,8 @@ subroutine amg_c_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
use amg_c_base_aggregator_mod
|
||||
use amg_c_dec_aggregator_mod
|
||||
use amg_c_symdec_aggregator_mod
|
||||
#if !defined(SERIAL_MPI)
|
||||
#endif
|
||||
use amg_c_jac_smoother
|
||||
use amg_c_as_smoother
|
||||
use amg_c_diag_solver
|
||||
|
||||
@@ -98,7 +98,8 @@ subroutine amg_c_base_onelev_memory_use(lv,il,nl,ilmin,info,iout,verbosity,prefi
|
||||
prefix_ = ""
|
||||
end if
|
||||
|
||||
if ((me == 0).or.(verbosity_>0)) write(iout_,*) trim(prefix_)
|
||||
write(iout_,*) trim(prefix_)
|
||||
|
||||
|
||||
if (global_) then
|
||||
allocate(sz(6))
|
||||
|
||||
@@ -98,7 +98,8 @@ subroutine amg_d_base_onelev_memory_use(lv,il,nl,ilmin,info,iout,verbosity,prefi
|
||||
prefix_ = ""
|
||||
end if
|
||||
|
||||
if ((me == 0).or.(verbosity_>0)) write(iout_,*) trim(prefix_)
|
||||
write(iout_,*) trim(prefix_)
|
||||
|
||||
|
||||
if (global_) then
|
||||
allocate(sz(6))
|
||||
|
||||
@@ -98,7 +98,8 @@ subroutine amg_s_base_onelev_memory_use(lv,il,nl,ilmin,info,iout,verbosity,prefi
|
||||
prefix_ = ""
|
||||
end if
|
||||
|
||||
if ((me == 0).or.(verbosity_>0)) write(iout_,*) trim(prefix_)
|
||||
write(iout_,*) trim(prefix_)
|
||||
|
||||
|
||||
if (global_) then
|
||||
allocate(sz(6))
|
||||
|
||||
@@ -42,6 +42,8 @@ subroutine amg_z_base_onelev_csetc(lv,what,val,info,pos,idx)
|
||||
use amg_z_base_aggregator_mod
|
||||
use amg_z_dec_aggregator_mod
|
||||
use amg_z_symdec_aggregator_mod
|
||||
#if !defined(SERIAL_MPI)
|
||||
#endif
|
||||
use amg_z_jac_smoother
|
||||
use amg_z_as_smoother
|
||||
use amg_z_diag_solver
|
||||
|
||||
@@ -98,7 +98,8 @@ subroutine amg_z_base_onelev_memory_use(lv,il,nl,ilmin,info,iout,verbosity,prefi
|
||||
prefix_ = ""
|
||||
end if
|
||||
|
||||
if ((me == 0).or.(verbosity_>0)) write(iout_,*) trim(prefix_)
|
||||
write(iout_,*) trim(prefix_)
|
||||
|
||||
|
||||
if (global_) then
|
||||
allocate(sz(6))
|
||||
|
||||
@@ -140,7 +140,7 @@ subroutine amg_d_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
call tz%zero()
|
||||
|
||||
select case(sm%variant)
|
||||
case(amg_cheb_4_)
|
||||
case(amg_poly_lottes_)
|
||||
if (do_timings) call psb_tic(poly_1)
|
||||
block
|
||||
real(psb_dpk_) :: cz, cr
|
||||
@@ -155,7 +155,7 @@ subroutine amg_d_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
cz = (2*i*done-3)/(2*i*done+done)
|
||||
cr = (8*i*done-4)/((2*i*done+done)*sm%rho_ba)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz(cr,cz,done,done,ty,tz,tx,desc_data,info) ! zk = cz * zk-1 + cr * rk-1
|
||||
call psb_abgdxyz(cr,cz,done,done,ty,tz,tx,desc_data,info) ! zk = cz * zk-1 + cr * rk-1
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
if (do_timings) call psb_tic(poly_mv)
|
||||
call psb_spmm(-done,sm%pa,tz,done,r,desc_data,info,work=aux,trans=trans_)
|
||||
@@ -167,12 +167,12 @@ subroutine amg_d_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
cz = (2*sm%pdegree*done-3)/(2*sm%pdegree*done+done)
|
||||
cr = (8*sm%pdegree*done-4)/((2*sm%pdegree*done+done)*sm%rho_ba)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz(cr,cz,done,done,ty,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz(cr,cz,done,done,ty,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
end block
|
||||
if (do_timings) call psb_toc(poly_1)
|
||||
|
||||
case(amg_cheb_4_opt_)
|
||||
case(amg_poly_lottes_beta_)
|
||||
if (do_timings) call psb_tic(poly_2)
|
||||
block
|
||||
real(psb_dpk_) :: cz, cr
|
||||
@@ -195,7 +195,7 @@ subroutine amg_d_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
cz = (2*i*done-3)/(2*i*done+done)
|
||||
cr = (8*i*done-4)/((2*i*done+done)*sm%rho_ba)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz(cr,cz,sm%poly_beta(i),done,ty,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz(cr,cz,sm%poly_beta(i),done,ty,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
if (do_timings) call psb_tic(poly_mv)
|
||||
call psb_spmm(-done,sm%pa,tz,done,r,desc_data,info,work=aux,trans=trans_)
|
||||
@@ -205,11 +205,11 @@ subroutine amg_d_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
cz = (2*sm%pdegree*done-3)/(2*sm%pdegree*done+done)
|
||||
cr = (8*sm%pdegree*done-4)/((2*sm%pdegree*done+done)*sm%rho_ba)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz(cr,cz,sm%poly_beta(sm%pdegree),done,ty,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz(cr,cz,sm%poly_beta(sm%pdegree),done,ty,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
end block
|
||||
if (do_timings) call psb_toc(poly_2)
|
||||
case(amg_cheb_1_opt_)
|
||||
case(amg_poly_new_)
|
||||
if (do_timings) call psb_tic(poly_3)
|
||||
block
|
||||
real(psb_dpk_) :: sigma, theta, delta, rho_old, rho
|
||||
@@ -226,7 +226,7 @@ subroutine amg_d_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
if (do_timings) call psb_toc(poly_sv)
|
||||
call psb_geaxpby((done/sm%rho_ba),ty,dzero,r,desc_data,info)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz((done/theta),dzero,done,done,r,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz((done/theta),dzero,done,done,r,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
|
||||
! tz == d
|
||||
@@ -244,7 +244,7 @@ subroutine amg_d_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
! d_{k+1} = (rho rho_old) d_k + 2(rho/delta) r_{k+1}
|
||||
rho = done/(2*sigma - rho_old)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz((2*rho/delta),(rho*rho_old),done,done,r,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz((2*rho/delta),(rho*rho_old),done,done,r,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
rho_old = rho
|
||||
end do
|
||||
|
||||
@@ -75,9 +75,9 @@ subroutine amg_d_poly_smoother_bld(a,desc_a,sm,info,amold,vmold,imold)
|
||||
nrow_a = a%get_nrows()
|
||||
nztota = a%get_nzeros()
|
||||
select case(sm%variant)
|
||||
case(amg_cheb_4_)
|
||||
case(amg_poly_lottes_)
|
||||
! do nothing
|
||||
case(amg_cheb_4_opt_)
|
||||
case(amg_poly_lottes_beta_)
|
||||
if ((1<=sm%pdegree).and.(sm%pdegree<=30)) then
|
||||
call psb_realloc(sm%pdegree,sm%poly_beta,info)
|
||||
sm%poly_beta(1:sm%pdegree) = amg_d_poly_beta_mat(1:sm%pdegree,sm%pdegree)
|
||||
@@ -87,7 +87,7 @@ subroutine amg_d_poly_smoother_bld(a,desc_a,sm,info,amold,vmold,imold)
|
||||
& a_err='invalid sm%degree for poly_beta')
|
||||
goto 9999
|
||||
end if
|
||||
case(amg_cheb_1_opt_)
|
||||
case(amg_poly_new_)
|
||||
|
||||
if ((1<=sm%pdegree).and.(sm%pdegree<=30)) then
|
||||
!Ok
|
||||
|
||||
@@ -58,11 +58,11 @@ subroutine amg_d_poly_smoother_cseti(sm,what,val,info,idx)
|
||||
sm%pdegree = val
|
||||
case('POLY_VARIANT')
|
||||
select case(val)
|
||||
case(amg_cheb_4_,amg_cheb_4_opt_,amg_cheb_1_opt_)
|
||||
case(amg_poly_lottes_,amg_poly_lottes_beta_,amg_poly_new_)
|
||||
sm%variant = val
|
||||
case default
|
||||
write(0,*) 'Invalid choice for POLY_VARIANT, defaulting to amg_cheb_4_',val
|
||||
sm%variant = amg_cheb_4_
|
||||
write(0,*) 'Invalid choice for POLY_VARIANT, defaulting to amg_poly_lottes_',val
|
||||
sm%variant = amg_poly_lottes_
|
||||
end select
|
||||
case('POLY_RHO_ESTIMATE')
|
||||
select case(val)
|
||||
|
||||
@@ -77,17 +77,17 @@ subroutine amg_d_poly_smoother_descr(sm,info,iout,coarse,prefix)
|
||||
|
||||
write(iout_,*) trim(prefix_), ' Polynomial smoother '
|
||||
select case(sm%variant)
|
||||
case(amg_cheb_4_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','CHEB_4'
|
||||
case(amg_poly_lottes_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','POLY_LOTTES'
|
||||
write(iout_,*) trim(prefix_), ' Degree: ',sm%pdegree
|
||||
write(iout_,*) trim(prefix_), ' rho_ba: ',sm%rho_ba
|
||||
case(amg_cheb_4_opt_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','CHEB_4_OPT'
|
||||
case(amg_poly_lottes_beta_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','POLY_LOTTES_BETA'
|
||||
write(iout_,*) trim(prefix_), ' Degree: ',sm%pdegree
|
||||
write(iout_,*) trim(prefix_), ' rho_ba: ',sm%rho_ba
|
||||
if (allocated(sm%poly_beta)) write(iout_,*) trim(prefix_), ' Coefficients: ',sm%poly_beta(1:sm%pdegree)
|
||||
case(amg_cheb_1_opt_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','CHEB_1_OPT'
|
||||
case(amg_poly_new_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','POLY_NEW'
|
||||
write(iout_,*) trim(prefix_), ' Degree: ',sm%pdegree
|
||||
write(iout_,*) trim(prefix_), ' rho_ba: ',sm%rho_ba
|
||||
write(iout_,*) trim(prefix_), ' Coefficient: ',sm%cf_a
|
||||
|
||||
@@ -140,7 +140,7 @@ subroutine amg_s_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
call tz%zero()
|
||||
|
||||
select case(sm%variant)
|
||||
case(amg_cheb_4_)
|
||||
case(amg_poly_lottes_)
|
||||
if (do_timings) call psb_tic(poly_1)
|
||||
block
|
||||
real(psb_spk_) :: cz, cr
|
||||
@@ -155,7 +155,7 @@ subroutine amg_s_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
cz = (2*i*sone-3)/(2*i*sone+sone)
|
||||
cr = (8*i*sone-4)/((2*i*sone+sone)*sm%rho_ba)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz(cr,cz,sone,sone,ty,tz,tx,desc_data,info) ! zk = cz * zk-1 + cr * rk-1
|
||||
call psb_abgdxyz(cr,cz,sone,sone,ty,tz,tx,desc_data,info) ! zk = cz * zk-1 + cr * rk-1
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
if (do_timings) call psb_tic(poly_mv)
|
||||
call psb_spmm(-sone,sm%pa,tz,sone,r,desc_data,info,work=aux,trans=trans_)
|
||||
@@ -167,12 +167,12 @@ subroutine amg_s_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
cz = (2*sm%pdegree*sone-3)/(2*sm%pdegree*sone+sone)
|
||||
cr = (8*sm%pdegree*sone-4)/((2*sm%pdegree*sone+sone)*sm%rho_ba)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz(cr,cz,sone,sone,ty,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz(cr,cz,sone,sone,ty,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
end block
|
||||
if (do_timings) call psb_toc(poly_1)
|
||||
|
||||
case(amg_cheb_4_opt_)
|
||||
case(amg_poly_lottes_beta_)
|
||||
if (do_timings) call psb_tic(poly_2)
|
||||
block
|
||||
real(psb_spk_) :: cz, cr
|
||||
@@ -195,7 +195,7 @@ subroutine amg_s_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
cz = (2*i*sone-3)/(2*i*sone+sone)
|
||||
cr = (8*i*sone-4)/((2*i*sone+sone)*sm%rho_ba)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz(cr,cz,sm%poly_beta(i),sone,ty,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz(cr,cz,sm%poly_beta(i),sone,ty,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
if (do_timings) call psb_tic(poly_mv)
|
||||
call psb_spmm(-sone,sm%pa,tz,sone,r,desc_data,info,work=aux,trans=trans_)
|
||||
@@ -205,11 +205,11 @@ subroutine amg_s_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
cz = (2*sm%pdegree*sone-3)/(2*sm%pdegree*sone+sone)
|
||||
cr = (8*sm%pdegree*sone-4)/((2*sm%pdegree*sone+sone)*sm%rho_ba)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz(cr,cz,sm%poly_beta(sm%pdegree),sone,ty,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz(cr,cz,sm%poly_beta(sm%pdegree),sone,ty,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
end block
|
||||
if (do_timings) call psb_toc(poly_2)
|
||||
case(amg_cheb_1_opt_)
|
||||
case(amg_poly_new_)
|
||||
if (do_timings) call psb_tic(poly_3)
|
||||
block
|
||||
real(psb_spk_) :: sigma, theta, delta, rho_old, rho
|
||||
@@ -226,7 +226,7 @@ subroutine amg_s_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
if (do_timings) call psb_toc(poly_sv)
|
||||
call psb_geaxpby((sone/sm%rho_ba),ty,szero,r,desc_data,info)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz((sone/theta),szero,sone,sone,r,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz((sone/theta),szero,sone,sone,r,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
|
||||
! tz == d
|
||||
@@ -244,7 +244,7 @@ subroutine amg_s_poly_smoother_apply_vect(alpha,sm,x,beta,y,desc_data,trans,&
|
||||
! d_{k+1} = (rho rho_old) d_k + 2(rho/delta) r_{k+1}
|
||||
rho = sone/(2*sigma - rho_old)
|
||||
if (do_timings) call psb_tic(poly_vect)
|
||||
call psb_upd_xyz((2*rho/delta),(rho*rho_old),sone,sone,r,tz,tx,desc_data,info)
|
||||
call psb_abgdxyz((2*rho/delta),(rho*rho_old),sone,sone,r,tz,tx,desc_data,info)
|
||||
if (do_timings) call psb_toc(poly_vect)
|
||||
rho_old = rho
|
||||
end do
|
||||
|
||||
@@ -75,9 +75,9 @@ subroutine amg_s_poly_smoother_bld(a,desc_a,sm,info,amold,vmold,imold)
|
||||
nrow_a = a%get_nrows()
|
||||
nztota = a%get_nzeros()
|
||||
select case(sm%variant)
|
||||
case(amg_cheb_4_)
|
||||
case(amg_poly_lottes_)
|
||||
! do nothing
|
||||
case(amg_cheb_4_opt_)
|
||||
case(amg_poly_lottes_beta_)
|
||||
if ((1<=sm%pdegree).and.(sm%pdegree<=30)) then
|
||||
call psb_realloc(sm%pdegree,sm%poly_beta,info)
|
||||
sm%poly_beta(1:sm%pdegree) = amg_d_poly_beta_mat(1:sm%pdegree,sm%pdegree)
|
||||
@@ -87,7 +87,7 @@ subroutine amg_s_poly_smoother_bld(a,desc_a,sm,info,amold,vmold,imold)
|
||||
& a_err='invalid sm%degree for poly_beta')
|
||||
goto 9999
|
||||
end if
|
||||
case(amg_cheb_1_opt_)
|
||||
case(amg_poly_new_)
|
||||
|
||||
if ((1<=sm%pdegree).and.(sm%pdegree<=30)) then
|
||||
!Ok
|
||||
|
||||
@@ -58,11 +58,11 @@ subroutine amg_s_poly_smoother_cseti(sm,what,val,info,idx)
|
||||
sm%pdegree = val
|
||||
case('POLY_VARIANT')
|
||||
select case(val)
|
||||
case(amg_cheb_4_,amg_cheb_4_opt_,amg_cheb_1_opt_)
|
||||
case(amg_poly_lottes_,amg_poly_lottes_beta_,amg_poly_new_)
|
||||
sm%variant = val
|
||||
case default
|
||||
write(0,*) 'Invalid choice for POLY_VARIANT, defaulting to amg_cheb_4_',val
|
||||
sm%variant = amg_cheb_4_
|
||||
write(0,*) 'Invalid choice for POLY_VARIANT, defaulting to amg_poly_lottes_',val
|
||||
sm%variant = amg_poly_lottes_
|
||||
end select
|
||||
case('POLY_RHO_ESTIMATE')
|
||||
select case(val)
|
||||
|
||||
@@ -77,17 +77,17 @@ subroutine amg_s_poly_smoother_descr(sm,info,iout,coarse,prefix)
|
||||
|
||||
write(iout_,*) trim(prefix_), ' Polynomial smoother '
|
||||
select case(sm%variant)
|
||||
case(amg_cheb_4_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','CHEB_4'
|
||||
case(amg_poly_lottes_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','POLY_LOTTES'
|
||||
write(iout_,*) trim(prefix_), ' Degree: ',sm%pdegree
|
||||
write(iout_,*) trim(prefix_), ' rho_ba: ',sm%rho_ba
|
||||
case(amg_cheb_4_opt_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','CHEB_4_OPT'
|
||||
case(amg_poly_lottes_beta_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','POLY_LOTTES_BETA'
|
||||
write(iout_,*) trim(prefix_), ' Degree: ',sm%pdegree
|
||||
write(iout_,*) trim(prefix_), ' rho_ba: ',sm%rho_ba
|
||||
if (allocated(sm%poly_beta)) write(iout_,*) trim(prefix_), ' Coefficients: ',sm%poly_beta(1:sm%pdegree)
|
||||
case(amg_cheb_1_opt_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','CHEB_1_OPT'
|
||||
case(amg_poly_new_)
|
||||
write(iout_,*) trim(prefix_), ' variant: ','POLY_NEW'
|
||||
write(iout_,*) trim(prefix_), ' Degree: ',sm%pdegree
|
||||
write(iout_,*) trim(prefix_), ' rho_ba: ',sm%rho_ba
|
||||
write(iout_,*) trim(prefix_), ' Coefficient: ',sm%cf_a
|
||||
|
||||
@@ -53,6 +53,7 @@ subroutine amg_c_ilu_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
class(psb_i_base_vect_type), intent(in), optional :: imold
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: n_row,n_col, nrow_a, nztota, psb_fctype
|
||||
!!$ complex(psb_spk_), pointer :: ww(:), aux(:), tx(:),ty(:)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='c_ilu_solver_bld', ch_err
|
||||
|
||||
@@ -53,6 +53,7 @@ subroutine amg_d_ilu_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
class(psb_i_base_vect_type), intent(in), optional :: imold
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: n_row,n_col, nrow_a, nztota, psb_fctype
|
||||
!!$ real(psb_dpk_), pointer :: ww(:), aux(:), tx(:),ty(:)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='d_ilu_solver_bld', ch_err
|
||||
|
||||
@@ -53,6 +53,7 @@ subroutine amg_s_ilu_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
class(psb_i_base_vect_type), intent(in), optional :: imold
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: n_row,n_col, nrow_a, nztota, psb_fctype
|
||||
!!$ real(psb_spk_), pointer :: ww(:), aux(:), tx(:),ty(:)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='s_ilu_solver_bld', ch_err
|
||||
|
||||
@@ -53,6 +53,7 @@ subroutine amg_z_ilu_solver_bld(a,desc_a,sv,info,b,amold,vmold,imold)
|
||||
class(psb_i_base_vect_type), intent(in), optional :: imold
|
||||
! Local variables
|
||||
integer(psb_ipk_) :: n_row,n_col, nrow_a, nztota, psb_fctype
|
||||
!!$ complex(psb_dpk_), pointer :: ww(:), aux(:), tx(:),ty(:)
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer(psb_ipk_) :: np, me, i, err_act, debug_unit, debug_level
|
||||
character(len=20) :: name='z_ilu_solver_bld', ch_err
|
||||
|
||||
@@ -8,37 +8,32 @@ FINCLUDES=$(FMFLAG). $(FMFLAG)$(AMGMODDIR) $(FMFLAG)$(AMGINCDIR) $(PSBLAS_INCLUD
|
||||
|
||||
LINKOPT=
|
||||
EXEDIR=./runs
|
||||
DGEN2D=amg_d_pde2d_poisson_mod.o amg_d_pde2d_exp_mod.o \
|
||||
amg_d_pde2d_gauss_mod.o amg_d_pde2d_box_mod.o
|
||||
DGEN3D=amg_d_pde3d_poisson_mod.o amg_d_pde3d_exp_mod.o \
|
||||
amg_d_pde3d_gauss_mod.o amg_d_pde3d_box_mod.o
|
||||
SGEN2D=amg_s_pde2d_poisson_mod.o amg_s_pde2d_exp_mod.o \
|
||||
amg_s_pde2d_gauss_mod.o amg_s_pde2d_box_mod.o
|
||||
SGEN3D=amg_s_pde3d_poisson_mod.o amg_s_pde3d_exp_mod.o \
|
||||
amg_s_pde3d_gauss_mod.o amg_s_pde3d_box_mod.o
|
||||
DGEN2D=amg_d_pde2d_base_mod.o amg_d_pde2d_exp_mod.o amg_d_pde2d_gauss_mod.o amg_d_pde2d_box_mod.o
|
||||
DGEN3D=amg_d_pde3d_base_mod.o amg_d_pde3d_exp_mod.o amg_d_pde3d_gauss_mod.o amg_d_pde3d_box_mod.o
|
||||
SGEN2D=amg_s_pde2d_base_mod.o amg_s_pde2d_exp_mod.o amg_s_pde2d_gauss_mod.o amg_s_pde2d_box_mod.o
|
||||
SGEN3D=amg_s_pde3d_base_mod.o amg_s_pde3d_exp_mod.o amg_s_pde3d_gauss_mod.o amg_s_pde3d_box_mod.o
|
||||
|
||||
all: amg_s_pde3d amg_d_pde3d amg_s_pde2d amg_d_pde2d
|
||||
|
||||
amg_d_pde3d: amg_d_pde3d.o amg_d_genpde_mod.o $(DGEN3D) data_input.o
|
||||
$(FLINK) $(LINKOPT) amg_d_pde3d.o amg_d_genpde_mod.o $(DGEN3D) data_input.o \
|
||||
-o amg_d_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
$(FLINK) $(LINKOPT) amg_d_pde3d.o amg_d_genpde_mod.o $(DGEN3D) data_input.o -o amg_d_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
/bin/mv amg_d_pde3d $(EXEDIR)
|
||||
|
||||
amg_s_pde3d: amg_s_pde3d.o amg_s_genpde_mod.o $(SGEN3D) data_input.o
|
||||
$(FLINK) $(LINKOPT) amg_s_pde3d.o amg_s_genpde_mod.o $(SGEN3D) data_input.o \
|
||||
-o amg_s_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
$(FLINK) $(LINKOPT) amg_s_pde3d.o amg_s_genpde_mod.o $(SGEN3D) data_input.o -o amg_s_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
/bin/mv amg_s_pde3d $(EXEDIR)
|
||||
|
||||
amg_d_pde2d: amg_d_pde2d.o amg_d_genpde_mod.o $(DGEN2D) data_input.o
|
||||
$(FLINK) $(LINKOPT) amg_d_pde2d.o amg_d_genpde_mod.o $(DGEN2D) data_input.o \
|
||||
-o amg_d_pde2d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
$(FLINK) $(LINKOPT) amg_d_pde2d.o amg_d_genpde_mod.o $(DGEN2D) data_input.o -o amg_d_pde2d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
/bin/mv amg_d_pde2d $(EXEDIR)
|
||||
|
||||
amg_s_pde2d: amg_s_pde2d.o amg_s_genpde_mod.o $(SGEN2D) data_input.o
|
||||
$(FLINK) $(LINKOPT) amg_s_pde2d.o amg_s_genpde_mod.o $(SGEN2D) data_input.o \
|
||||
-o amg_s_pde2d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
$(FLINK) $(LINKOPT) amg_s_pde2d.o amg_s_genpde_mod.o $(SGEN2D) data_input.o -o amg_s_pde2d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
/bin/mv amg_s_pde2d $(EXEDIR)
|
||||
|
||||
amg_d_pde3d_rebld: amg_d_pde3d_rebld.o data_input.o
|
||||
$(FLINK) $(LINKOPT) amg_d_pde3d_rebld.o data_input.o -o amg_d_pde3d_rebld $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
|
||||
/bin/mv amg_d_pde3d_rebld $(EXEDIR)
|
||||
|
||||
amg_d_pde3d.o amg_s_pde3d.o amg_d_pde2d.o amg_s_pde2d.o: data_input.o
|
||||
|
||||
@@ -47,11 +42,6 @@ amg_s_pde3d.o: amg_s_genpde_mod.o $(SGEN3D)
|
||||
amg_d_pde2d.o: amg_d_genpde_mod.o $(DGEN2D)
|
||||
amg_s_pde2d.o: amg_s_genpde_mod.o $(SGEN2D)
|
||||
|
||||
amg_d_genpde_mod.o: $(DGEN3D)
|
||||
amg_s_genpde_mod.o: $(SGEN3D)
|
||||
amg_d_genpde_mod.o: $(DGEN2D)
|
||||
amg_s_genpde_mod.o: $(SGEN2D)
|
||||
|
||||
check: all
|
||||
cd runs && ./amg_d_pde2d <amg_pde2d.inp && ./amg_s_pde2d<amg_pde2d.inp
|
||||
|
||||
|
||||
@@ -69,7 +69,7 @@ program amg_d_pde2d
|
||||
use psb_krylov_mod
|
||||
use psb_util_mod
|
||||
use data_input
|
||||
use amg_d_pde2d_poisson_mod
|
||||
use amg_d_pde2d_base_mod
|
||||
use amg_d_pde2d_exp_mod
|
||||
use amg_d_pde2d_box_mod
|
||||
use amg_d_pde2d_gauss_mod
|
||||
@@ -81,8 +81,7 @@ program amg_d_pde2d
|
||||
|
||||
! input parameters
|
||||
character(len=20) :: kmethd, ptype
|
||||
character(len=5) :: afmt
|
||||
character(len=32) :: pdecoeff
|
||||
character(len=5) :: afmt, pdecoeff
|
||||
integer(psb_ipk_) :: idim
|
||||
integer(psb_epk_) :: system_size
|
||||
|
||||
@@ -127,7 +126,7 @@ program amg_d_pde2d
|
||||
! AMG cycles for ML
|
||||
! general AMG data
|
||||
character(len=32) :: mlcycle ! AMG cycle type
|
||||
integer(psb_ipk_) :: maxlevs ! maximum number of levels in AMG preconditioner
|
||||
integer(psb_ipk_) :: maxlevs ! maximum number of levels in AMG preconditioner
|
||||
|
||||
! AMG aggregation
|
||||
character(len=32) :: aggr_prol ! aggregation type: SMOOTHED, NONSMOOTHED
|
||||
@@ -147,8 +146,6 @@ program amg_d_pde2d
|
||||
integer(psb_ipk_) :: jsweeps ! (pre-)smoother / 1-lev prec. sweeps
|
||||
integer(psb_ipk_) :: degree ! degree for polynomial smoother
|
||||
character(len=32) :: pvariant ! polynomial variant
|
||||
character(len=32) :: prhovariant ! how to estimate rho(M^{-1}A)
|
||||
real(psb_dpk_) :: prhovalue ! if previous is set value, we set it from this one
|
||||
integer(psb_ipk_) :: novr ! number of overlap layers
|
||||
character(len=32) :: restr ! restriction over application of AS
|
||||
character(len=32) :: prol ! prolongation over application of AS
|
||||
@@ -165,8 +162,6 @@ program amg_d_pde2d
|
||||
integer(psb_ipk_) :: jsweeps2 ! post-smoother sweeps
|
||||
integer(psb_ipk_) :: degree2 ! degree for polynomial smoother
|
||||
character(len=32) :: pvariant2 ! polynomial variant
|
||||
character(len=32) :: prhovariant2 ! how to estimate rho(M^{-1}A)
|
||||
real(psb_dpk_) :: prhovalue2 ! if previous is set value, we set it from this one
|
||||
integer(psb_ipk_) :: novr2 ! number of overlap layers
|
||||
character(len=32) :: restr2 ! restriction over application of AS
|
||||
character(len=32) :: prol2 ! prolongation over application of AS
|
||||
@@ -179,15 +174,15 @@ program amg_d_pde2d
|
||||
real(psb_dpk_) :: thr2 ! threshold for ILUT factorization
|
||||
|
||||
! coarsest-level solver
|
||||
character(len=32) :: cmat ! coarsest matrix layout: REPL, DIST
|
||||
character(len=32) :: csolve ! coarsest-lev solver: BJAC, SLUDIST (distr.
|
||||
! mat.); UMF, MUMPS, SLU, ILU, ILUT, MILU
|
||||
! (repl. mat.)
|
||||
character(len=32) :: csbsolve ! coarsest-lev local subsolver: ILU, ILUT,
|
||||
! MILU, UMF, MUMPS, SLU
|
||||
integer(psb_ipk_) :: cfill ! fill-in for incomplete LU factorization
|
||||
real(psb_dpk_) :: cthres ! threshold for ILUT factorization
|
||||
integer(psb_ipk_) :: cjswp ! sweeps for GS or JAC coarsest-lev subsolver
|
||||
character(len=32) :: cmat ! coarsest matrix layout: REPL, DIST
|
||||
character(len=32) :: csolve ! coarsest-lev solver: BJAC, SLUDIST (distr.
|
||||
! mat.); UMF, MUMPS, SLU, ILU, ILUT, MILU
|
||||
! (repl. mat.)
|
||||
character(len=32) :: csbsolve ! coarsest-lev local subsolver: ILU, ILUT,
|
||||
! MILU, UMF, MUMPS, SLU
|
||||
integer(psb_ipk_) :: cfill ! fill-in for incomplete LU factorization
|
||||
real(psb_dpk_) :: cthres ! threshold for ILUT factorization
|
||||
integer(psb_ipk_) :: cjswp ! sweeps for GS or JAC coarsest-lev subsolver
|
||||
|
||||
! Dump data
|
||||
logical :: dump = .false.
|
||||
@@ -249,22 +244,18 @@ program amg_d_pde2d
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
select case(psb_toupper(trim(pdecoeff)))
|
||||
case("POISSON")
|
||||
case("CONST")
|
||||
call amg_gen_pde2d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_poisson,a2_poisson,&
|
||||
& b1_poisson,b2_poisson,c_poisson,g_poisson,info)
|
||||
& a1_base,a2_base,b1_base,b2_base,c_base,g_base,info)
|
||||
case("EXP")
|
||||
call amg_gen_pde2d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_exp,a2_exp,&
|
||||
& b1_exp,b2_exp,c_exp,g_exp,info)
|
||||
& a1_exp,a2_exp,b1_exp,b2_exp,c_exp,g_exp,info)
|
||||
case("BOX")
|
||||
call amg_gen_pde2d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_box,a2_box,&
|
||||
& b1_box,b2_box,c_box,g_box,info)
|
||||
& a1_box,a2_box,b1_box,b2_box,c_box,g_box,info)
|
||||
case("GAUSS")
|
||||
call amg_gen_pde2d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_gauss,a2_gauss,&
|
||||
& b1_gauss,b2_gauss,c_gauss,g_gauss,info)
|
||||
& a1_gauss,a2_gauss,b1_gauss,b2_gauss,c_gauss,g_gauss,info)
|
||||
case default
|
||||
info=psb_err_from_subroutine_
|
||||
ch_err='amg_gen_pdecoeff'
|
||||
@@ -353,12 +344,7 @@ program amg_d_pde2d
|
||||
call prec%set('smoother_sweeps', p_choice%jsweeps, info)
|
||||
call prec%set('poly_degree', p_choice%degree, info)
|
||||
call prec%set('poly_variant', p_choice%pvariant, info)
|
||||
if (p_choice%prhovalue > dzero ) then
|
||||
call prec%set('poly_rho_ba', p_choice%prhovalue, info)
|
||||
else
|
||||
call prec%set('poly_rho_estimate', p_choice%prhovariant, info)
|
||||
end if
|
||||
|
||||
|
||||
select case (psb_toupper(p_choice%smther))
|
||||
case ('GS','BWGS','FBGS','JACOBI','L1-JACOBI','L1-FBGS')
|
||||
! do nothing
|
||||
@@ -390,12 +376,6 @@ program amg_d_pde2d
|
||||
call prec%set('smoother_sweeps', p_choice%jsweeps2, info,pos='post')
|
||||
call prec%set('poly_degree', p_choice%degree2, info,pos='post')
|
||||
call prec%set('poly_variant', p_choice%pvariant2, info,pos='post')
|
||||
if (p_choice%prhovalue > dzero ) then
|
||||
call prec%set('poly_rho_ba', p_choice%prhovalue2, info,pos='post')
|
||||
else
|
||||
call prec%set('poly_rho_estimate', p_choice%prhovariant2, info,pos='post')
|
||||
end if
|
||||
|
||||
select case (psb_toupper(p_choice%smther2))
|
||||
case ('GS','BWGS','FBGS','JACOBI','L1-JACOBI','L1-FBGS')
|
||||
! do nothing
|
||||
@@ -612,9 +592,7 @@ contains
|
||||
call read_data(prec%smther,inp_unit) ! smoother type
|
||||
call read_data(prec%jsweeps,inp_unit) ! (pre-)smoother / 1-lev prec sweeps
|
||||
call read_data(prec%degree,inp_unit) ! (pre-)smoother / 1-lev prec sweeps
|
||||
call read_data(prec%pvariant,inp_unit) !
|
||||
call read_data(prec%prhovariant,inp_unit)! how to estimate rho(M^{-1}A)
|
||||
call read_data(prec%prhovalue,inp_unit) ! if previous is set value, we set it from this one
|
||||
call read_data(prec%pvariant,inp_unit) !
|
||||
call read_data(prec%novr,inp_unit) ! number of overlap layers
|
||||
call read_data(prec%restr,inp_unit) ! restriction over application of AS
|
||||
call read_data(prec%prol,inp_unit) ! prolongation over application of AS
|
||||
@@ -629,8 +607,6 @@ contains
|
||||
call read_data(prec%jsweeps2,inp_unit) ! (post-)smoother sweeps
|
||||
call read_data(prec%degree2,inp_unit) ! (post-)smoother sweeps
|
||||
call read_data(prec%pvariant2,inp_unit) !
|
||||
call read_data(prec%prhovariant2,inp_unit)! how to estimate rho(M^{-1}A)
|
||||
call read_data(prec%prhovalue2,inp_unit) ! if previous is set value, we set it from this one
|
||||
call read_data(prec%novr2,inp_unit) ! number of overlap layers
|
||||
call read_data(prec%restr2,inp_unit) ! restriction over application of AS
|
||||
call read_data(prec%prol2,inp_unit) ! prolongation over application of AS
|
||||
@@ -703,8 +679,6 @@ contains
|
||||
call psb_bcast(ctxt,prec%jsweeps)
|
||||
call psb_bcast(ctxt,prec%degree)
|
||||
call psb_bcast(ctxt,prec%pvariant)
|
||||
call psb_bcast(ctxt,prec%prhovariant)
|
||||
call psb_bcast(ctxt,prec%prhovalue)
|
||||
call psb_bcast(ctxt,prec%novr)
|
||||
call psb_bcast(ctxt,prec%restr)
|
||||
call psb_bcast(ctxt,prec%prol)
|
||||
@@ -719,8 +693,6 @@ contains
|
||||
call psb_bcast(ctxt,prec%jsweeps2)
|
||||
call psb_bcast(ctxt,prec%degree2)
|
||||
call psb_bcast(ctxt,prec%pvariant2)
|
||||
call psb_bcast(ctxt,prec%prhovariant2)
|
||||
call psb_bcast(ctxt,prec%prhovalue2)
|
||||
call psb_bcast(ctxt,prec%novr2)
|
||||
call psb_bcast(ctxt,prec%restr2)
|
||||
call psb_bcast(ctxt,prec%prol2)
|
||||
|
||||
+30
-30
@@ -34,56 +34,56 @@
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
module amg_d_pde2d_poisson_mod
|
||||
module amg_d_pde2d_base_mod
|
||||
use psb_base_mod, only : psb_dpk_, dzero, done
|
||||
real(psb_dpk_), save, private :: epsilon=done/80
|
||||
contains
|
||||
subroutine pde_set_parm2d_poisson(dat)
|
||||
subroutine pde_set_parm2d_base(dat)
|
||||
real(psb_dpk_), intent(in) :: dat
|
||||
epsilon = dat
|
||||
end subroutine pde_set_parm2d_poisson
|
||||
end subroutine pde_set_parm2d_base
|
||||
!
|
||||
! functions parametrizing the differential equation
|
||||
!
|
||||
function b1_poisson(x,y)
|
||||
function b1_base(x,y)
|
||||
implicit none
|
||||
real(psb_dpk_) :: b1_poisson
|
||||
real(psb_dpk_) :: b1_base
|
||||
real(psb_dpk_), intent(in) :: x,y
|
||||
b1_poisson = dzero
|
||||
end function b1_poisson
|
||||
function b2_poisson(x,y)
|
||||
b1_base = dzero/1.414_psb_dpk_
|
||||
end function b1_base
|
||||
function b2_base(x,y)
|
||||
implicit none
|
||||
real(psb_dpk_) :: b2_poisson
|
||||
real(psb_dpk_) :: b2_base
|
||||
real(psb_dpk_), intent(in) :: x,y
|
||||
b2_poisson = dzero
|
||||
end function b2_poisson
|
||||
function c_poisson(x,y)
|
||||
b2_base = dzero/1.414_psb_dpk_
|
||||
end function b2_base
|
||||
function c_base(x,y)
|
||||
implicit none
|
||||
real(psb_dpk_) :: c_poisson
|
||||
real(psb_dpk_) :: c_base
|
||||
real(psb_dpk_), intent(in) :: x,y
|
||||
c_poisson = dzero
|
||||
end function c_poisson
|
||||
function a1_poisson(x,y)
|
||||
c_base = dzero
|
||||
end function c_base
|
||||
function a1_base(x,y)
|
||||
implicit none
|
||||
real(psb_dpk_) :: a1_poisson
|
||||
real(psb_dpk_) :: a1_base
|
||||
real(psb_dpk_), intent(in) :: x,y
|
||||
a1_poisson=done*epsilon
|
||||
end function a1_poisson
|
||||
function a2_poisson(x,y)
|
||||
a1_base=done*epsilon
|
||||
end function a1_base
|
||||
function a2_base(x,y)
|
||||
implicit none
|
||||
real(psb_dpk_) :: a2_poisson
|
||||
real(psb_dpk_) :: a2_base
|
||||
real(psb_dpk_), intent(in) :: x,y
|
||||
a2_poisson=done*epsilon
|
||||
end function a2_poisson
|
||||
function g_poisson(x,y)
|
||||
a2_base=done*epsilon
|
||||
end function a2_base
|
||||
function g_base(x,y)
|
||||
implicit none
|
||||
real(psb_dpk_) :: g_poisson
|
||||
real(psb_dpk_) :: g_base
|
||||
real(psb_dpk_), intent(in) :: x,y
|
||||
g_poisson = dzero
|
||||
g_base = dzero
|
||||
if (x == done) then
|
||||
g_poisson = done
|
||||
g_base = done
|
||||
else if (x == dzero) then
|
||||
g_poisson = done
|
||||
g_base = done
|
||||
end if
|
||||
end function g_poisson
|
||||
end module amg_d_pde2d_poisson_mod
|
||||
end function g_base
|
||||
end module amg_d_pde2d_base_mod
|
||||
@@ -70,7 +70,7 @@ program amg_d_pde3d
|
||||
use psb_krylov_mod
|
||||
use psb_util_mod
|
||||
use data_input
|
||||
use amg_d_pde3d_poisson_mod
|
||||
use amg_d_pde3d_base_mod
|
||||
use amg_d_pde3d_exp_mod
|
||||
use amg_d_pde3d_box_mod
|
||||
use amg_d_pde3d_gauss_mod
|
||||
@@ -82,8 +82,7 @@ program amg_d_pde3d
|
||||
|
||||
! input parameters
|
||||
character(len=20) :: kmethd, ptype
|
||||
character(len=5) :: afmt
|
||||
character(len=32) :: pdecoeff
|
||||
character(len=5) :: afmt, pdecoeff
|
||||
integer(psb_ipk_) :: idim
|
||||
integer(psb_epk_) :: system_size
|
||||
|
||||
@@ -148,8 +147,6 @@ program amg_d_pde3d
|
||||
integer(psb_ipk_) :: jsweeps ! (pre-)smoother / 1-lev prec. sweeps
|
||||
integer(psb_ipk_) :: degree ! degree for polynomial smoother
|
||||
character(len=32) :: pvariant ! polynomial variant
|
||||
character(len=32) :: prhovariant ! how to estimate rho(M^{-1}A)
|
||||
real(psb_dpk_) :: prhovalue ! if previous is set value, we set it from this one
|
||||
integer(psb_ipk_) :: novr ! number of overlap layers
|
||||
character(len=32) :: restr ! restriction over application of AS
|
||||
character(len=32) :: prol ! prolongation over application of AS
|
||||
@@ -166,8 +163,6 @@ program amg_d_pde3d
|
||||
integer(psb_ipk_) :: jsweeps2 ! post-smoother sweeps
|
||||
integer(psb_ipk_) :: degree2 ! degree for polynomial smoother
|
||||
character(len=32) :: pvariant2 ! polynomial variant
|
||||
character(len=32) :: prhovariant2 ! how to estimate rho(M^{-1}A)
|
||||
real(psb_dpk_) :: prhovalue2 ! if previous is set value, we set it from this one
|
||||
integer(psb_ipk_) :: novr2 ! number of overlap layers
|
||||
character(len=32) :: restr2 ! restriction over application of AS
|
||||
character(len=32) :: prol2 ! prolongation over application of AS
|
||||
@@ -251,22 +246,18 @@ program amg_d_pde3d
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
select case(psb_toupper(trim(pdecoeff)))
|
||||
case("POISSON")
|
||||
case("CONST")
|
||||
call amg_gen_pde3d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_poisson,a2_poisson,a3_poisson,&
|
||||
& b1_poisson,b2_poisson,b3_poisson,c_poisson,g_poisson,info)
|
||||
& a1_base,a2_base,a3_base,b1_base,b2_base,b3_base,c_base,g_base,info)
|
||||
case("EXP")
|
||||
call amg_gen_pde3d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_exp,a2_exp,a3_exp,&
|
||||
& b1_exp,b2_exp,b3_exp,c_exp,g_exp,info)
|
||||
& a1_exp,a2_exp,a3_exp,b1_exp,b2_exp,b3_exp,c_exp,g_exp,info)
|
||||
case("BOX")
|
||||
call amg_gen_pde3d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_box,a2_box,a3_box,&
|
||||
& b1_box,b2_box,b3_box,c_box,g_box,info)
|
||||
& a1_box,a2_box,a3_box,b1_box,b2_box,b3_box,c_box,g_box,info)
|
||||
case("GAUSS")
|
||||
call amg_gen_pde3d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_gauss,a2_gauss,a3_gauss,&
|
||||
& b1_gauss,b2_gauss,b3_gauss,c_gauss,g_gauss,info)
|
||||
& a1_gauss,a2_gauss,a3_gauss,b1_gauss,b2_gauss,b3_gauss,c_gauss,g_gauss,info)
|
||||
case default
|
||||
info=psb_err_from_subroutine_
|
||||
ch_err='amg_gen_pdecoeff'
|
||||
@@ -357,12 +348,7 @@ program amg_d_pde3d
|
||||
call prec%set('smoother_sweeps', p_choice%jsweeps, info)
|
||||
call prec%set('poly_degree', p_choice%degree, info)
|
||||
call prec%set('poly_variant', p_choice%pvariant, info)
|
||||
if (p_choice%prhovalue > dzero ) then
|
||||
call prec%set('poly_rho_ba', p_choice%prhovalue, info)
|
||||
else
|
||||
call prec%set('poly_rho_estimate', p_choice%prhovariant, info)
|
||||
end if
|
||||
|
||||
|
||||
select case (psb_toupper(p_choice%smther))
|
||||
case ('GS','BWGS','FBGS','JACOBI','L1-JACOBI','L1-FBGS')
|
||||
! do nothing
|
||||
@@ -394,12 +380,6 @@ program amg_d_pde3d
|
||||
call prec%set('smoother_sweeps', p_choice%jsweeps2, info,pos='post')
|
||||
call prec%set('poly_degree', p_choice%degree2, info,pos='post')
|
||||
call prec%set('poly_variant', p_choice%pvariant2, info,pos='post')
|
||||
if (p_choice%prhovalue > dzero ) then
|
||||
call prec%set('poly_rho_ba', p_choice%prhovalue2, info,pos='post')
|
||||
else
|
||||
call prec%set('poly_rho_estimate', p_choice%prhovariant2, info,pos='post')
|
||||
end if
|
||||
|
||||
select case (psb_toupper(p_choice%smther2))
|
||||
case ('GS','BWGS','FBGS','JACOBI','L1-JACOBI','L1-FBGS')
|
||||
! do nothing
|
||||
@@ -616,9 +596,7 @@ contains
|
||||
call read_data(prec%smther,inp_unit) ! smoother type
|
||||
call read_data(prec%jsweeps,inp_unit) ! (pre-)smoother / 1-lev prec sweeps
|
||||
call read_data(prec%degree,inp_unit) ! (pre-)smoother / 1-lev prec sweeps
|
||||
call read_data(prec%pvariant,inp_unit) !
|
||||
call read_data(prec%prhovariant,inp_unit)! how to estimate rho(M^{-1}A)
|
||||
call read_data(prec%prhovalue,inp_unit) ! if previous is set value, we set it from this one
|
||||
call read_data(prec%pvariant,inp_unit) !
|
||||
call read_data(prec%novr,inp_unit) ! number of overlap layers
|
||||
call read_data(prec%restr,inp_unit) ! restriction over application of AS
|
||||
call read_data(prec%prol,inp_unit) ! prolongation over application of AS
|
||||
@@ -633,8 +611,6 @@ contains
|
||||
call read_data(prec%jsweeps2,inp_unit) ! (post-)smoother sweeps
|
||||
call read_data(prec%degree2,inp_unit) ! (post-)smoother sweeps
|
||||
call read_data(prec%pvariant2,inp_unit) !
|
||||
call read_data(prec%prhovariant2,inp_unit)! how to estimate rho(M^{-1}A)
|
||||
call read_data(prec%prhovalue2,inp_unit) ! if previous is set value, we set it from this one
|
||||
call read_data(prec%novr2,inp_unit) ! number of overlap layers
|
||||
call read_data(prec%restr2,inp_unit) ! restriction over application of AS
|
||||
call read_data(prec%prol2,inp_unit) ! prolongation over application of AS
|
||||
@@ -707,8 +683,6 @@ contains
|
||||
call psb_bcast(ctxt,prec%jsweeps)
|
||||
call psb_bcast(ctxt,prec%degree)
|
||||
call psb_bcast(ctxt,prec%pvariant)
|
||||
call psb_bcast(ctxt,prec%prhovariant)
|
||||
call psb_bcast(ctxt,prec%prhovalue)
|
||||
call psb_bcast(ctxt,prec%novr)
|
||||
call psb_bcast(ctxt,prec%restr)
|
||||
call psb_bcast(ctxt,prec%prol)
|
||||
@@ -723,8 +697,6 @@ contains
|
||||
call psb_bcast(ctxt,prec%jsweeps2)
|
||||
call psb_bcast(ctxt,prec%degree2)
|
||||
call psb_bcast(ctxt,prec%pvariant2)
|
||||
call psb_bcast(ctxt,prec%prhovariant2)
|
||||
call psb_bcast(ctxt,prec%prhovalue2)
|
||||
call psb_bcast(ctxt,prec%novr2)
|
||||
call psb_bcast(ctxt,prec%restr2)
|
||||
call psb_bcast(ctxt,prec%prol2)
|
||||
|
||||
+38
-38
@@ -34,68 +34,68 @@
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
module amg_d_pde3d_poisson_mod
|
||||
module amg_d_pde3d_base_mod
|
||||
use psb_base_mod, only : psb_dpk_, done, dzero
|
||||
real(psb_dpk_), save, private :: epsilon=done/80
|
||||
contains
|
||||
subroutine pde_set_parm3d_poisson(dat)
|
||||
subroutine pde_set_parm3d_base(dat)
|
||||
real(psb_dpk_), intent(in) :: dat
|
||||
epsilon = dat
|
||||
end subroutine pde_set_parm3d_poisson
|
||||
end subroutine pde_set_parm3d_base
|
||||
!
|
||||
! functions parametrizing the differential equation
|
||||
!
|
||||
function b1_poisson(x,y,z)
|
||||
function b1_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_dpk_) :: b1_poisson
|
||||
real(psb_dpk_) :: b1_base
|
||||
real(psb_dpk_), intent(in) :: x,y,z
|
||||
b1_poisson=dzero
|
||||
end function b1_poisson
|
||||
function b2_poisson(x,y,z)
|
||||
b1_base=dzero/sqrt(3.0_psb_dpk_)
|
||||
end function b1_base
|
||||
function b2_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_dpk_) :: b2_poisson
|
||||
real(psb_dpk_) :: b2_base
|
||||
real(psb_dpk_), intent(in) :: x,y,z
|
||||
b2_poisson=dzero
|
||||
end function b2_poisson
|
||||
function b3_poisson(x,y,z)
|
||||
b2_base=dzero/sqrt(3.0_psb_dpk_)
|
||||
end function b2_base
|
||||
function b3_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_dpk_) :: b3_poisson
|
||||
real(psb_dpk_) :: b3_base
|
||||
real(psb_dpk_), intent(in) :: x,y,z
|
||||
b3_poisson=dzero
|
||||
end function b3_poisson
|
||||
function c_poisson(x,y,z)
|
||||
b3_base=dzero/sqrt(3.0_psb_dpk_)
|
||||
end function b3_base
|
||||
function c_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_dpk_) :: c_poisson
|
||||
real(psb_dpk_) :: c_base
|
||||
real(psb_dpk_), intent(in) :: x,y,z
|
||||
c_poisson=dzero
|
||||
end function c_poisson
|
||||
function a1_poisson(x,y,z)
|
||||
c_base=dzero
|
||||
end function c_base
|
||||
function a1_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_dpk_) :: a1_poisson
|
||||
real(psb_dpk_) :: a1_base
|
||||
real(psb_dpk_), intent(in) :: x,y,z
|
||||
a1_poisson=epsilon
|
||||
end function a1_poisson
|
||||
function a2_poisson(x,y,z)
|
||||
a1_base=epsilon
|
||||
end function a1_base
|
||||
function a2_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_dpk_) :: a2_poisson
|
||||
real(psb_dpk_) :: a2_base
|
||||
real(psb_dpk_), intent(in) :: x,y,z
|
||||
a2_poisson=epsilon
|
||||
end function a2_poisson
|
||||
function a3_poisson(x,y,z)
|
||||
a2_base=epsilon
|
||||
end function a2_base
|
||||
function a3_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_dpk_) :: a3_poisson
|
||||
real(psb_dpk_) :: a3_base
|
||||
real(psb_dpk_), intent(in) :: x,y,z
|
||||
a3_poisson=epsilon
|
||||
end function a3_poisson
|
||||
function g_poisson(x,y,z)
|
||||
a3_base=epsilon
|
||||
end function a3_base
|
||||
function g_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_dpk_) :: g_poisson
|
||||
real(psb_dpk_) :: g_base
|
||||
real(psb_dpk_), intent(in) :: x,y,z
|
||||
g_poisson = dzero
|
||||
g_base = dzero
|
||||
if (x == done) then
|
||||
g_poisson = done
|
||||
g_base = done
|
||||
else if (x == dzero) then
|
||||
g_poisson = done
|
||||
g_base = done
|
||||
end if
|
||||
end function g_poisson
|
||||
end module amg_d_pde3d_poisson_mod
|
||||
end function g_base
|
||||
end module amg_d_pde3d_base_mod
|
||||
@@ -69,7 +69,7 @@ program amg_s_pde2d
|
||||
use psb_krylov_mod
|
||||
use psb_util_mod
|
||||
use data_input
|
||||
use amg_s_pde2d_poisson_mod
|
||||
use amg_s_pde2d_base_mod
|
||||
use amg_s_pde2d_exp_mod
|
||||
use amg_s_pde2d_box_mod
|
||||
use amg_s_pde2d_gauss_mod
|
||||
@@ -81,8 +81,7 @@ program amg_s_pde2d
|
||||
|
||||
! input parameters
|
||||
character(len=20) :: kmethd, ptype
|
||||
character(len=5) :: afmt
|
||||
character(len=32) :: pdecoeff
|
||||
character(len=5) :: afmt, pdecoeff
|
||||
integer(psb_ipk_) :: idim
|
||||
integer(psb_epk_) :: system_size
|
||||
|
||||
@@ -127,16 +126,16 @@ program amg_s_pde2d
|
||||
! AMG cycles for ML
|
||||
! general AMG data
|
||||
character(len=32) :: mlcycle ! AMG cycle type
|
||||
integer(psb_ipk_) :: maxlevs ! maximum number of levels in AMG preconditioner
|
||||
integer(psb_ipk_) :: maxlevs ! maximum number of levels in AMG preconditioner
|
||||
|
||||
! AMG aggregation
|
||||
character(len=32) :: aggr_prol ! aggregation type: SMOOTHED, NONSMOOTHED
|
||||
character(len=32) :: par_aggr_alg ! parallel aggregation algorithm: DEC, SYMDEC
|
||||
character(len=32) :: aggr_type ! Type of aggregation SOC1, SOC2, MATCHBOXP
|
||||
integer(psb_ipk_) :: aggr_size ! Requested size of the aggregates for MATCHBOXP
|
||||
character(len=32) :: aggr_ord ! ordering for aggregation: NATURAL, DEGREE
|
||||
character(len=32) :: aggr_filter ! filtering: FILTER, NO_FILTER
|
||||
real(psb_spk_) :: mncrratio ! minimum aggregation ratio
|
||||
character(len=32) :: aggr_type ! Type of aggregation SOC1, SOC2, MATCHBOXP
|
||||
integer(psb_ipk_) :: aggr_size ! Requested size of the aggregates for MATCHBOXP
|
||||
character(len=32) :: aggr_ord ! ordering for aggregation: NATURAL, DEGREE
|
||||
character(len=32) :: aggr_filter ! filtering: FILTER, NO_FILTER
|
||||
real(psb_spk_) :: mncrratio ! minimum aggregation ratio
|
||||
real(psb_spk_), allocatable :: athresv(:) ! smoothed aggregation threshold vector
|
||||
integer(psb_ipk_) :: thrvsz ! size of threshold vector
|
||||
real(psb_spk_) :: athres ! smoothed aggregation threshold
|
||||
@@ -147,8 +146,6 @@ program amg_s_pde2d
|
||||
integer(psb_ipk_) :: jsweeps ! (pre-)smoother / 1-lev prec. sweeps
|
||||
integer(psb_ipk_) :: degree ! degree for polynomial smoother
|
||||
character(len=32) :: pvariant ! polynomial variant
|
||||
character(len=32) :: prhovariant ! how to estimate rho(M^{-1}A)
|
||||
real(psb_spk_) :: prhovalue ! if previous is set value, we set it from this one
|
||||
integer(psb_ipk_) :: novr ! number of overlap layers
|
||||
character(len=32) :: restr ! restriction over application of AS
|
||||
character(len=32) :: prol ! prolongation over application of AS
|
||||
@@ -165,8 +162,6 @@ program amg_s_pde2d
|
||||
integer(psb_ipk_) :: jsweeps2 ! post-smoother sweeps
|
||||
integer(psb_ipk_) :: degree2 ! degree for polynomial smoother
|
||||
character(len=32) :: pvariant2 ! polynomial variant
|
||||
character(len=32) :: prhovariant2 ! how to estimate rho(M^{-1}A)
|
||||
real(psb_spk_) :: prhovalue2 ! if previous is set value, we set it from this one
|
||||
integer(psb_ipk_) :: novr2 ! number of overlap layers
|
||||
character(len=32) :: restr2 ! restriction over application of AS
|
||||
character(len=32) :: prol2 ! prolongation over application of AS
|
||||
@@ -179,15 +174,15 @@ program amg_s_pde2d
|
||||
real(psb_spk_) :: thr2 ! threshold for ILUT factorization
|
||||
|
||||
! coarsest-level solver
|
||||
character(len=32) :: cmat ! coarsest matrix layout: REPL, DIST
|
||||
character(len=32) :: csolve ! coarsest-lev solver: BJAC, SLUDIST (distr.
|
||||
! mat.); UMF, MUMPS, SLU, ILU, ILUT, MILU
|
||||
! (repl. mat.)
|
||||
character(len=32) :: csbsolve ! coarsest-lev local subsolver: ILU, ILUT,
|
||||
! MILU, UMF, MUMPS, SLU
|
||||
integer(psb_ipk_) :: cfill ! fill-in for incomplete LU factorization
|
||||
real(psb_spk_) :: cthres ! threshold for ILUT factorization
|
||||
integer(psb_ipk_) :: cjswp ! sweeps for GS or JAC coarsest-lev subsolver
|
||||
character(len=32) :: cmat ! coarsest matrix layout: REPL, DIST
|
||||
character(len=32) :: csolve ! coarsest-lev solver: BJAC, SLUDIST (distr.
|
||||
! mat.); UMF, MUMPS, SLU, ILU, ILUT, MILU
|
||||
! (repl. mat.)
|
||||
character(len=32) :: csbsolve ! coarsest-lev local subsolver: ILU, ILUT,
|
||||
! MILU, UMF, MUMPS, SLU
|
||||
integer(psb_ipk_) :: cfill ! fill-in for incomplete LU factorization
|
||||
real(psb_spk_) :: cthres ! threshold for ILUT factorization
|
||||
integer(psb_ipk_) :: cjswp ! sweeps for GS or JAC coarsest-lev subsolver
|
||||
|
||||
! Dump data
|
||||
logical :: dump = .false.
|
||||
@@ -249,22 +244,18 @@ program amg_s_pde2d
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
select case(psb_toupper(trim(pdecoeff)))
|
||||
case("POISSON")
|
||||
case("CONST")
|
||||
call amg_gen_pde2d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_poisson,a2_poisson,&
|
||||
& b1_poisson,b2_poisson,c_poisson,g_poisson,info)
|
||||
& a1_base,a2_base,b1_base,b2_base,c_base,g_base,info)
|
||||
case("EXP")
|
||||
call amg_gen_pde2d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_exp,a2_exp,&
|
||||
& b1_exp,b2_exp,c_exp,g_exp,info)
|
||||
& a1_exp,a2_exp,b1_exp,b2_exp,c_exp,g_exp,info)
|
||||
case("BOX")
|
||||
call amg_gen_pde2d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_box,a2_box,&
|
||||
& b1_box,b2_box,c_box,g_box,info)
|
||||
& a1_box,a2_box,b1_box,b2_box,c_box,g_box,info)
|
||||
case("GAUSS")
|
||||
call amg_gen_pde2d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_gauss,a2_gauss,&
|
||||
& b1_gauss,b2_gauss,c_gauss,g_gauss,info)
|
||||
& a1_gauss,a2_gauss,b1_gauss,b2_gauss,c_gauss,g_gauss,info)
|
||||
case default
|
||||
info=psb_err_from_subroutine_
|
||||
ch_err='amg_gen_pdecoeff'
|
||||
@@ -353,12 +344,7 @@ program amg_s_pde2d
|
||||
call prec%set('smoother_sweeps', p_choice%jsweeps, info)
|
||||
call prec%set('poly_degree', p_choice%degree, info)
|
||||
call prec%set('poly_variant', p_choice%pvariant, info)
|
||||
if (p_choice%prhovalue > szero ) then
|
||||
call prec%set('poly_rho_ba', p_choice%prhovalue, info)
|
||||
else
|
||||
call prec%set('poly_rho_estimate', p_choice%prhovariant, info)
|
||||
end if
|
||||
|
||||
|
||||
select case (psb_toupper(p_choice%smther))
|
||||
case ('GS','BWGS','FBGS','JACOBI','L1-JACOBI','L1-FBGS')
|
||||
! do nothing
|
||||
@@ -390,12 +376,6 @@ program amg_s_pde2d
|
||||
call prec%set('smoother_sweeps', p_choice%jsweeps2, info,pos='post')
|
||||
call prec%set('poly_degree', p_choice%degree2, info,pos='post')
|
||||
call prec%set('poly_variant', p_choice%pvariant2, info,pos='post')
|
||||
if (p_choice%prhovalue > szero ) then
|
||||
call prec%set('poly_rho_ba', p_choice%prhovalue2, info,pos='post')
|
||||
else
|
||||
call prec%set('poly_rho_estimate', p_choice%prhovariant2, info,pos='post')
|
||||
end if
|
||||
|
||||
select case (psb_toupper(p_choice%smther2))
|
||||
case ('GS','BWGS','FBGS','JACOBI','L1-JACOBI','L1-FBGS')
|
||||
! do nothing
|
||||
@@ -612,9 +592,7 @@ contains
|
||||
call read_data(prec%smther,inp_unit) ! smoother type
|
||||
call read_data(prec%jsweeps,inp_unit) ! (pre-)smoother / 1-lev prec sweeps
|
||||
call read_data(prec%degree,inp_unit) ! (pre-)smoother / 1-lev prec sweeps
|
||||
call read_data(prec%pvariant,inp_unit) !
|
||||
call read_data(prec%prhovariant,inp_unit)! how to estimate rho(M^{-1}A)
|
||||
call read_data(prec%prhovalue,inp_unit) ! if previous is set value, we set it from this one
|
||||
call read_data(prec%pvariant,inp_unit) !
|
||||
call read_data(prec%novr,inp_unit) ! number of overlap layers
|
||||
call read_data(prec%restr,inp_unit) ! restriction over application of AS
|
||||
call read_data(prec%prol,inp_unit) ! prolongation over application of AS
|
||||
@@ -629,8 +607,6 @@ contains
|
||||
call read_data(prec%jsweeps2,inp_unit) ! (post-)smoother sweeps
|
||||
call read_data(prec%degree2,inp_unit) ! (post-)smoother sweeps
|
||||
call read_data(prec%pvariant2,inp_unit) !
|
||||
call read_data(prec%prhovariant2,inp_unit)! how to estimate rho(M^{-1}A)
|
||||
call read_data(prec%prhovalue2,inp_unit) ! if previous is set value, we set it from this one
|
||||
call read_data(prec%novr2,inp_unit) ! number of overlap layers
|
||||
call read_data(prec%restr2,inp_unit) ! restriction over application of AS
|
||||
call read_data(prec%prol2,inp_unit) ! prolongation over application of AS
|
||||
@@ -703,8 +679,6 @@ contains
|
||||
call psb_bcast(ctxt,prec%jsweeps)
|
||||
call psb_bcast(ctxt,prec%degree)
|
||||
call psb_bcast(ctxt,prec%pvariant)
|
||||
call psb_bcast(ctxt,prec%prhovariant)
|
||||
call psb_bcast(ctxt,prec%prhovalue)
|
||||
call psb_bcast(ctxt,prec%novr)
|
||||
call psb_bcast(ctxt,prec%restr)
|
||||
call psb_bcast(ctxt,prec%prol)
|
||||
@@ -719,8 +693,6 @@ contains
|
||||
call psb_bcast(ctxt,prec%jsweeps2)
|
||||
call psb_bcast(ctxt,prec%degree2)
|
||||
call psb_bcast(ctxt,prec%pvariant2)
|
||||
call psb_bcast(ctxt,prec%prhovariant2)
|
||||
call psb_bcast(ctxt,prec%prhovalue2)
|
||||
call psb_bcast(ctxt,prec%novr2)
|
||||
call psb_bcast(ctxt,prec%restr2)
|
||||
call psb_bcast(ctxt,prec%prol2)
|
||||
|
||||
+30
-30
@@ -34,56 +34,56 @@
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
module amg_s_pde2d_poisson_mod
|
||||
module amg_s_pde2d_base_mod
|
||||
use psb_base_mod, only : psb_spk_, szero, sone
|
||||
real(psb_spk_), save, private :: epsilon=sone/80
|
||||
contains
|
||||
subroutine pde_set_parm2d_poisson(dat)
|
||||
subroutine pde_set_parm2d_base(dat)
|
||||
real(psb_spk_), intent(in) :: dat
|
||||
epsilon = dat
|
||||
end subroutine pde_set_parm2d_poisson
|
||||
end subroutine pde_set_parm2d_base
|
||||
!
|
||||
! functions parametrizing the differential equation
|
||||
!
|
||||
function b1_poisson(x,y)
|
||||
function b1_base(x,y)
|
||||
implicit none
|
||||
real(psb_spk_) :: b1_poisson
|
||||
real(psb_spk_) :: b1_base
|
||||
real(psb_spk_), intent(in) :: x,y
|
||||
b1_poisson = szero
|
||||
end function b1_poisson
|
||||
function b2_poisson(x,y)
|
||||
b1_base = szero/1.414_psb_spk_
|
||||
end function b1_base
|
||||
function b2_base(x,y)
|
||||
implicit none
|
||||
real(psb_spk_) :: b2_poisson
|
||||
real(psb_spk_) :: b2_base
|
||||
real(psb_spk_), intent(in) :: x,y
|
||||
b2_poisson = szero
|
||||
end function b2_poisson
|
||||
function c_poisson(x,y)
|
||||
b2_base = szero/1.414_psb_spk_
|
||||
end function b2_base
|
||||
function c_base(x,y)
|
||||
implicit none
|
||||
real(psb_spk_) :: c_poisson
|
||||
real(psb_spk_) :: c_base
|
||||
real(psb_spk_), intent(in) :: x,y
|
||||
c_poisson = szero
|
||||
end function c_poisson
|
||||
function a1_poisson(x,y)
|
||||
c_base = szero
|
||||
end function c_base
|
||||
function a1_base(x,y)
|
||||
implicit none
|
||||
real(psb_spk_) :: a1_poisson
|
||||
real(psb_spk_) :: a1_base
|
||||
real(psb_spk_), intent(in) :: x,y
|
||||
a1_poisson=sone*epsilon
|
||||
end function a1_poisson
|
||||
function a2_poisson(x,y)
|
||||
a1_base=sone*epsilon
|
||||
end function a1_base
|
||||
function a2_base(x,y)
|
||||
implicit none
|
||||
real(psb_spk_) :: a2_poisson
|
||||
real(psb_spk_) :: a2_base
|
||||
real(psb_spk_), intent(in) :: x,y
|
||||
a2_poisson=sone*epsilon
|
||||
end function a2_poisson
|
||||
function g_poisson(x,y)
|
||||
a2_base=sone*epsilon
|
||||
end function a2_base
|
||||
function g_base(x,y)
|
||||
implicit none
|
||||
real(psb_spk_) :: g_poisson
|
||||
real(psb_spk_) :: g_base
|
||||
real(psb_spk_), intent(in) :: x,y
|
||||
g_poisson = szero
|
||||
g_base = szero
|
||||
if (x == sone) then
|
||||
g_poisson = sone
|
||||
g_base = sone
|
||||
else if (x == szero) then
|
||||
g_poisson = sone
|
||||
g_base = sone
|
||||
end if
|
||||
end function g_poisson
|
||||
end module amg_s_pde2d_poisson_mod
|
||||
end function g_base
|
||||
end module amg_s_pde2d_base_mod
|
||||
@@ -70,7 +70,7 @@ program amg_s_pde3d
|
||||
use psb_krylov_mod
|
||||
use psb_util_mod
|
||||
use data_input
|
||||
use amg_s_pde3d_poisson_mod
|
||||
use amg_s_pde3d_base_mod
|
||||
use amg_s_pde3d_exp_mod
|
||||
use amg_s_pde3d_box_mod
|
||||
use amg_s_pde3d_gauss_mod
|
||||
@@ -82,8 +82,7 @@ program amg_s_pde3d
|
||||
|
||||
! input parameters
|
||||
character(len=20) :: kmethd, ptype
|
||||
character(len=5) :: afmt
|
||||
character(len=32) :: pdecoeff
|
||||
character(len=5) :: afmt, pdecoeff
|
||||
integer(psb_ipk_) :: idim
|
||||
integer(psb_epk_) :: system_size
|
||||
|
||||
@@ -148,8 +147,6 @@ program amg_s_pde3d
|
||||
integer(psb_ipk_) :: jsweeps ! (pre-)smoother / 1-lev prec. sweeps
|
||||
integer(psb_ipk_) :: degree ! degree for polynomial smoother
|
||||
character(len=32) :: pvariant ! polynomial variant
|
||||
character(len=32) :: prhovariant ! how to estimate rho(M^{-1}A)
|
||||
real(psb_spk_) :: prhovalue ! if previous is set value, we set it from this one
|
||||
integer(psb_ipk_) :: novr ! number of overlap layers
|
||||
character(len=32) :: restr ! restriction over application of AS
|
||||
character(len=32) :: prol ! prolongation over application of AS
|
||||
@@ -166,8 +163,6 @@ program amg_s_pde3d
|
||||
integer(psb_ipk_) :: jsweeps2 ! post-smoother sweeps
|
||||
integer(psb_ipk_) :: degree2 ! degree for polynomial smoother
|
||||
character(len=32) :: pvariant2 ! polynomial variant
|
||||
character(len=32) :: prhovariant2 ! how to estimate rho(M^{-1}A)
|
||||
real(psb_spk_) :: prhovalue2 ! if previous is set value, we set it from this one
|
||||
integer(psb_ipk_) :: novr2 ! number of overlap layers
|
||||
character(len=32) :: restr2 ! restriction over application of AS
|
||||
character(len=32) :: prol2 ! prolongation over application of AS
|
||||
@@ -251,22 +246,18 @@ program amg_s_pde3d
|
||||
call psb_barrier(ctxt)
|
||||
t1 = psb_wtime()
|
||||
select case(psb_toupper(trim(pdecoeff)))
|
||||
case("POISSON")
|
||||
case("CONST")
|
||||
call amg_gen_pde3d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_poisson,a2_poisson,a3_poisson,&
|
||||
& b1_poisson,b2_poisson,b3_poisson,c_poisson,g_poisson,info)
|
||||
& a1_base,a2_base,a3_base,b1_base,b2_base,b3_base,c_base,g_base,info)
|
||||
case("EXP")
|
||||
call amg_gen_pde3d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_exp,a2_exp,a3_exp,&
|
||||
& b1_exp,b2_exp,b3_exp,c_exp,g_exp,info)
|
||||
& a1_exp,a2_exp,a3_exp,b1_exp,b2_exp,b3_exp,c_exp,g_exp,info)
|
||||
case("BOX")
|
||||
call amg_gen_pde3d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_box,a2_box,a3_box,&
|
||||
& b1_box,b2_box,b3_box,c_box,g_box,info)
|
||||
& a1_box,a2_box,a3_box,b1_box,b2_box,b3_box,c_box,g_box,info)
|
||||
case("GAUSS")
|
||||
call amg_gen_pde3d(ctxt,idim,a,b,x,desc_a,afmt,&
|
||||
& a1_gauss,a2_gauss,a3_gauss,&
|
||||
& b1_gauss,b2_gauss,b3_gauss,c_gauss,g_gauss,info)
|
||||
& a1_gauss,a2_gauss,a3_gauss,b1_gauss,b2_gauss,b3_gauss,c_gauss,g_gauss,info)
|
||||
case default
|
||||
info=psb_err_from_subroutine_
|
||||
ch_err='amg_gen_pdecoeff'
|
||||
@@ -357,12 +348,7 @@ program amg_s_pde3d
|
||||
call prec%set('smoother_sweeps', p_choice%jsweeps, info)
|
||||
call prec%set('poly_degree', p_choice%degree, info)
|
||||
call prec%set('poly_variant', p_choice%pvariant, info)
|
||||
if (p_choice%prhovalue > szero ) then
|
||||
call prec%set('poly_rho_ba', p_choice%prhovalue, info)
|
||||
else
|
||||
call prec%set('poly_rho_estimate', p_choice%prhovariant, info)
|
||||
end if
|
||||
|
||||
|
||||
select case (psb_toupper(p_choice%smther))
|
||||
case ('GS','BWGS','FBGS','JACOBI','L1-JACOBI','L1-FBGS')
|
||||
! do nothing
|
||||
@@ -394,12 +380,6 @@ program amg_s_pde3d
|
||||
call prec%set('smoother_sweeps', p_choice%jsweeps2, info,pos='post')
|
||||
call prec%set('poly_degree', p_choice%degree2, info,pos='post')
|
||||
call prec%set('poly_variant', p_choice%pvariant2, info,pos='post')
|
||||
if (p_choice%prhovalue > szero ) then
|
||||
call prec%set('poly_rho_ba', p_choice%prhovalue2, info,pos='post')
|
||||
else
|
||||
call prec%set('poly_rho_estimate', p_choice%prhovariant2, info,pos='post')
|
||||
end if
|
||||
|
||||
select case (psb_toupper(p_choice%smther2))
|
||||
case ('GS','BWGS','FBGS','JACOBI','L1-JACOBI','L1-FBGS')
|
||||
! do nothing
|
||||
@@ -616,9 +596,7 @@ contains
|
||||
call read_data(prec%smther,inp_unit) ! smoother type
|
||||
call read_data(prec%jsweeps,inp_unit) ! (pre-)smoother / 1-lev prec sweeps
|
||||
call read_data(prec%degree,inp_unit) ! (pre-)smoother / 1-lev prec sweeps
|
||||
call read_data(prec%pvariant,inp_unit) !
|
||||
call read_data(prec%prhovariant,inp_unit)! how to estimate rho(M^{-1}A)
|
||||
call read_data(prec%prhovalue,inp_unit) ! if previous is set value, we set it from this one
|
||||
call read_data(prec%pvariant,inp_unit) !
|
||||
call read_data(prec%novr,inp_unit) ! number of overlap layers
|
||||
call read_data(prec%restr,inp_unit) ! restriction over application of AS
|
||||
call read_data(prec%prol,inp_unit) ! prolongation over application of AS
|
||||
@@ -633,8 +611,6 @@ contains
|
||||
call read_data(prec%jsweeps2,inp_unit) ! (post-)smoother sweeps
|
||||
call read_data(prec%degree2,inp_unit) ! (post-)smoother sweeps
|
||||
call read_data(prec%pvariant2,inp_unit) !
|
||||
call read_data(prec%prhovariant2,inp_unit)! how to estimate rho(M^{-1}A)
|
||||
call read_data(prec%prhovalue2,inp_unit) ! if previous is set value, we set it from this one
|
||||
call read_data(prec%novr2,inp_unit) ! number of overlap layers
|
||||
call read_data(prec%restr2,inp_unit) ! restriction over application of AS
|
||||
call read_data(prec%prol2,inp_unit) ! prolongation over application of AS
|
||||
@@ -707,8 +683,6 @@ contains
|
||||
call psb_bcast(ctxt,prec%jsweeps)
|
||||
call psb_bcast(ctxt,prec%degree)
|
||||
call psb_bcast(ctxt,prec%pvariant)
|
||||
call psb_bcast(ctxt,prec%prhovariant)
|
||||
call psb_bcast(ctxt,prec%prhovalue)
|
||||
call psb_bcast(ctxt,prec%novr)
|
||||
call psb_bcast(ctxt,prec%restr)
|
||||
call psb_bcast(ctxt,prec%prol)
|
||||
@@ -723,8 +697,6 @@ contains
|
||||
call psb_bcast(ctxt,prec%jsweeps2)
|
||||
call psb_bcast(ctxt,prec%degree2)
|
||||
call psb_bcast(ctxt,prec%pvariant2)
|
||||
call psb_bcast(ctxt,prec%prhovariant2)
|
||||
call psb_bcast(ctxt,prec%prhovalue2)
|
||||
call psb_bcast(ctxt,prec%novr2)
|
||||
call psb_bcast(ctxt,prec%restr2)
|
||||
call psb_bcast(ctxt,prec%prol2)
|
||||
|
||||
+38
-38
@@ -34,68 +34,68 @@
|
||||
! ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
! POSSIBILITY OF SUCH DAMAGE.
|
||||
!
|
||||
module amg_s_pde3d_poisson_mod
|
||||
module amg_s_pde3d_base_mod
|
||||
use psb_base_mod, only : psb_spk_, sone, szero
|
||||
real(psb_spk_), save, private :: epsilon=sone/80
|
||||
contains
|
||||
subroutine pde_set_parm3d_poisson(dat)
|
||||
subroutine pde_set_parm3d_base(dat)
|
||||
real(psb_spk_), intent(in) :: dat
|
||||
epsilon = dat
|
||||
end subroutine pde_set_parm3d_poisson
|
||||
end subroutine pde_set_parm3d_base
|
||||
!
|
||||
! functions parametrizing the differential equation
|
||||
!
|
||||
function b1_poisson(x,y,z)
|
||||
function b1_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_spk_) :: b1_poisson
|
||||
real(psb_spk_) :: b1_base
|
||||
real(psb_spk_), intent(in) :: x,y,z
|
||||
b1_poisson=szero
|
||||
end function b1_poisson
|
||||
function b2_poisson(x,y,z)
|
||||
b1_base=szero/sqrt(3.0_psb_spk_)
|
||||
end function b1_base
|
||||
function b2_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_spk_) :: b2_poisson
|
||||
real(psb_spk_) :: b2_base
|
||||
real(psb_spk_), intent(in) :: x,y,z
|
||||
b2_poisson=szero
|
||||
end function b2_poisson
|
||||
function b3_poisson(x,y,z)
|
||||
b2_base=szero/sqrt(3.0_psb_spk_)
|
||||
end function b2_base
|
||||
function b3_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_spk_) :: b3_poisson
|
||||
real(psb_spk_) :: b3_base
|
||||
real(psb_spk_), intent(in) :: x,y,z
|
||||
b3_poisson=szero
|
||||
end function b3_poisson
|
||||
function c_poisson(x,y,z)
|
||||
b3_base=szero/sqrt(3.0_psb_spk_)
|
||||
end function b3_base
|
||||
function c_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_spk_) :: c_poisson
|
||||
real(psb_spk_) :: c_base
|
||||
real(psb_spk_), intent(in) :: x,y,z
|
||||
c_poisson=szero
|
||||
end function c_poisson
|
||||
function a1_poisson(x,y,z)
|
||||
c_base=szero
|
||||
end function c_base
|
||||
function a1_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_spk_) :: a1_poisson
|
||||
real(psb_spk_) :: a1_base
|
||||
real(psb_spk_), intent(in) :: x,y,z
|
||||
a1_poisson=epsilon
|
||||
end function a1_poisson
|
||||
function a2_poisson(x,y,z)
|
||||
a1_base=epsilon
|
||||
end function a1_base
|
||||
function a2_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_spk_) :: a2_poisson
|
||||
real(psb_spk_) :: a2_base
|
||||
real(psb_spk_), intent(in) :: x,y,z
|
||||
a2_poisson=epsilon
|
||||
end function a2_poisson
|
||||
function a3_poisson(x,y,z)
|
||||
a2_base=epsilon
|
||||
end function a2_base
|
||||
function a3_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_spk_) :: a3_poisson
|
||||
real(psb_spk_) :: a3_base
|
||||
real(psb_spk_), intent(in) :: x,y,z
|
||||
a3_poisson=epsilon
|
||||
end function a3_poisson
|
||||
function g_poisson(x,y,z)
|
||||
a3_base=epsilon
|
||||
end function a3_base
|
||||
function g_base(x,y,z)
|
||||
implicit none
|
||||
real(psb_spk_) :: g_poisson
|
||||
real(psb_spk_) :: g_base
|
||||
real(psb_spk_), intent(in) :: x,y,z
|
||||
g_poisson = szero
|
||||
g_base = szero
|
||||
if (x == sone) then
|
||||
g_poisson = sone
|
||||
g_base = sone
|
||||
else if (x == szero) then
|
||||
g_poisson = sone
|
||||
g_base = sone
|
||||
end if
|
||||
end function g_poisson
|
||||
end module amg_s_pde3d_poisson_mod
|
||||
end function g_base
|
||||
end module amg_s_pde3d_base_mod
|
||||
@@ -1,10 +1,8 @@
|
||||
%%%%%%%%%%% General arguments % Lines starting with % are ignored.
|
||||
CSR ! Storage format CSR COO JAD
|
||||
0150 ! IDIM; domain size. Linear system size is IDIM**2
|
||||
POISSON ! PDECOEFF: POISSON, EXP, BOX, GAUSS
|
||||
% ! Coefficients of the PDE
|
||||
CG ! Iterative method:
|
||||
% ! BiCGSTAB BiCGSTABL BiCG CG CGS FCG GCR RGMRES
|
||||
0200 ! IDIM; domain size. Linear system size is IDIM**2
|
||||
CONST ! PDECOEFF: CONST, EXP, BOX, GAUSS Coefficients of the PDE
|
||||
CG ! Iterative method: BiCGSTAB BiCGSTABL BiCG CG CGS FCG GCR RGMRES
|
||||
2 ! ISTOPC
|
||||
00500 ! ITMAX
|
||||
1 ! ITRACE
|
||||
@@ -18,12 +16,11 @@ ML ! Preconditioner type: NONE JACOBI GS FBGS BJAC AS ML
|
||||
FBGS ! Smoother type JACOBI FBGS GS BWGS BJAC AS POLY r 1-level, repeats previous.
|
||||
6 ! Number of sweeps for smoother
|
||||
1 ! degree for polynomial smoother
|
||||
CHEB_4_OPT ! Polynomial variant
|
||||
POLY_LOTTES_BETA ! Polynomial variant
|
||||
% Fields to be added for POLY
|
||||
POLY_RHO_EST_POWER % POLY_RHO_ESTIMATE Currently only POLY_RHO_EST_POWER
|
||||
% POLY_RHO_ESTIMATE Currently only POLY_RHO_EST_POWER
|
||||
% POLY_RHO_ESTIMATE_ITERATIONS default = 20
|
||||
% POLY_RHO_BA set to value
|
||||
1.0
|
||||
% POLY_RHO_BA set to value
|
||||
%
|
||||
0 ! Number of overlap layers for AS preconditioner
|
||||
HALO ! AS restriction operator: NONE HALO
|
||||
@@ -38,11 +35,7 @@ LLK ! AINV variant
|
||||
NONE ! Second (post) smoother, ignored if NONE
|
||||
6 ! Number of sweeps for (post) smoother
|
||||
1 ! degree for polynomial smoother
|
||||
CHEB_4_OPT ! Polynomial variant
|
||||
POLY_RHO_EST_POWER % POLY_RHO_ESTIMATE Currently only POLY_RHO_EST_POWER
|
||||
% POLY_RHO_ESTIMATE_ITERATIONS default = 20
|
||||
% POLY_RHO_BA set to value
|
||||
1.0
|
||||
POLY_LOTTES_BETA ! Polynomial variant
|
||||
0 ! Number of overlap layers for AS preconditioner
|
||||
HALO ! AS restriction operator: NONE HALO
|
||||
NONE ! AS prolongation operator: NONE SUM AVG
|
||||
|
||||
@@ -1,12 +1,10 @@
|
||||
%%%%%%%%%%% General arguments % Lines starting with % are ignored.
|
||||
CSR ! Storage format CSR COO JAD
|
||||
0150 ! IDIM; domain size. Linear system size is IDIM**3
|
||||
POISSON ! PDECOEFF: POISSON, EXP, BOX, GAUSS
|
||||
% ! Coefficients of the PDE
|
||||
CG ! Iterative method:
|
||||
% ! BiCGSTAB BiCGSTABL BiCG CG CGS FCG GCR RGMRES
|
||||
0040 ! IDIM; domain size. Linear system size is IDIM**3
|
||||
CONST ! PDECOEFF: CONST, EXP, GAUSS Coefficients of the PDE
|
||||
CG ! Iterative method: BiCGSTAB BiCGSTABL BiCG CG CGS FCG GCR RGMRES
|
||||
2 ! ISTOPC
|
||||
00500 ! ITMAX
|
||||
00008 ! ITMAX
|
||||
1 ! ITRACE
|
||||
30 ! IRST (restart for RGMRES and BiCGSTABL)
|
||||
1.d-6 ! EPS
|
||||
@@ -15,15 +13,14 @@ ML-VBM-VCYCLE-FBGS-D-BJAC ! Longer descriptive name for preconditioner (up
|
||||
ML ! Preconditioner type: NONE JACOBI GS FBGS BJAC AS ML POLY
|
||||
%
|
||||
%%%%%%%%%%% First smoother (for all levels but coarsest) %%%%%%%%%%%%%%%%
|
||||
POLY ! Smoother type JACOBI FBGS GS BWGS BJAC AS POLY r 1-level, repeats previous.
|
||||
FBGS ! Smoother type JACOBI FBGS GS BWGS BJAC AS POLY r 1-level, repeats previous.
|
||||
6 ! Number of sweeps for smoother
|
||||
6 ! degree for polynomial smoother
|
||||
CHEB_4_OPT ! Polynomial variant
|
||||
1 ! degree for polynomial smoother
|
||||
POLY_LOTTES_BETA ! Polynomial variant
|
||||
% Fields to be added for POLY
|
||||
POLY_RHO_EST_POWER % POLY_RHO_ESTIMATE Currently only POLY_RHO_EST_POWER
|
||||
% POLY_RHO_ESTIMATE Currently only POLY_RHO_EST_POWER
|
||||
% POLY_RHO_ESTIMATE_ITERATIONS default = 20
|
||||
% POLY_RHO_BA set to value
|
||||
1.0
|
||||
% POLY_RHO_BA set to value
|
||||
%
|
||||
0 ! Number of overlap layers for AS preconditioner
|
||||
HALO ! AS restriction operator: NONE HALO
|
||||
@@ -38,11 +35,7 @@ LLK ! AINV variant
|
||||
NONE ! Second (post) smoother, ignored if NONE
|
||||
6 ! Number of sweeps for (post) smoother
|
||||
1 ! degree for polynomial smoother
|
||||
CHEB_4_OPT ! Polynomial variant
|
||||
POLY_RHO_EST_POWER % POLY_RHO_ESTIMATE Currently only POLY_RHO_EST_POWER
|
||||
% POLY_RHO_ESTIMATE_ITERATIONS default = 20
|
||||
% POLY_RHO_BA set to value
|
||||
1.0
|
||||
POLY_LOTTES_BETA ! Polynomial variant
|
||||
0 ! Number of overlap layers for AS preconditioner
|
||||
HALO ! AS restriction operator: NONE HALO
|
||||
NONE ! AS prolongation operator: NONE SUM AVG
|
||||
|
||||
Reference in New Issue
Block a user