Compare commits

...
19 Commits
Author SHA1 Message Date
sfilippone 8f259367de Merge branch 'gpucinterfaces' of github.com:sfilippone/amg4psblas into gpucinterfaces 2026-01-27 14:49:48 +01:00
fdurastante 15d1386b73 Removed debug prints, fixed name of function in error strings 2026-01-27 14:46:35 +01:00
sfilippone 14d24b1d7c Inconsistent C function names 2026-01-27 12:32:46 +01:00
sfilippone 92425e2478 Merge branch 'gpucinterfaces' of github.com:sfilippone/amg4psblas into gpucinterfaces 2026-01-27 12:19:52 +01:00
sfilippone 228f4e46a9 Fix select type 2026-01-27 12:15:14 +01:00
fdurastante 492ae602f2 Added interface for smoother build. Improved options for preconditioner. There is still a memory error. 2026-01-23 14:29:21 +01:00
fdurastante 3230c70308 Standard input file for GPU experiment 2026-01-23 09:40:51 +01:00
fdurastante eb99c74fae Added L1-Jacobi smoother 2026-01-23 08:15:13 +01:00
fdurastante 1c69b2f635 Tester compiling, but misterious CUDA double free to be debugged 2026-01-22 14:35:30 +01:00
sfilippone 86731fe5bb Fix makefiles 2026-01-19 11:46:20 +01:00
fdurastante 7a5ba06622 Added interface for the allocate_wrk member function of a preconditioner. It handles also allocation on the CUDA/GPU side. 2026-01-16 14:08:36 +01:00
sfilippone d4c0428704 Add FCUDEFINES for CBIND 2026-01-16 12:24:22 +01:00
sfilippone 1a969488e3 Merge branch 'gpucinterfaces' of github.com:sfilippone/amg4psblas into gpucinterfaces 2026-01-15 15:27:53 +01:00
Salvatore Filippone 95382fbb01 Fix use of psb_stringc2f 2026-01-15 15:26:11 +01:00
sfilippone f586df1ac3 Fix use of stringc2f 2026-01-15 11:40:36 +01:00
Salvatore Filippone bab8a27962 Merge branch 'development' of github.com:sfilippone/amg4psblas into development 2026-01-13 17:45:52 +01:00
sfilippone 28ccefa4bb Merge branch 'development' of github.com:sfilippone/amg4psblas into development 2026-01-13 17:43:25 +01:00
sfilippone 3f33b2ce71 Update docs 2026-01-13 17:41:47 +01:00
Salvatore Filippone 7279347055 Doc fixes 2025-12-30 19:42:01 +01:00
22 changed files with 1073 additions and 45 deletions
+3 -2
View File
@@ -58,7 +58,8 @@ MODOBJS=amg_base_prec_type.o amg_prec_type.o amg_prec_mod.o \
$(SMODOBJS) $(DMODOBJS) $(CMODOBJS) $(ZMODOBJS)
LOCAL_MODS=$(MODOBJS:.o=$(.mod))
LOCAL_MODS=$(MODOBJS:.o=$(.mod)) amg_c_l1_diag_solver$(.mod) amg_s_l1_diag_solver$(.mod) \
amg_d_l1_diag_solver$(.mod) amg_z_l1_diag_solver$(.mod)
LIBNAME=libamg_prec.a
all: mods objs impld
@@ -70,7 +71,7 @@ objs: mods impld
impld: mods
cd impl && $(MAKE)
lib: mods impld
lib: objs
cd impl && $(MAKE) lib
$(AR) $(HERE)/$(LIBNAME) $(MODOBJS)
$(RANLIB) $(HERE)/$(LIBNAME)
@@ -74,7 +74,7 @@ subroutine amg_c_as_smoother_clone(sm,smout,info)
if (info == psb_success_) &
& call sm%desc_data%clone(smo%desc_data,info)
if ((info==psb_success_).and.(allocated(sm%sv))) then
allocate(smout%sv,mold=sm%sv,stat=info)
allocate(smo%sv,mold=sm%sv,stat=info)
if (info == psb_success_) call sm%sv%clone(smo%sv,info)
end if
@@ -74,7 +74,7 @@ subroutine amg_c_jac_smoother_clone(sm,smout,info)
smo%tol = sm%tol
call sm%nd%clone(smo%nd,info)
if ((info==psb_success_).and.(allocated(sm%sv))) then
allocate(smout%sv,mold=sm%sv,stat=info)
allocate(smo%sv,mold=sm%sv,stat=info)
if (info == psb_success_) call sm%sv%clone(smo%sv,info)
end if
@@ -74,7 +74,7 @@ subroutine amg_d_as_smoother_clone(sm,smout,info)
if (info == psb_success_) &
& call sm%desc_data%clone(smo%desc_data,info)
if ((info==psb_success_).and.(allocated(sm%sv))) then
allocate(smout%sv,mold=sm%sv,stat=info)
allocate(smo%sv,mold=sm%sv,stat=info)
if (info == psb_success_) call sm%sv%clone(smo%sv,info)
end if
@@ -74,7 +74,7 @@ subroutine amg_d_jac_smoother_clone(sm,smout,info)
smo%tol = sm%tol
call sm%nd%clone(smo%nd,info)
if ((info==psb_success_).and.(allocated(sm%sv))) then
allocate(smout%sv,mold=sm%sv,stat=info)
allocate(smo%sv,mold=sm%sv,stat=info)
if (info == psb_success_) call sm%sv%clone(smo%sv,info)
end if
@@ -74,7 +74,7 @@ subroutine amg_s_as_smoother_clone(sm,smout,info)
if (info == psb_success_) &
& call sm%desc_data%clone(smo%desc_data,info)
if ((info==psb_success_).and.(allocated(sm%sv))) then
allocate(smout%sv,mold=sm%sv,stat=info)
allocate(smo%sv,mold=sm%sv,stat=info)
if (info == psb_success_) call sm%sv%clone(smo%sv,info)
end if
@@ -74,7 +74,7 @@ subroutine amg_s_jac_smoother_clone(sm,smout,info)
smo%tol = sm%tol
call sm%nd%clone(smo%nd,info)
if ((info==psb_success_).and.(allocated(sm%sv))) then
allocate(smout%sv,mold=sm%sv,stat=info)
allocate(smo%sv,mold=sm%sv,stat=info)
if (info == psb_success_) call sm%sv%clone(smo%sv,info)
end if
@@ -74,7 +74,7 @@ subroutine amg_z_as_smoother_clone(sm,smout,info)
if (info == psb_success_) &
& call sm%desc_data%clone(smo%desc_data,info)
if ((info==psb_success_).and.(allocated(sm%sv))) then
allocate(smout%sv,mold=sm%sv,stat=info)
allocate(smo%sv,mold=sm%sv,stat=info)
if (info == psb_success_) call sm%sv%clone(smo%sv,info)
end if
@@ -74,7 +74,7 @@ subroutine amg_z_jac_smoother_clone(sm,smout,info)
smo%tol = sm%tol
call sm%nd%clone(smo%nd,info)
if ((info==psb_success_).and.(allocated(sm%sv))) then
allocate(smout%sv,mold=sm%sv,stat=info)
allocate(smo%sv,mold=sm%sv,stat=info)
if (info == psb_success_) call sm%sv%clone(smo%sv,info)
end if
+3
View File
@@ -7,6 +7,8 @@ INCDIR=../include
MODDIR=../modules/
LIBNAME=$(CBINDLIBNAME)
LIBNAME=libamg_cbind.a
AMGFDEFINES=-DPSB_HAVE_LAPACK -DPSB_HAVE_FLUSH_STMT -DPSB_MPI_MOD $(PSBFDEFINES) $(FCUDEFINES)
FDEFINES=$(AMGFDEFINES)
objs: amgprecd
@@ -26,3 +28,4 @@ clean:
veryclean: clean
cd test/pargen && $(MAKE) clean
/bin/rm -f $(HERE)/$(LIBNAME) $(LIBMOD) *$(.mod) *.h
+2
View File
@@ -8,6 +8,8 @@ DEST=../
CINCLUDES=-I. -I$(INCDIR) -I$(PSBLAS_INCDIR)
FINCLUDES=$(FMFLAG)$(HERE) $(FMFLAG)$(INCDIR) $(FMFLAG)$(MODDIR) $(PSBLAS_INCLUDES)
AMGFDEFINES=-DPSB_HAVE_LAPACK -DPSB_HAVE_FLUSH_STMT -DPSB_MPI_MOD $(PSBFDEFINES) $(FCUDEFINES)
FDEFINES=$(AMGFDEFINES)
OBJS=amg_prec_cbind_mod.o amg_dprec_cbind_mod.o amg_c_dprec.o amg_zprec_cbind_mod.o amg_c_zprec.o
+2
View File
@@ -26,12 +26,14 @@ extern "C" {
psb_i_t amg_c_dprecbld(psb_c_dspmat *ah, psb_c_descriptor *cdh, amg_c_dprec *ph);
psb_i_t amg_c_dhierarchy_build(psb_c_dspmat *ah, psb_c_descriptor *cdh, amg_c_dprec *ph);
psb_i_t amg_c_dsmoothers_build(psb_c_dspmat *ah, psb_c_descriptor *cdh, amg_c_dprec *ph);
psb_i_t amg_c_dsmoothers_build_opt(psb_c_dspmat *ah, psb_c_descriptor *cdh, amg_c_dprec *ph, const char *afmt, const char *chfmt);
psb_i_t amg_c_dprecapply(amg_c_dprec *ph, psb_c_dvector *x, psb_c_dvector *b, psb_c_descriptor *cdh);
psb_i_t amg_c_dprecapply_opt(amg_c_dprec *ph, psb_c_dvector *x, psb_c_dvector *b, psb_c_descriptor *cdh, const char *ctrans);
psb_i_t amg_c_dprecfree(amg_c_dprec *ph);
psb_i_t amg_c_dprecbld_opt(psb_c_dspmat *ah, psb_c_descriptor *cdh,
amg_c_dprec *ph, const char *afmt);
psb_i_t amg_c_ddescr(amg_c_dprec *ph);
psb_i_t amg_c_dallocate_wrk(amg_c_dprec *ph, const char *chfmt);
psb_i_t amg_c_dkrylov(const char *method, psb_c_dspmat *ah, amg_c_dprec *ph,
psb_c_dvector *bh, psb_c_dvector *xh,
+16 -13
View File
@@ -10,15 +10,17 @@
/* Note: amg_get_XXX_handle returns: <= 0 unsuccessful */
/* >0 valid handle */
#ifdef __cplusplus
extern "C" {
extern "C"
{
#endif
typedef struct AMG_C_ZPREC {
typedef struct AMG_C_ZPREC
{
void *dprec;
} amg_c_zprec;
amg_c_zprec* amg_c_zprec_new();
psb_i_t amg_c_zprec_delete(amg_c_zprec* p);
} amg_c_zprec;
amg_c_zprec *amg_c_zprec_new();
psb_i_t amg_c_zprec_delete(amg_c_zprec *p);
psb_i_t amg_c_zprecinit(psb_c_ctxt cctxt, amg_c_zprec *ph, const char *ptype);
psb_i_t amg_c_zprecseti(amg_c_zprec *ph, const char *what, psb_i_t val);
psb_i_t amg_c_zprecsetc(amg_c_zprec *ph, const char *what, const char *val);
@@ -26,18 +28,19 @@ extern "C" {
psb_i_t amg_c_zprecbld(psb_c_zspmat *ah, psb_c_descriptor *cdh, amg_c_zprec *ph);
psb_i_t amg_c_zhierarchy_build(psb_c_zspmat *ah, psb_c_descriptor *cdh, amg_c_zprec *ph);
psb_i_t amg_c_zsmoothers_build(psb_c_zspmat *ah, psb_c_descriptor *cdh, amg_c_zprec *ph);
psb_i_t amg_c_zsmoothers_build_opt(psb_c_zspmat *ah, psb_c_descriptor *cdh, amg_c_zprec *ph, const char *afmt, const char *chfmt);
psb_i_t amg_c_zprecapply(amg_c_zprec *ph, psb_c_zvector *x, psb_c_zvector *b, psb_c_descriptor *cdh);
psb_i_t amg_c_zprecapply_opt(amg_c_zprec *ph, psb_c_zvector *x, psb_c_zvector *b, psb_c_descriptor *cdh, const char *ctrans);
psb_i_t amg_c_zprecfree(amg_c_zprec *ph);
psb_i_t amg_c_zprecbld_opt(psb_c_zspmat *ah, psb_c_descriptor *cdh,
amg_c_zprec *ph, const char *afmt);
psb_i_t amg_c_zprecbld_opt(psb_c_zspmat *ah, psb_c_descriptor *cdh,
amg_c_zprec *ph, const char *afmt);
psb_i_t amg_c_zdescr(amg_c_zprec *ph);
psb_i_t amg_c_zallocate_wrk(amg_c_zprec *ph, const char *chfmt);
psb_i_t amg_c_zkrylov(const char *method, psb_c_zspmat *ah, amg_c_zprec *ph,
psb_c_zvector *bh, psb_c_zvector *xh,
psb_c_descriptor *cdh, psb_c_SolverOptions *opt);
psb_i_t amg_c_zkrylov(const char *method, psb_c_zspmat *ah, amg_c_zprec *ph,
psb_c_zvector *bh, psb_c_zvector *xh,
psb_c_descriptor *cdh, psb_c_SolverOptions *opt);
#ifdef __cplusplus
}
+172 -8
View File
@@ -43,7 +43,7 @@ contains
ph%item = c_loc(precp)
call stringc2f(ptype,fptype)
call psb_stringc2f(ptype,fptype)
call precp%init(psb_c2f_ctxt(cctxt),fptype,iret)
@@ -70,7 +70,7 @@ contains
return
end if
call stringc2f(what,fwhat)
call psb_stringc2f(what,fwhat)
call precp%set(fwhat,val,iret)
@@ -98,7 +98,7 @@ contains
return
end if
call stringc2f(what,fwhat)
call psb_stringc2f(what,fwhat)
call precp%set(fwhat,val,iret)
@@ -124,8 +124,8 @@ contains
return
end if
call stringc2f(what,fwhat)
call stringc2f(val,fval)
call psb_stringc2f(what,fwhat)
call psb_stringc2f(val,fval)
call precp%set(fwhat,fval,iret)
@@ -245,6 +245,120 @@ contains
return
end function amg_c_dsmoothers_build
function amg_c_dsmoothers_build_opt(ah,cdh,ph,afmt,cdfmt) bind(c) result(res)
#if defined (PSB_HAVE_CUDA)
use psb_cuda_mod
#endif
implicit none
integer(psb_c_ipk_) :: res
type(psb_c_object_type) :: ph,ah,cdh
type(amg_dprec_type), pointer :: precp
type(psb_dspmat_type), pointer :: ap
type(psb_desc_type), pointer :: descp
character(c_char) :: afmt(*), cdfmt(*)
character(len=80) :: fptype
integer(psb_ipk_) :: iret
! Local variables for formats
character(len=10) :: fafmt, fcdfmt
#if defined (PSB_HAVE_CUDA)
type(psb_d_vect_cuda), target :: dvgpu
type(psb_i_vect_cuda), target :: ivgpu
! GPU matrix molds
type(psb_d_cuda_hlg_sparse_mat), target :: ahlg
type(psb_d_cuda_hdiag_sparse_mat), target :: ahdiag
type(psb_d_cuda_csrg_sparse_mat), target :: acsrg
type(psb_d_cuda_elg_sparse_mat), target :: aelg
#endif
type(psb_d_base_vect_type), target :: dvhost
type(psb_i_base_vect_type), target :: ivhost
! CPU matrix molds
type(psb_d_ell_sparse_mat), target :: aell
type(psb_d_csr_sparse_mat), target :: acsr
type(psb_d_coo_sparse_mat), target :: acoo
type(psb_d_hll_sparse_mat), target :: ahll
type(psb_d_hdia_sparse_mat), target :: ahdia
type(psb_d_dns_sparse_mat), target :: adns
! molding variables
class(psb_d_base_vect_type), pointer :: vmold
class(psb_d_base_sparse_mat), pointer :: amold
class(psb_i_base_vect_type), pointer :: imold
res = -1
if (c_associated(cdh%item)) then
call c_f_pointer(cdh%item,descp)
else
return
end if
if (c_associated(ah%item)) then
call c_f_pointer(ah%item,ap)
else
return
end if
if (c_associated(ph%item)) then
call c_f_pointer(ph%item,precp)
else
return
end if
! Convert formats
call psb_stringc2f(afmt,fafmt)
call psb_stringc2f(cdfmt,fcdfmt)
! Select matrix mold
select case (psb_toupper(fafmt))
#if defined (PSB_HAVE_CUDA)
case('CSRG')
amold => acsrg
case('ELG')
amold => aelg
case('HLG')
amold => ahlg
case('HDIAG')
amold => ahdiag
#endif
case('CSR')
amold => acsr
case('ELL')
amold => aell
case('COO')
amold => acoo
case('HLL')
amold => ahll
case('HDIA')
amold => ahdia
case('DNS')
amold => adns
case default
write(psb_err_unit,'(A)') 'amg_c_dsmoothers_build_format: Unknown format ', fafmt, ' defaulting to CSR'
amold => acsr
end select
! Select vector mold
select case (psb_toupper(fcdfmt))
#if defined (PSB_HAVE_CUDA)
case('GPU','DEVICE')
vmold => dvgpu
imold => ivgpu
#endif
case('HOST','CPU')
vmold => dvhost
imold => ivhost
case default
write(psb_err_unit,'(A)') 'amg_c_dsmoothers_build_format: Unknown format ', fcdfmt, ' defaulting to HOST/CPU'
vmold => dvhost
imold => ivhost
end select
call precp%smoothers_build(ap,descp,iret,amold=amold,vmold=vmold,imold=imold)
res = MLDC_ERR_FILTER(iret)
MLDC_ERR_HANDLE(res)
return
end function amg_c_dsmoothers_build_opt
function amg_c_dkrylov(methd,&
& ah,ph,bh,xh,cdh,options) bind(c) result(res)
use psb_base_mod
@@ -273,7 +387,6 @@ contains
use psb_linsolve_mod
use psb_objhandle_mod
use psb_prec_cbind_mod
use psb_base_string_cbind_mod
implicit none
integer(psb_c_ipk_) :: res
type(psb_c_object_type) :: ah,cdh,ph,bh,xh
@@ -319,7 +432,7 @@ contains
end if
call stringc2f(methd,fmethd)
call psb_stringc2f(methd,fmethd)
feps = eps
fitmax = itmax
fitrace = itrace
@@ -431,7 +544,7 @@ end function amg_c_dprecapply
end if
! Convert transpose flag
call stringc2f(ctrans,ftrans)
call psb_stringc2f(ctrans,ftrans)
! Apply preconditioner
call precp%apply(bp,xp,descp,info,trans=ftrans)
@@ -492,4 +605,55 @@ end function amg_c_dprecapply_opt
return
end function amg_c_ddescr
function amg_c_dallocate_wrk(ph,chfmt) bind(c, name="amg_c_dallocate_wrk") result(res)
#if defined (PSB_HAVE_CUDA)
use psb_cuda_mod
#endif
implicit none
integer(psb_c_ipk_) :: res
type(psb_c_object_type) :: ph
character(c_char) :: chfmt(*)
integer(psb_ipk_) :: iret
type(amg_dprec_type), pointer :: precp
character(len=6) :: fchfmt
! Local variable
integer(psb_ipk_) :: info
! Local mold variables
#if defined (PSB_HAVE_CUDA)
type(psb_d_vect_cuda), target :: dvgpu
#endif
type(psb_d_base_vect_type), target :: dvhost
class(psb_d_base_vect_type), pointer :: vmold
res = -1
if (c_associated(ph%item)) then
call c_f_pointer(ph%item,precp)
else
return
end if
call psb_stringc2f(chfmt,fchfmt)
select case (psb_toupper(fchfmt))
case('HOST','CPU')
vmold => dvhost
#if defined (PSB_HAVE_CUDA)
case('GPU','DEVICE')
vmold => dvgpu
#endif
case default
write(psb_err_unit,'(A)') 'amg_c_dallocate_wrk: Unknown format ', fchfmt, ' defaulting to HOST/CPU'
vmold => dvhost
end select
call precp%allocate_wrk(info,vmold=vmold)
iret = info
res = MLDC_ERR_FILTER(iret)
MLDC_ERR_HANDLE(res)
return
end function amg_c_dallocate_wrk
end module amg_dprec_cbind_mod
+166 -8
View File
@@ -44,7 +44,7 @@ contains
ph%item = c_loc(precp)
call stringc2f(ptype,fptype)
call psb_stringc2f(ptype,fptype)
call precp%init(psb_c2f_ctxt(cctxt),fptype,iret)
@@ -71,7 +71,7 @@ contains
return
end if
call stringc2f(what,fwhat)
call psb_stringc2f(what,fwhat)
call precp%set(fwhat,val,iret)
@@ -99,7 +99,7 @@ contains
return
end if
call stringc2f(what,fwhat)
call psb_stringc2f(what,fwhat)
call precp%set(fwhat,val,iret)
@@ -125,8 +125,8 @@ contains
return
end if
call stringc2f(what,fwhat)
call stringc2f(val,fval)
call psb_stringc2f(what,fwhat)
call psb_stringc2f(val,fval)
call precp%set(fwhat,fval,iret)
@@ -246,6 +246,114 @@ contains
return
end function amg_c_zsmoothers_build
function amg_c_zsmoothers_build_format(ah,cdh,ph,afmt,cdfmt) bind(c) result(res)
#if defined (PSB_HAVE_CUDA)
use psb_cuda_mod
#endif
implicit none
integer(psb_c_ipk_) :: res
type(psb_c_object_type) :: ph,ah,cdh
type(amg_zprec_type), pointer :: precp
type(psb_zspmat_type), pointer :: ap
type(psb_desc_type), pointer :: descp
character(c_char) :: afmt(*), cdfmt(*)
character(len=80) :: fptype
integer(psb_ipk_) :: iret
! Local variables for formats
character(len=10) :: fafmt, fcdfmt
#if defined (PSB_HAVE_CUDA)
type(psb_z_vect_cuda), target :: dvgpu
type(psb_i_vect_cuda), target :: ivgpu
! GPU matrix molds
type(psb_z_cuda_hlg_sparse_mat), target :: ahlg
type(psb_z_cuda_csrg_sparse_mat), target :: acsrg
type(psb_z_cuda_elg_sparse_mat), target :: aelg
#endif
type(psb_z_base_vect_type), target :: dvhost
type(psb_i_base_vect_type), target :: ivhost
! CPU matrix molds
type(psb_z_ell_sparse_mat), target :: aell
type(psb_z_csr_sparse_mat), target :: acsr
type(psb_z_coo_sparse_mat), target :: acoo
type(psb_z_hll_sparse_mat), target :: ahll
type(psb_z_dns_sparse_mat), target :: adns
! molding variables
class(psb_z_base_vect_type), pointer :: vmold
class(psb_z_base_sparse_mat), pointer :: amold
class(psb_i_base_vect_type), pointer :: imold
res = -1
if (c_associated(cdh%item)) then
call c_f_pointer(cdh%item,descp)
else
return
end if
if (c_associated(ah%item)) then
call c_f_pointer(ah%item,ap)
else
return
end if
if (c_associated(ph%item)) then
call c_f_pointer(ph%item,precp)
else
return
end if
! Convert formats
call psb_stringc2f(afmt,fafmt)
call psb_stringc2f(cdfmt,fcdfmt)
! Select matrix mold
select case (psb_toupper(fafmt))
#if defined (PSB_HAVE_CUDA)
case('CSRG')
amold => acsrg
case('ELG')
amold => aelg
case('HLG')
amold => ahlg
#endif
case('CSR')
amold => acsr
case('ELL')
amold => aell
case('COO')
amold => acoo
case('HLL')
amold => ahll
case('DNS')
amold => adns
case default
write(psb_err_unit,'(A)') 'amg_c_zsmoothers_build_format: Unknown format ', fafmt, ' defaulting to CSR'
amold => acsr
end select
! Select vector mold
select case (psb_toupper(fcdfmt))
#if defined (PSB_HAVE_CUDA)
case('GPU','DEVICE')
vmold => dvgpu
imold => ivgpu
#endif
case('HOST','CPU')
vmold => dvhost
imold => ivhost
case default
write(psb_err_unit,'(A)') 'amg_c_zsmoothers_build_format: Unknown format ', fcdfmt, ' defaulting to HOST/CPU'
vmold => dvhost
imold => ivhost
end select
call precp%smoothers_build(ap,descp,iret,amold=amold,vmold=vmold,imold=imold)
res = MLDC_ERR_FILTER(iret)
MLDC_ERR_HANDLE(res)
return
end function amg_c_zsmoothers_build_format
function amg_c_zkrylov(methd,&
& ah,ph,bh,xh,cdh,options) bind(c) result(res)
use psb_linsolve_mod
@@ -269,7 +377,6 @@ contains
& ah,ph,bh,xh,eps,cdh,itmax,iter,err,itrace,irst,istop) bind(c) result(res)
use psb_linsolve_mod
use psb_objhandle_mod
use psb_base_string_cbind_mod
implicit none
integer(psb_c_ipk_) :: res
type(psb_c_object_type) :: ah,cdh,ph,bh,xh
@@ -315,7 +422,7 @@ contains
end if
call stringc2f(methd,fmethd)
call psb_stringc2f(methd,fmethd)
feps = eps
fitmax = itmax
fitrace = itrace
@@ -427,7 +534,7 @@ contains
end if
! Convert transpose flag
call stringc2f(ctrans,ftrans)
call psb_stringc2f(ctrans,ftrans)
! Apply preconditioner
call precp%apply(bp,xp,descp,info,trans=ftrans)
@@ -488,4 +595,55 @@ contains
return
end function amg_c_zdescr
function amg_c_zallocate_wrk(ph,chfmt) bind(c, name="amg_c_zallocate_wrk") result(res)
#if defined (PSB_HAVE_CUDA)
use psb_cuda_mod
#endif
implicit none
integer(psb_c_ipk_) :: res
type(psb_c_object_type) :: ph
character(c_char) :: chfmt(*)
integer(psb_ipk_) :: iret
type(amg_zprec_type), pointer :: precp
character(len=6) :: fchfmt
! Local variable
integer(psb_ipk_) :: info
! Local mold variables
#if defined (PSB_HAVE_CUDA)
type(psb_z_vect_cuda), target :: zvgpu
#endif
type(psb_z_base_vect_type), target :: zvhost
class(psb_z_base_vect_type), pointer :: vmold
res = -1
if (c_associated(ph%item)) then
call c_f_pointer(ph%item,precp)
else
return
end if
call psb_stringc2f(chfmt,fchfmt)
select case (psb_toupper(fchfmt))
case('HOST','CPU')
vmold => zvhost
#if defined (PSB_HAVE_CUDA)
case('GPU','DEVICE')
vmold => zvgpu
#endif
case default
write(psb_err_unit,'(A)') 'amg_c_zallocate_wrk: Unknown format ', fchfmt, ' defaulting to HOST/CPU'
vmold => zvhost
end select
call precp%allocate_wrk(info,vmold=vmold)
iret = info
res = MLDC_ERR_FILTER(iret)
MLDC_ERR_HANDLE(res)
return
end function amg_c_zallocate_wrk
end module amg_zprec_cbind_mod
+9 -3
View File
@@ -23,16 +23,21 @@ EXEDIR=./runs
#UMFLIBS=-lumfpack -lamd -lcholmod -lcolamd -lcamd -lccolamd -L/usr/include/suitesparse
#UMFFLAGS=-DHave_UMF_ -I/usr/include/suitesparse
all: amgec
all: amgec amgecgpu
amgec: amgec.o
$(MPFC) amgec.o -o amgec $(AMGC_LIBS) $(PSBC_LIBS) $(PSBCLDLIBS) $(PSBLAS_LIBS) \
$(UMFLIBS) $(PSBLDLIBS) $(LDLIBS) -lm -lgfortran
$(UMFLIBS) $(PSBLDLIBS) $(AMGLDLIBS) $(PSBGPULDLIBS) $(LDLIBS) -lm -lgfortran -fopenmp
# \
# -lifcore -lifcoremt -lguide -limf -lirc -lintlc -lcxaguard -L/opt/intel/fc/10.0.023/lib/ -lm
/bin/mv amgec $(EXEDIR)
amgecgpu: amgecgpu.o
$(MPFC) amgecgpu.o -o amgecgpu $(AMGC_LIBS) $(PSBC_LIBS) $(PSBCLDLIBS) $(PSBLAS_LIBS) \
$(UMFLIBS) $(PSBLDLIBS) $(AMGLDLIBS) $(PSBGPULDLIBS) $(LDLIBS) -lm -lgfortran -fopenmp
/bin/mv amgecgpu $(EXEDIR)
.f90.o:
$(MPFC) $(F90COPT) $(FINCLUDES) $(FDEFINES) -c $<
.c.o:
@@ -40,7 +45,7 @@ amgec: amgec.o
clean:
/bin/rm -f amgec.o $(EXEDIR)/amgec
/bin/rm -f amgec.o $(EXEDIR)/amgec $(EXEDIR)/amgecgpu
verycleanlib:
(cd ../..; make veryclean)
lib:
@@ -48,5 +53,6 @@ lib:
tests: all
cd runs ; ./amgec < amge.inp
cd runs ; ./amgecgpu < amgegpu.inp
+679
View File
@@ -0,0 +1,679 @@
/*----------------------------------------------------------------------------------*/
/* Parallel Sparse BLAS v2.2 */
/* (C) Copyright 2007 Salvatore Filippone University of Rome Tor Vergata */
/* */
/* Redistribution and use in source and binary forms, with or without */
/* modification, are permitted provided that the following conditions */
/* are met: */
/* 1. Redistributions of source code must retain the above copyright */
/* notice, this list of conditions and the following disclaimer. */
/* 2. Redistributions in binary form must reproduce the above copyright */
/* notice, this list of conditions, and the following disclaimer in the */
/* documentation and/or other materials provided with the distribution. */
/* 3. The name of the PSBLAS group or the names of its contributors may */
/* not be used to endorse or promote products derived from this */
/* software without specific written permission. */
/* */
/* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS */
/* ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED */
/* TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR */
/* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE PSBLAS GROUP OR ITS CONTRIBUTORS */
/* BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR */
/* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF */
/* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS */
/* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN */
/* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) */
/* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE */
/* POSSIBILITY OF SUCH DAMAGE. */
/* */
/* */
/* File: ppdec.c */
/* */
/* Program: ppdec */
/* This sample program shows how to build and solve a sparse linear */
/* */
/* The program solves a linear system based on the partial differential */
/* equation */
/* */
/* */
/* */
/* The equation generated is */
/* */
/* b1 d d (u) b2 d d (u) a1 d (u)) a2 d (u))) */
/* - ------ - ------ + ----- + ------ + a3 u = 0 */
/* dx dx dy dy dx dy */
/* */
/* */
/* with Dirichlet boundary conditions on the unit cube */
/* */
/* 0<=x,y,z<=1 */
/* */
/* The equation is discretized with finite differences and uniform stepsize; */
/* the resulting discrete equation is */
/* */
/* ( u(x,y,z)(2b1+2b2+a1+a2)+u(x-1,y)(-b1-a1)+u(x,y-1)(-b2-a2)+ */
/* -u(x+1,y)b1-u(x,y+1)b2)*(1/h**2) */
/* */
/* Example adapted from: C.T.Kelley */
/* Iterative Methods for Linear and Nonlinear Equations */
/* SIAM 1995 */
/* */
/* */
/* In this sample program the index space of the discretized */
/* computational domain is first numbered sequentially in a standard way, */
/* then the corresponding vector is distributed according to an HPF BLOCK */
/* distribution directive. */
/* */
/* Boundary conditions are set in a very simple way, by adding */
/* equations of the form */
/* */
/* u(x,y) = rhs(x,y) */
/* */
/*----------------------------------------------------------------------------------*/
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <math.h>
#include "psb_base_cbind.h"
#include "amg_cbind.h"
double a1(double x, double y, double z)
{
return (1.0 / 80.0);
}
double a2(double x, double y, double z)
{
return (1.0 / 80.0);
}
double a3(double x, double y, double z)
{
return (1.0 / 80.0);
}
double c(double x, double y, double z)
{
return (0.0);
}
double b1(double x, double y, double z)
{
return (0.0 / sqrt(3.0));
}
double b2(double x, double y, double z)
{
return (0.0 / sqrt(3.0));
}
double b3(double x, double y, double z)
{
return (0.0 / sqrt(3.0));
}
double g(double x, double y, double z)
{
if (x == 1.0)
{
return (1.0);
}
else if (x == 0.0)
{
return (exp(-y * y - z * z));
}
else
{
return (0.0);
}
}
#define NBMAX 20
psb_i_t matgen(psb_c_ctxt cctxt, psb_i_t nl, psb_i_t idim, psb_l_t vl[],
psb_c_dspmat *ah, const char *afmt,
psb_c_descriptor *cdh, const char *cdfmt,
psb_c_dvector *xh, psb_c_dvector *bh, psb_c_dvector *rh)
{
psb_i_t iam, np;
psb_l_t ix, iy, iz, el, glob_row;
psb_i_t i, k, info, ret;
double x, y, z, deltah, sqdeltah, deltah2;
double val[10 * NBMAX], zt[NBMAX];
psb_l_t irow[10 * NBMAX], icol[10 * NBMAX];
info = 0;
psb_c_info(cctxt, &iam, &np);
if (iam == 0)
{
fprintf(stdout, "Starting matrix generation: local number of rows %d\n", nl);
fprintf(stdout, "Matrix format: %s\n", afmt);
fprintf(stdout, "Descriptor format: %s\n", cdfmt);
fflush(stdout);
}
fprintf(stderr, "Matrix generation: local number of rows %d on process %d\n", nl, iam);
deltah = (double)1.0 / (idim + 1);
sqdeltah = deltah * deltah;
deltah2 = 2.0 * deltah;
psb_c_set_index_base(0);
for (i = 0; i < nl; i++)
{
glob_row = vl[i];
// if ((i%100000 == 0)||(i<10)) fprintf(stderr,"%d: generation loop at %d %ld \n",iam,i,glob_row);
el = 0;
ix = glob_row / (idim * idim);
iy = (glob_row - ix * idim * idim) / idim;
iz = glob_row - ix * idim * idim - iy * idim;
x = (ix + 1) * deltah;
y = (iy + 1) * deltah;
z = (iz + 1) * deltah;
zt[0] = 0.0;
/* internal point: build discretization */
/* term depending on (x-1,y,z) */
val[el] = -a1(x, y, z) / sqdeltah - b1(x, y, z) / deltah2;
if (ix == 0)
{
zt[0] += g(0.0, y, z) * (-val[el]);
}
else
{
icol[el] = (ix - 1) * idim * idim + (iy)*idim + (iz);
el = el + 1;
}
/* term depending on (x,y-1,z) */
val[el] = -a2(x, y, z) / sqdeltah - b2(x, y, z) / deltah2;
if (iy == 0)
{
zt[0] += g(x, 0.0, z) * (-val[el]);
}
else
{
icol[el] = (ix)*idim * idim + (iy - 1) * idim + (iz);
el = el + 1;
}
/* term depending on (x,y,z-1)*/
val[el] = -a3(x, y, z) / sqdeltah - b3(x, y, z) / deltah2;
if (iz == 0)
{
zt[0] += g(x, y, 0.0) * (-val[el]);
}
else
{
icol[el] = (ix)*idim * idim + (iy)*idim + (iz - 1);
el = el + 1;
}
/* term depending on (x,y,z)*/
val[el] = 2.0 * (a1(x, y, z) + a2(x, y, z) + a3(x, y, z)) / sqdeltah + c(x, y, z);
icol[el] = (ix)*idim * idim + (iy)*idim + (iz);
el = el + 1;
/* term depending on (x,y,z+1) */
val[el] = -a3(x, y, z) / sqdeltah + b3(x, y, z) / deltah2;
if (iz == idim - 1)
{
zt[0] += g(x, y, 1.0) * (-val[el]);
}
else
{
icol[el] = (ix)*idim * idim + (iy)*idim + (iz + 1);
el = el + 1;
}
/* term depending on (x,y+1,z) */
val[el] = -a2(x, y, z) / sqdeltah + b2(x, y, z) / deltah2;
if (iy == idim - 1)
{
zt[0] += g(x, 1.0, z) * (-val[el]);
}
else
{
icol[el] = (ix)*idim * idim + (iy + 1) * idim + (iz);
el = el + 1;
}
/* term depending on (x+1,y,z) */
val[el] = -a1(x, y, z) / sqdeltah + b1(x, y, z) / deltah2;
if (ix == idim - 1)
{
zt[0] += g(1.0, y, z) * (-val[el]);
}
else
{
icol[el] = (ix + 1) * idim * idim + (iy)*idim + (iz);
el = el + 1;
}
for (k = 0; k < el; k++)
irow[k] = glob_row;
if ((ret = psb_c_dspins(el, irow, icol, val, ah, cdh)) != 0)
fprintf(stderr, "From psb_c_dspins: %d\n", ret);
irow[0] = glob_row;
psb_c_dgeins(1, irow, zt, bh, cdh);
zt[0] = 0.0;
psb_c_dgeins(1, irow, zt, xh, cdh);
}
#ifdef PSB_HAVE_CUDA
/* Execute assembly and final allocation on CUDA device */
info = psb_c_cdasb_format(cdh, cdfmt);
if (info != 0)
{
return (info);
}
#ifdef DEBUG
else if (iam == 0)
{
fprintf(stdout, "Completed descriptor assembly format\n");
fflush(stdout);
}
#endif
info = psb_c_dspasb_opt(ah, cdh, afmt, PSB_UPD_DFLT, PSB_DUPL_DEF);
if (info != 0)
{
return (info);
}
#ifdef DEBUG
else if (iam == 0)
{
fprintf(stdout, "Completed matrix assembly\n");
fflush(stdout);
}
#endif
info = psb_c_dgeasb_options_format(xh, cdh, PSB_DUPL_ADD, cdfmt);
if (info != 0)
{
return (info);
}
#ifdef DEBUG
else if (iam == 0)
{
fprintf(stdout, "Completed x vector assembly\n");
fflush(stdout);
}
#endif
info = psb_c_dgeasb_options_format(bh, cdh, PSB_DUPL_ADD, cdfmt);
if (info != 0)
{
return (info);
}
#ifdef DEBUG
else if (iam == 0)
{
fprintf(stdout, "Completed b vector assembly\n");
fflush(stdout);
}
#endif
info = psb_c_dgeasb_options_format(rh, cdh, PSB_DUPL_ADD, cdfmt);
if (info != 0)
{
return (info);
}
#ifdef DEBUG
else if (iam == 0)
{
fprintf(stdout, "Completed r vector assembly\n");
fflush(stdout);
}
#endif
#else
/* Execute assembly and final allocation HOST side */
if ((info = psb_c_cdasb(cdh)) != 0)
return (info);
if ((info = psb_c_dspasb(ah, cdh)) != 0)
return (info);
if ((info = psb_c_dgeasb(xh, cdh)) != 0)
return (info);
if ((info = psb_c_dgeasb(bh, cdh)) != 0)
return (info);
if ((info = psb_c_dgeasb(rh, cdh)) != 0)
return (info);
return (info);
#endif
}
#define LINEBUFSIZE 1024
static char buffer[LINEBUFSIZE + 1];
int get_buffer(FILE *fp)
{
while (!feof(fp))
{
fgets(buffer, LINEBUFSIZE, fp);
if (buffer[0] != '%')
break;
}
}
void get_iparm(FILE *fp, int *val)
{
get_buffer(fp);
// fprintf(stderr,"Reading int parm: %s\n",buffer);
sscanf(buffer, "%d ", val);
}
void get_dparm(FILE *fp, double *val)
{
get_buffer(fp);
sscanf(buffer, "%lf ", val);
}
void get_hparm(FILE *fp, char *val)
{
get_buffer(fp);
sscanf(buffer, "%s ", val);
}
#define DUMPMATRIX 0
int main(int argc, char *argv[])
{
psb_c_ctxt *cctxt;
psb_i_t iam, np;
char methd[40], ptype[40], afmt[8], cdfmt[8];
psb_i_t nparms;
psb_i_t idim, info, istop, itmax, itrace, irst, iter, ret;
amg_c_dprec *ph;
psb_c_dspmat *ah;
psb_c_dvector *bh, *xh, *rh;
psb_i_t nb, nlr, nl;
psb_l_t i, ng, *vl, k;
double t1, t2, eps, err;
double *xv, *bv, *rv;
double one = 1.0, zero = 0.0, res2;
psb_c_SolverOptions options;
psb_c_descriptor *cdh;
FILE *vectfile;
cctxt = psb_c_new_ctxt();
psb_c_init(cctxt);
#ifdef PSB_HAVE_CUDA
psb_c_cuda_init(cctxt);
#endif
psb_c_info(*cctxt, &iam, &np);
#ifdef PSB_HAVE_CUDA
if (iam == 0)
{
fprintf(stderr, "-- CUDA initialized --\n");
fprintf(stderr, "Number of available GPU devices: %d\n", psb_c_cuda_getDeviceCount());
}
#endif
fprintf(stdout, "Initialization: am %d of %d\n", iam, np);
fflush(stdout);
psb_c_barrier(*cctxt);
if (iam == 0)
{
get_iparm(stdin, &nparms);
get_hparm(stdin, methd);
get_hparm(stdin, ptype);
get_hparm(stdin, afmt);
get_hparm(stdin, cdfmt);
get_iparm(stdin, &idim);
get_iparm(stdin, &istop);
get_iparm(stdin, &itmax);
get_iparm(stdin, &itrace);
get_iparm(stdin, &irst);
#if 0
/* Display paremeters */
fprintf(stderr, "Input parameters:\n");
fprintf(stderr, " Number of parameters: %d\n", nparms);
fprintf(stderr, " Method: %s\n", methd);
fprintf(stderr, " Preconditioner type: %s\n", ptype);
fprintf(stderr, " Matrix format: %s\n", afmt);
fprintf(stderr, " Descriptor format: %s\n", cdfmt);
fprintf(stderr, " Problem dimension: %d\n", idim);
fprintf(stderr, " Stopping criterion: %d\n", istop);
fprintf(stderr, " Maximum iterations: %d\n", itmax);
fprintf(stderr, " Trace frequency: %d\n", itrace);
fprintf(stderr, " Restart depth: %d\n", irst);
#endif
}
/* Now broadcast the values, and check they're OK */
psb_c_ibcast(*cctxt, 1, &nparms, 0);
psb_c_hbcast(*cctxt, methd, 0);
psb_c_hbcast(*cctxt, ptype, 0);
psb_c_hbcast(*cctxt, afmt, 0);
psb_c_hbcast(*cctxt, cdfmt, 0);
psb_c_ibcast(*cctxt, 1, &idim, 0);
psb_c_ibcast(*cctxt, 1, &istop, 0);
psb_c_ibcast(*cctxt, 1, &itmax, 0);
psb_c_ibcast(*cctxt, 1, &itrace, 0);
psb_c_ibcast(*cctxt, 1, &irst, 0);
psb_c_barrier(*cctxt);
cdh = psb_c_new_descriptor();
psb_c_set_index_base(0);
/* Simple minded BLOCK data distribution */
ng = ((psb_l_t)idim) * idim * idim;
nb = (ng + np - 1) / np;
nl = nb;
if ((ng - iam * nb) < nl)
nl = ng - iam * nb;
fprintf(stderr, "%d: Input data %d %ld %d %d\n", iam, idim, ng, nb, nl);
if ((vl = malloc(nb * sizeof(psb_l_t))) == NULL)
{
fprintf(stderr, "On %d: malloc failure\n", iam);
psb_c_abort(*cctxt);
}
i = ((psb_l_t)iam) * nb;
for (k = 0; k < nl; k++)
vl[k] = i + k;
if ((info = psb_c_cdall_vl(nl, vl, *cctxt, cdh)) != 0)
{
fprintf(stderr, "From cdall: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
}
bh = psb_c_new_dvector();
xh = psb_c_new_dvector();
rh = psb_c_new_dvector();
ah = psb_c_new_dspmat();
// fprintf(stderr,"From psb_c_new_dspmat: %p\n",ah);
/* Allocate mem space for sparse matrix and vectors */
ret = psb_c_dspall(ah, cdh);
// fprintf(stderr,"From psb_c_dspall: %d\n",ret);
psb_c_dgeall(bh, cdh);
psb_c_dgeall(xh, cdh);
psb_c_dgeall(rh, cdh);
if (iam == 0)
{
fprintf(stdout, "Matrix and vectors allocated\n");
}
psb_c_barrier(*cctxt);
/* Matrix generation */
if (matgen(*cctxt, nl, idim, vl, ah, afmt, cdh, cdfmt, xh, bh, rh) != 0)
{
fprintf(stderr, "Error during matrix build loop\n");
psb_c_abort(*cctxt);
}
psb_c_barrier(*cctxt);
#if 0
if (iam == 0)
{
fprintf(stdout, "Matrix and vectors generated\n");
}
#endif
fflush(stdout);
/* Set up the preconditioner */
ph = amg_c_dprec_new();
amg_c_dprecinit(*cctxt, ph, ptype);
amg_c_dprecsetc(ph, "SMOOTHER_TYPE", "L1-JACOBI");
amg_c_dprecseti(ph, "SMOOTHER_SWEEPS", 2);
amg_c_dprecsetc(ph, "COARSE_SOLVE", "BJAC");
amg_c_dprecsetc(ph, "COARSE_SUBSOLVE", "L1-JACOBI");
amg_c_dprecsetc(ph, "AGGR_FILTER", "FILTER");
if ((ret = amg_c_dhierarchy_build(ah, cdh, ph)) != 0){
fprintf(stderr, "From hierarchy_build: %d\n", ret);
}{
#if 0
if (iam == 0) {
fprintf(stdout, "Hierarchy built\n");
}
#endif
}
#if defined (PSB_HAVE_CUDA)
if ((ret = amg_c_dsmoothers_build_opt(ah, cdh, ph, afmt, cdfmt)) != 0)
fprintf(stderr, "From smoothers_build_format: %d\n", ret);
#else
if ((ret = amg_c_dsmoothers_build(ah, cdh, ph)) != 0)
fprintf(stderr, "From smoothers_build: %d\n", ret);
#endif
#if 0
if ( ret == 0){
if (iam == 0) {
fprintf(stdout, "Smoothers built\n");
}
}
#endif
#ifdef PSB_HAVE_CUDA
/* Allocate work vectors for the preconditioner on the GPU */
info = amg_c_dallocate_wrk(ph, cdfmt);
if (info != 0)
{
fprintf(stderr, "From dallocate_wrk: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
} else {
if (iam == 0) {
fprintf(stdout, "Preconditioner work vectors allocated\n");
}
}
#endif
psb_c_barrier(*cctxt);
/* Do a dry run of the preconditioner */
info = amg_c_dprecapply(ph, bh, xh, cdh);
if (info != 0)
{
fprintf(stderr, "From dprec_apply: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
}
/* Do a dry run of the preconditioner with the option routine */
info = amg_c_dprecapply_opt(ph, bh, xh, cdh, "N");
if (info != 0)
{
fprintf(stderr, "From dprec_apply_opt: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
}
/*
info = amg_c_dprecapply_opt(ph,bh,xh,cdh,"T");
if (info != 0) {
fprintf(stderr,"From dprec_apply_opt: %d\nBailing out\n",info);
psb_c_abort(*cctxt);
}
*/
/* Print the information on the preconditioner */
if (iam == 0)
{
info = amg_c_ddescr(ph);
if (info != 0)
{
fprintf(stderr, "From ddescr: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
}
}
/* Set up the solver options */
psb_c_DefaultSolverOptions(&options);
options.eps = 1.e-6;
options.itmax = itmax;
options.irst = irst;
options.itrace = 1;
options.istop = istop;
psb_c_seterraction_ret();
t1 = psb_c_wtime();
ret = amg_c_dkrylov(methd, ah, ph, bh, xh, cdh, &options);
t2 = psb_c_wtime();
iter = options.iter;
err = options.err;
// fprintf(stderr,"From krylov: %d %lf, %d %d\n",iter,err,ret,psb_c_get_errstatus());
if (psb_c_get_errstatus() != 0)
{
psb_c_print_errmsg();
}
// fprintf(stderr,"After cleanup %d\n",psb_c_get_errstatus());
/* Check 2-norm of residual on exit */
psb_c_dgeaxpby(one, bh, zero, rh, cdh);
psb_c_dspmm(-one, ah, xh, one, rh, cdh);
res2 = psb_c_dgenrm2(rh, cdh);
if (iam == 0)
{
fprintf(stdout, "Time: %lf\n", (t2 - t1));
fprintf(stdout, "Iter: %d\n", iter);
fprintf(stdout, "Err: %lg\n", err);
fprintf(stdout, "||r||_2: %lg\n", res2);
}
#if DUMPATRIX
psb_c_dmat_name_print(ah, "cbindmat.mtx");
nlr = psb_c_cd_get_local_rows(cdh);
bv = psb_c_dvect_get_cpy(bh);
vectfile = fopen("cbindb.mtx", "w");
for (i = 0; i < nlr; i++)
fprintf(vectfile, "%lf\n", bv[i]);
fclose(vectfile);
xv = psb_c_dvect_get_cpy(xh);
nlr = psb_c_cd_get_local_rows(cdh);
for (i = 0; i < nlr; i++)
fprintf(stdout, "SOL: %d %d %lf\n", iam, i, xv[i]);
rv = psb_c_dvect_get_cpy(rh);
nlr = psb_c_cd_get_local_rows(cdh);
for (i = 0; i < nlr; i++)
fprintf(stdout, "RES: %d %d %lf\n", iam, i, rv[i]);
#endif
/* Clean up memory */
if ((info = psb_c_dgefree(xh, cdh)) != 0)
{
fprintf(stderr, "From dgefree: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
}
#if 0
fprintf(stderr, "after dgefree xh\n");
#endif
if ((info = psb_c_dgefree(bh, cdh)) != 0)
{
fprintf(stderr, "From dgefree: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
}
#if 0
fprintf(stderr, "after dgefree bh\n");
#endif
if ((info = psb_c_dgefree(rh, cdh)) != 0)
{
fprintf(stderr, "From dgefree: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
}
#if 0
fprintf(stderr, "after dgefree rh\n");
#endif
if ((info = psb_c_cdfree(cdh)) != 0)
{
fprintf(stderr, "From cdfree: %d\nBailing out\n", info);
psb_c_abort(*cctxt);
}
#if 0
fprintf(stderr, "after cdfree\n");
#endif
// fprintf(stderr,"pointer from cdfree: %p\n",cdh->descriptor);
/* Clean up object handles */
free(ph);
free(xh);
free(bh);
free(ah);
free(cdh);
if (iam == 0)
fprintf(stderr, "program completed successfully\n");
#ifdef PSB_HAVE_CUDA
psb_c_cuda_exit();
#endif
psb_c_barrier(*cctxt);
psb_c_exit(*cctxt);
free(cctxt);
}
+10
View File
@@ -0,0 +1,10 @@
9 Number of entries below this
BICGSTAB Iterative method BICGSTAB CGS BICG BICGSTABL RGMRES
ML Preconditioner NONE DIAG BJAC
HLG A Storage format CSR COO JAD (HOST) CSRG HLG (GPU)
DEVICE Communication format CSR COO JAD (HOST) CSRG HLG (GPU)
60 Domain size (acutal system is this**3)
1 Stopping criterion
80 MAXIT
01 ITRACE
20 IRST restart for RGMRES and BiCGSTABL
Binary file not shown.
+1 -1
View File
@@ -199,7 +199,7 @@ class="cmr-12">.</span>
<span
class="cmr-12">This step is complementary to step 1 and should be performed when the</span>
<span
class="cmr-12">preconditioner is no more used.</span></li></ol>
class="cmr-12">preconditioner is no longer used.</span></li></ol>
<!--l. 59--><p class="indent" > <span
class="cmr-12">All the previous routines are available as methods of the preconditioner object. A</span>
<span
+1 -1
View File
@@ -53,7 +53,7 @@ performed by the routine \fortinline|bld|.
by the PSBLAS routine implementing the Krylov solver (\fortinline|psb_krylov|).
\item \emph{Free the preconditioner data structure}. This is performed by
the routine \fortinline|free|. This step is complementary to step 1 and should
be performed when the preconditioner is no more used.
be performed when the preconditioner is no longer used.
\end{enumerate}
All the previous routines are available as methods of the preconditioner object.
+1 -1
View File
@@ -21,7 +21,7 @@ all: amg_s_pde3d amg_d_pde3d amg_s_pde2d amg_d_pde2d
amg_d_pde3d: amg_d_pde3d.o amg_d_genpde_mod.o $(DGEN3D) data_input.o
$(FLINK) $(LINKOPT) amg_d_pde3d.o amg_d_genpde_mod.o $(DGEN3D) data_input.o \
-o amg_d_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(LDLIBS)
-o amg_d_pde3d $(AMG_LIBS) $(PSBLAS_LIBS) $(AMG_LDLIBS) $(LDLIBS)
/bin/mv amg_d_pde3d $(EXEDIR)
amg_s_pde3d: amg_s_pde3d.o amg_s_genpde_mod.o $(SGEN3D) data_input.o