mirror of
https://github.com/sfilippone/psblas3.git
synced 2026-10-07 15:14:57 +00:00
Fix fix_coo for OpenMP
This commit is contained in:
+17
-1279
File diff suppressed because it is too large
Load Diff
+146
-146
@@ -2902,78 +2902,78 @@ subroutine psb_c_cp_csr_from_coo(a,b,info)
|
||||
|
||||
a%irp(:) = 0
|
||||
|
||||
if (use_openmp) then
|
||||
!$ maxthreads = omp_get_max_threads()
|
||||
!$ allocate(sum(maxthreads+1))
|
||||
!$ sum(:) = 0
|
||||
!$ sum(1) = 1
|
||||
|
||||
!$OMP PARALLEL default(none) &
|
||||
!$OMP shared(nza,itemp,a,nthreads,sum,nr) &
|
||||
!$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
|
||||
!$OMP DO schedule(STATIC) &
|
||||
!$OMP private(k,i)
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
end do
|
||||
!$OMP END DO
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ nthreads = omp_get_num_threads()
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ ithread = omp_get_thread_num()
|
||||
|
||||
!$ work = nr/nthreads
|
||||
!$ if (ithread < MOD(nr,nthreads)) then
|
||||
!$ work = work + 1
|
||||
!$ first_idx = ithread*work + 1
|
||||
!$ else
|
||||
!$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!$ end if
|
||||
|
||||
!$ last_idx = first_idx + work - 1
|
||||
|
||||
!$ s = 0
|
||||
!$ do i=first_idx,last_idx
|
||||
!$ s = s + a%irp(i)
|
||||
!$ end do
|
||||
!$ if (work > 0) then
|
||||
!$ sum(ithread+2) = s
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ do i=2,nthreads+1
|
||||
!$ sum(i) = sum(i) + sum(i-1)
|
||||
!$ end do
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ saved_elem = a%irp(first_idx)
|
||||
!$ end if
|
||||
!$ if (ithread == 0) then
|
||||
!$ a%irp(1) = 1
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ old_val = a%irp(first_idx+1)
|
||||
!$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!$ end if
|
||||
|
||||
!$ do i=first_idx+2,last_idx+1
|
||||
!$ nxt_val = a%irp(i)
|
||||
!$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!$ old_val = nxt_val
|
||||
!$ end do
|
||||
|
||||
!$OMP END PARALLEL
|
||||
else
|
||||
!!$ if (use_openmp) then
|
||||
!!$ !$ maxthreads = omp_get_max_threads()
|
||||
!!$ !$ allocate(sum(maxthreads+1))
|
||||
!!$ !$ sum(:) = 0
|
||||
!!$ !$ sum(1) = 1
|
||||
!!$
|
||||
!!$ !$OMP PARALLEL default(none) &
|
||||
!!$ !$OMP shared(nza,itemp,a,nthreads,sum,nr) &
|
||||
!!$ !$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
!!$
|
||||
!!$ !$OMP DO schedule(STATIC) &
|
||||
!!$ !$OMP private(k,i)
|
||||
!!$ do k=1,nza
|
||||
!!$ i = itemp(k)
|
||||
!!$ a%irp(i) = a%irp(i) + 1
|
||||
!!$ end do
|
||||
!!$ !$OMP END DO
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ nthreads = omp_get_num_threads()
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ ithread = omp_get_thread_num()
|
||||
!!$
|
||||
!!$ !$ work = nr/nthreads
|
||||
!!$ !$ if (ithread < MOD(nr,nthreads)) then
|
||||
!!$ !$ work = work + 1
|
||||
!!$ !$ first_idx = ithread*work + 1
|
||||
!!$ !$ else
|
||||
!!$ !$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ last_idx = first_idx + work - 1
|
||||
!!$
|
||||
!!$ !$ s = 0
|
||||
!!$ !$ do i=first_idx,last_idx
|
||||
!!$ !$ s = s + a%irp(i)
|
||||
!!$ !$ end do
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ sum(ithread+2) = s
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ do i=2,nthreads+1
|
||||
!!$ !$ sum(i) = sum(i) + sum(i-1)
|
||||
!!$ !$ end do
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ saved_elem = a%irp(first_idx)
|
||||
!!$ !$ end if
|
||||
!!$ !$ if (ithread == 0) then
|
||||
!!$ !$ a%irp(1) = 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ old_val = a%irp(first_idx+1)
|
||||
!!$ !$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ do i=first_idx+2,last_idx+1
|
||||
!!$ !$ nxt_val = a%irp(i)
|
||||
!!$ !$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!!$ !$ old_val = nxt_val
|
||||
!!$ !$ end do
|
||||
!!$
|
||||
!!$ !$OMP END PARALLEL
|
||||
!!$ else
|
||||
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
@@ -2986,7 +2986,7 @@ subroutine psb_c_cp_csr_from_coo(a,b,info)
|
||||
ip = ip + ncl
|
||||
end do
|
||||
a%irp(nr+1) = ip
|
||||
end if
|
||||
!!$ end if
|
||||
call a%set_host()
|
||||
|
||||
|
||||
@@ -3104,10 +3104,10 @@ subroutine psb_c_mv_csr_from_coo(a,b,info)
|
||||
character(len=20) :: name='mv_from_coo'
|
||||
logical :: use_openmp = .false.
|
||||
|
||||
!$ integer(psb_ipk_), allocatable :: sum(:)
|
||||
!$ integer(psb_ipk_) :: first_idx,last_idx,work,ithread,nthreads,s
|
||||
!$ integer(psb_ipk_) :: nxt_val,old_val,saved_elem
|
||||
!$ use_openmp = .true.
|
||||
! $ integer(psb_ipk_), allocatable :: sum(:)
|
||||
! $ integer(psb_ipk_) :: first_idx,last_idx,work,ithread,nthreads,s
|
||||
! $ integer(psb_ipk_) :: nxt_val,old_val,saved_elem
|
||||
! $ use_openmp = .true.
|
||||
|
||||
|
||||
info = psb_success_
|
||||
@@ -3135,74 +3135,74 @@ subroutine psb_c_mv_csr_from_coo(a,b,info)
|
||||
|
||||
a%irp(:) = 0
|
||||
|
||||
if (use_openmp) then
|
||||
!$OMP PARALLEL default(none) &
|
||||
!$OMP shared(sum,nthreads,nr,a,itemp,nza) &
|
||||
!$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
|
||||
!$OMP DO schedule(STATIC) &
|
||||
!$OMP private(k,i)
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
end do
|
||||
!$OMP END DO
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ nthreads = omp_get_num_threads()
|
||||
!$ allocate(sum(nthreads+1))
|
||||
!$ sum(:) = 0
|
||||
!$ sum(1) = 1
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ ithread = omp_get_thread_num()
|
||||
|
||||
!$ work = nr/nthreads
|
||||
!$ if (ithread < MOD(nr,nthreads)) then
|
||||
!$ work = work + 1
|
||||
!$ first_idx = ithread*work + 1
|
||||
!$ else
|
||||
!$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!$ end if
|
||||
|
||||
!$ last_idx = first_idx + work - 1
|
||||
|
||||
!$ s = 0
|
||||
!$ do i=first_idx,last_idx
|
||||
!$ s = s + a%irp(i)
|
||||
!$ end do
|
||||
!$ if (work > 0) then
|
||||
!$ sum(ithread+2) = s
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ do i=2,nthreads+1
|
||||
!$ sum(i) = sum(i) + sum(i-1)
|
||||
!$ end do
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ saved_elem = a%irp(first_idx)
|
||||
!$ end if
|
||||
!$ if (ithread == 0) then
|
||||
!$ a%irp(1) = 1
|
||||
!$ end if
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ old_val = a%irp(first_idx+1)
|
||||
!$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!$ end if
|
||||
|
||||
!$ do i=first_idx+2,last_idx+1
|
||||
!$ nxt_val = a%irp(i)
|
||||
!$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!$ old_val = nxt_val
|
||||
!$ end do
|
||||
|
||||
!$OMP END PARALLEL
|
||||
else
|
||||
!!$ if (use_openmp) then
|
||||
!!$ !$OMP PARALLEL default(none) &
|
||||
!!$ !$OMP shared(sum,nthreads,nr,a,itemp,nza) &
|
||||
!!$ !$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
!!$
|
||||
!!$ !$OMP DO schedule(STATIC) &
|
||||
!!$ !$OMP private(k,i)
|
||||
!!$ do k=1,nza
|
||||
!!$ i = itemp(k)
|
||||
!!$ a%irp(i) = a%irp(i) + 1
|
||||
!!$ end do
|
||||
!!$ !$OMP END DO
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ nthreads = omp_get_num_threads()
|
||||
!!$ !$ allocate(sum(nthreads+1))
|
||||
!!$ !$ sum(:) = 0
|
||||
!!$ !$ sum(1) = 1
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ ithread = omp_get_thread_num()
|
||||
!!$
|
||||
!!$ !$ work = nr/nthreads
|
||||
!!$ !$ if (ithread < MOD(nr,nthreads)) then
|
||||
!!$ !$ work = work + 1
|
||||
!!$ !$ first_idx = ithread*work + 1
|
||||
!!$ !$ else
|
||||
!!$ !$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ last_idx = first_idx + work - 1
|
||||
!!$
|
||||
!!$ !$ s = 0
|
||||
!!$ !$ do i=first_idx,last_idx
|
||||
!!$ !$ s = s + a%irp(i)
|
||||
!!$ !$ end do
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ sum(ithread+2) = s
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ do i=2,nthreads+1
|
||||
!!$ !$ sum(i) = sum(i) + sum(i-1)
|
||||
!!$ !$ end do
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ saved_elem = a%irp(first_idx)
|
||||
!!$ !$ end if
|
||||
!!$ !$ if (ithread == 0) then
|
||||
!!$ !$ a%irp(1) = 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ old_val = a%irp(first_idx+1)
|
||||
!!$ !$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ do i=first_idx+2,last_idx+1
|
||||
!!$ !$ nxt_val = a%irp(i)
|
||||
!!$ !$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!!$ !$ old_val = nxt_val
|
||||
!!$ !$ end do
|
||||
!!$
|
||||
!!$ !$OMP END PARALLEL
|
||||
!!$ else
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
@@ -3214,7 +3214,7 @@ subroutine psb_c_mv_csr_from_coo(a,b,info)
|
||||
ip = ip + ncl
|
||||
end do
|
||||
a%irp(nr+1) = ip
|
||||
end if
|
||||
!!$ end if
|
||||
|
||||
call a%set_host()
|
||||
|
||||
|
||||
+17
-1279
File diff suppressed because it is too large
Load Diff
+146
-146
@@ -2902,78 +2902,78 @@ subroutine psb_d_cp_csr_from_coo(a,b,info)
|
||||
|
||||
a%irp(:) = 0
|
||||
|
||||
if (use_openmp) then
|
||||
!$ maxthreads = omp_get_max_threads()
|
||||
!$ allocate(sum(maxthreads+1))
|
||||
!$ sum(:) = 0
|
||||
!$ sum(1) = 1
|
||||
|
||||
!$OMP PARALLEL default(none) &
|
||||
!$OMP shared(nza,itemp,a,nthreads,sum,nr) &
|
||||
!$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
|
||||
!$OMP DO schedule(STATIC) &
|
||||
!$OMP private(k,i)
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
end do
|
||||
!$OMP END DO
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ nthreads = omp_get_num_threads()
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ ithread = omp_get_thread_num()
|
||||
|
||||
!$ work = nr/nthreads
|
||||
!$ if (ithread < MOD(nr,nthreads)) then
|
||||
!$ work = work + 1
|
||||
!$ first_idx = ithread*work + 1
|
||||
!$ else
|
||||
!$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!$ end if
|
||||
|
||||
!$ last_idx = first_idx + work - 1
|
||||
|
||||
!$ s = 0
|
||||
!$ do i=first_idx,last_idx
|
||||
!$ s = s + a%irp(i)
|
||||
!$ end do
|
||||
!$ if (work > 0) then
|
||||
!$ sum(ithread+2) = s
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ do i=2,nthreads+1
|
||||
!$ sum(i) = sum(i) + sum(i-1)
|
||||
!$ end do
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ saved_elem = a%irp(first_idx)
|
||||
!$ end if
|
||||
!$ if (ithread == 0) then
|
||||
!$ a%irp(1) = 1
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ old_val = a%irp(first_idx+1)
|
||||
!$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!$ end if
|
||||
|
||||
!$ do i=first_idx+2,last_idx+1
|
||||
!$ nxt_val = a%irp(i)
|
||||
!$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!$ old_val = nxt_val
|
||||
!$ end do
|
||||
|
||||
!$OMP END PARALLEL
|
||||
else
|
||||
!!$ if (use_openmp) then
|
||||
!!$ !$ maxthreads = omp_get_max_threads()
|
||||
!!$ !$ allocate(sum(maxthreads+1))
|
||||
!!$ !$ sum(:) = 0
|
||||
!!$ !$ sum(1) = 1
|
||||
!!$
|
||||
!!$ !$OMP PARALLEL default(none) &
|
||||
!!$ !$OMP shared(nza,itemp,a,nthreads,sum,nr) &
|
||||
!!$ !$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
!!$
|
||||
!!$ !$OMP DO schedule(STATIC) &
|
||||
!!$ !$OMP private(k,i)
|
||||
!!$ do k=1,nza
|
||||
!!$ i = itemp(k)
|
||||
!!$ a%irp(i) = a%irp(i) + 1
|
||||
!!$ end do
|
||||
!!$ !$OMP END DO
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ nthreads = omp_get_num_threads()
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ ithread = omp_get_thread_num()
|
||||
!!$
|
||||
!!$ !$ work = nr/nthreads
|
||||
!!$ !$ if (ithread < MOD(nr,nthreads)) then
|
||||
!!$ !$ work = work + 1
|
||||
!!$ !$ first_idx = ithread*work + 1
|
||||
!!$ !$ else
|
||||
!!$ !$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ last_idx = first_idx + work - 1
|
||||
!!$
|
||||
!!$ !$ s = 0
|
||||
!!$ !$ do i=first_idx,last_idx
|
||||
!!$ !$ s = s + a%irp(i)
|
||||
!!$ !$ end do
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ sum(ithread+2) = s
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ do i=2,nthreads+1
|
||||
!!$ !$ sum(i) = sum(i) + sum(i-1)
|
||||
!!$ !$ end do
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ saved_elem = a%irp(first_idx)
|
||||
!!$ !$ end if
|
||||
!!$ !$ if (ithread == 0) then
|
||||
!!$ !$ a%irp(1) = 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ old_val = a%irp(first_idx+1)
|
||||
!!$ !$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ do i=first_idx+2,last_idx+1
|
||||
!!$ !$ nxt_val = a%irp(i)
|
||||
!!$ !$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!!$ !$ old_val = nxt_val
|
||||
!!$ !$ end do
|
||||
!!$
|
||||
!!$ !$OMP END PARALLEL
|
||||
!!$ else
|
||||
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
@@ -2986,7 +2986,7 @@ subroutine psb_d_cp_csr_from_coo(a,b,info)
|
||||
ip = ip + ncl
|
||||
end do
|
||||
a%irp(nr+1) = ip
|
||||
end if
|
||||
!!$ end if
|
||||
call a%set_host()
|
||||
|
||||
|
||||
@@ -3104,10 +3104,10 @@ subroutine psb_d_mv_csr_from_coo(a,b,info)
|
||||
character(len=20) :: name='mv_from_coo'
|
||||
logical :: use_openmp = .false.
|
||||
|
||||
!$ integer(psb_ipk_), allocatable :: sum(:)
|
||||
!$ integer(psb_ipk_) :: first_idx,last_idx,work,ithread,nthreads,s
|
||||
!$ integer(psb_ipk_) :: nxt_val,old_val,saved_elem
|
||||
!$ use_openmp = .true.
|
||||
! $ integer(psb_ipk_), allocatable :: sum(:)
|
||||
! $ integer(psb_ipk_) :: first_idx,last_idx,work,ithread,nthreads,s
|
||||
! $ integer(psb_ipk_) :: nxt_val,old_val,saved_elem
|
||||
! $ use_openmp = .true.
|
||||
|
||||
|
||||
info = psb_success_
|
||||
@@ -3135,74 +3135,74 @@ subroutine psb_d_mv_csr_from_coo(a,b,info)
|
||||
|
||||
a%irp(:) = 0
|
||||
|
||||
if (use_openmp) then
|
||||
!$OMP PARALLEL default(none) &
|
||||
!$OMP shared(sum,nthreads,nr,a,itemp,nza) &
|
||||
!$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
|
||||
!$OMP DO schedule(STATIC) &
|
||||
!$OMP private(k,i)
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
end do
|
||||
!$OMP END DO
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ nthreads = omp_get_num_threads()
|
||||
!$ allocate(sum(nthreads+1))
|
||||
!$ sum(:) = 0
|
||||
!$ sum(1) = 1
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ ithread = omp_get_thread_num()
|
||||
|
||||
!$ work = nr/nthreads
|
||||
!$ if (ithread < MOD(nr,nthreads)) then
|
||||
!$ work = work + 1
|
||||
!$ first_idx = ithread*work + 1
|
||||
!$ else
|
||||
!$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!$ end if
|
||||
|
||||
!$ last_idx = first_idx + work - 1
|
||||
|
||||
!$ s = 0
|
||||
!$ do i=first_idx,last_idx
|
||||
!$ s = s + a%irp(i)
|
||||
!$ end do
|
||||
!$ if (work > 0) then
|
||||
!$ sum(ithread+2) = s
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ do i=2,nthreads+1
|
||||
!$ sum(i) = sum(i) + sum(i-1)
|
||||
!$ end do
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ saved_elem = a%irp(first_idx)
|
||||
!$ end if
|
||||
!$ if (ithread == 0) then
|
||||
!$ a%irp(1) = 1
|
||||
!$ end if
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ old_val = a%irp(first_idx+1)
|
||||
!$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!$ end if
|
||||
|
||||
!$ do i=first_idx+2,last_idx+1
|
||||
!$ nxt_val = a%irp(i)
|
||||
!$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!$ old_val = nxt_val
|
||||
!$ end do
|
||||
|
||||
!$OMP END PARALLEL
|
||||
else
|
||||
!!$ if (use_openmp) then
|
||||
!!$ !$OMP PARALLEL default(none) &
|
||||
!!$ !$OMP shared(sum,nthreads,nr,a,itemp,nza) &
|
||||
!!$ !$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
!!$
|
||||
!!$ !$OMP DO schedule(STATIC) &
|
||||
!!$ !$OMP private(k,i)
|
||||
!!$ do k=1,nza
|
||||
!!$ i = itemp(k)
|
||||
!!$ a%irp(i) = a%irp(i) + 1
|
||||
!!$ end do
|
||||
!!$ !$OMP END DO
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ nthreads = omp_get_num_threads()
|
||||
!!$ !$ allocate(sum(nthreads+1))
|
||||
!!$ !$ sum(:) = 0
|
||||
!!$ !$ sum(1) = 1
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ ithread = omp_get_thread_num()
|
||||
!!$
|
||||
!!$ !$ work = nr/nthreads
|
||||
!!$ !$ if (ithread < MOD(nr,nthreads)) then
|
||||
!!$ !$ work = work + 1
|
||||
!!$ !$ first_idx = ithread*work + 1
|
||||
!!$ !$ else
|
||||
!!$ !$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ last_idx = first_idx + work - 1
|
||||
!!$
|
||||
!!$ !$ s = 0
|
||||
!!$ !$ do i=first_idx,last_idx
|
||||
!!$ !$ s = s + a%irp(i)
|
||||
!!$ !$ end do
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ sum(ithread+2) = s
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ do i=2,nthreads+1
|
||||
!!$ !$ sum(i) = sum(i) + sum(i-1)
|
||||
!!$ !$ end do
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ saved_elem = a%irp(first_idx)
|
||||
!!$ !$ end if
|
||||
!!$ !$ if (ithread == 0) then
|
||||
!!$ !$ a%irp(1) = 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ old_val = a%irp(first_idx+1)
|
||||
!!$ !$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ do i=first_idx+2,last_idx+1
|
||||
!!$ !$ nxt_val = a%irp(i)
|
||||
!!$ !$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!!$ !$ old_val = nxt_val
|
||||
!!$ !$ end do
|
||||
!!$
|
||||
!!$ !$OMP END PARALLEL
|
||||
!!$ else
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
@@ -3214,7 +3214,7 @@ subroutine psb_d_mv_csr_from_coo(a,b,info)
|
||||
ip = ip + ncl
|
||||
end do
|
||||
a%irp(nr+1) = ip
|
||||
end if
|
||||
!!$ end if
|
||||
|
||||
call a%set_host()
|
||||
|
||||
|
||||
+17
-1279
File diff suppressed because it is too large
Load Diff
+146
-146
@@ -2902,78 +2902,78 @@ subroutine psb_s_cp_csr_from_coo(a,b,info)
|
||||
|
||||
a%irp(:) = 0
|
||||
|
||||
if (use_openmp) then
|
||||
!$ maxthreads = omp_get_max_threads()
|
||||
!$ allocate(sum(maxthreads+1))
|
||||
!$ sum(:) = 0
|
||||
!$ sum(1) = 1
|
||||
|
||||
!$OMP PARALLEL default(none) &
|
||||
!$OMP shared(nza,itemp,a,nthreads,sum,nr) &
|
||||
!$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
|
||||
!$OMP DO schedule(STATIC) &
|
||||
!$OMP private(k,i)
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
end do
|
||||
!$OMP END DO
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ nthreads = omp_get_num_threads()
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ ithread = omp_get_thread_num()
|
||||
|
||||
!$ work = nr/nthreads
|
||||
!$ if (ithread < MOD(nr,nthreads)) then
|
||||
!$ work = work + 1
|
||||
!$ first_idx = ithread*work + 1
|
||||
!$ else
|
||||
!$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!$ end if
|
||||
|
||||
!$ last_idx = first_idx + work - 1
|
||||
|
||||
!$ s = 0
|
||||
!$ do i=first_idx,last_idx
|
||||
!$ s = s + a%irp(i)
|
||||
!$ end do
|
||||
!$ if (work > 0) then
|
||||
!$ sum(ithread+2) = s
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ do i=2,nthreads+1
|
||||
!$ sum(i) = sum(i) + sum(i-1)
|
||||
!$ end do
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ saved_elem = a%irp(first_idx)
|
||||
!$ end if
|
||||
!$ if (ithread == 0) then
|
||||
!$ a%irp(1) = 1
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ old_val = a%irp(first_idx+1)
|
||||
!$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!$ end if
|
||||
|
||||
!$ do i=first_idx+2,last_idx+1
|
||||
!$ nxt_val = a%irp(i)
|
||||
!$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!$ old_val = nxt_val
|
||||
!$ end do
|
||||
|
||||
!$OMP END PARALLEL
|
||||
else
|
||||
!!$ if (use_openmp) then
|
||||
!!$ !$ maxthreads = omp_get_max_threads()
|
||||
!!$ !$ allocate(sum(maxthreads+1))
|
||||
!!$ !$ sum(:) = 0
|
||||
!!$ !$ sum(1) = 1
|
||||
!!$
|
||||
!!$ !$OMP PARALLEL default(none) &
|
||||
!!$ !$OMP shared(nza,itemp,a,nthreads,sum,nr) &
|
||||
!!$ !$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
!!$
|
||||
!!$ !$OMP DO schedule(STATIC) &
|
||||
!!$ !$OMP private(k,i)
|
||||
!!$ do k=1,nza
|
||||
!!$ i = itemp(k)
|
||||
!!$ a%irp(i) = a%irp(i) + 1
|
||||
!!$ end do
|
||||
!!$ !$OMP END DO
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ nthreads = omp_get_num_threads()
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ ithread = omp_get_thread_num()
|
||||
!!$
|
||||
!!$ !$ work = nr/nthreads
|
||||
!!$ !$ if (ithread < MOD(nr,nthreads)) then
|
||||
!!$ !$ work = work + 1
|
||||
!!$ !$ first_idx = ithread*work + 1
|
||||
!!$ !$ else
|
||||
!!$ !$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ last_idx = first_idx + work - 1
|
||||
!!$
|
||||
!!$ !$ s = 0
|
||||
!!$ !$ do i=first_idx,last_idx
|
||||
!!$ !$ s = s + a%irp(i)
|
||||
!!$ !$ end do
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ sum(ithread+2) = s
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ do i=2,nthreads+1
|
||||
!!$ !$ sum(i) = sum(i) + sum(i-1)
|
||||
!!$ !$ end do
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ saved_elem = a%irp(first_idx)
|
||||
!!$ !$ end if
|
||||
!!$ !$ if (ithread == 0) then
|
||||
!!$ !$ a%irp(1) = 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ old_val = a%irp(first_idx+1)
|
||||
!!$ !$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ do i=first_idx+2,last_idx+1
|
||||
!!$ !$ nxt_val = a%irp(i)
|
||||
!!$ !$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!!$ !$ old_val = nxt_val
|
||||
!!$ !$ end do
|
||||
!!$
|
||||
!!$ !$OMP END PARALLEL
|
||||
!!$ else
|
||||
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
@@ -2986,7 +2986,7 @@ subroutine psb_s_cp_csr_from_coo(a,b,info)
|
||||
ip = ip + ncl
|
||||
end do
|
||||
a%irp(nr+1) = ip
|
||||
end if
|
||||
!!$ end if
|
||||
call a%set_host()
|
||||
|
||||
|
||||
@@ -3104,10 +3104,10 @@ subroutine psb_s_mv_csr_from_coo(a,b,info)
|
||||
character(len=20) :: name='mv_from_coo'
|
||||
logical :: use_openmp = .false.
|
||||
|
||||
!$ integer(psb_ipk_), allocatable :: sum(:)
|
||||
!$ integer(psb_ipk_) :: first_idx,last_idx,work,ithread,nthreads,s
|
||||
!$ integer(psb_ipk_) :: nxt_val,old_val,saved_elem
|
||||
!$ use_openmp = .true.
|
||||
! $ integer(psb_ipk_), allocatable :: sum(:)
|
||||
! $ integer(psb_ipk_) :: first_idx,last_idx,work,ithread,nthreads,s
|
||||
! $ integer(psb_ipk_) :: nxt_val,old_val,saved_elem
|
||||
! $ use_openmp = .true.
|
||||
|
||||
|
||||
info = psb_success_
|
||||
@@ -3135,74 +3135,74 @@ subroutine psb_s_mv_csr_from_coo(a,b,info)
|
||||
|
||||
a%irp(:) = 0
|
||||
|
||||
if (use_openmp) then
|
||||
!$OMP PARALLEL default(none) &
|
||||
!$OMP shared(sum,nthreads,nr,a,itemp,nza) &
|
||||
!$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
|
||||
!$OMP DO schedule(STATIC) &
|
||||
!$OMP private(k,i)
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
end do
|
||||
!$OMP END DO
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ nthreads = omp_get_num_threads()
|
||||
!$ allocate(sum(nthreads+1))
|
||||
!$ sum(:) = 0
|
||||
!$ sum(1) = 1
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ ithread = omp_get_thread_num()
|
||||
|
||||
!$ work = nr/nthreads
|
||||
!$ if (ithread < MOD(nr,nthreads)) then
|
||||
!$ work = work + 1
|
||||
!$ first_idx = ithread*work + 1
|
||||
!$ else
|
||||
!$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!$ end if
|
||||
|
||||
!$ last_idx = first_idx + work - 1
|
||||
|
||||
!$ s = 0
|
||||
!$ do i=first_idx,last_idx
|
||||
!$ s = s + a%irp(i)
|
||||
!$ end do
|
||||
!$ if (work > 0) then
|
||||
!$ sum(ithread+2) = s
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ do i=2,nthreads+1
|
||||
!$ sum(i) = sum(i) + sum(i-1)
|
||||
!$ end do
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ saved_elem = a%irp(first_idx)
|
||||
!$ end if
|
||||
!$ if (ithread == 0) then
|
||||
!$ a%irp(1) = 1
|
||||
!$ end if
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ old_val = a%irp(first_idx+1)
|
||||
!$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!$ end if
|
||||
|
||||
!$ do i=first_idx+2,last_idx+1
|
||||
!$ nxt_val = a%irp(i)
|
||||
!$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!$ old_val = nxt_val
|
||||
!$ end do
|
||||
|
||||
!$OMP END PARALLEL
|
||||
else
|
||||
!!$ if (use_openmp) then
|
||||
!!$ !$OMP PARALLEL default(none) &
|
||||
!!$ !$OMP shared(sum,nthreads,nr,a,itemp,nza) &
|
||||
!!$ !$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
!!$
|
||||
!!$ !$OMP DO schedule(STATIC) &
|
||||
!!$ !$OMP private(k,i)
|
||||
!!$ do k=1,nza
|
||||
!!$ i = itemp(k)
|
||||
!!$ a%irp(i) = a%irp(i) + 1
|
||||
!!$ end do
|
||||
!!$ !$OMP END DO
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ nthreads = omp_get_num_threads()
|
||||
!!$ !$ allocate(sum(nthreads+1))
|
||||
!!$ !$ sum(:) = 0
|
||||
!!$ !$ sum(1) = 1
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ ithread = omp_get_thread_num()
|
||||
!!$
|
||||
!!$ !$ work = nr/nthreads
|
||||
!!$ !$ if (ithread < MOD(nr,nthreads)) then
|
||||
!!$ !$ work = work + 1
|
||||
!!$ !$ first_idx = ithread*work + 1
|
||||
!!$ !$ else
|
||||
!!$ !$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ last_idx = first_idx + work - 1
|
||||
!!$
|
||||
!!$ !$ s = 0
|
||||
!!$ !$ do i=first_idx,last_idx
|
||||
!!$ !$ s = s + a%irp(i)
|
||||
!!$ !$ end do
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ sum(ithread+2) = s
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ do i=2,nthreads+1
|
||||
!!$ !$ sum(i) = sum(i) + sum(i-1)
|
||||
!!$ !$ end do
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ saved_elem = a%irp(first_idx)
|
||||
!!$ !$ end if
|
||||
!!$ !$ if (ithread == 0) then
|
||||
!!$ !$ a%irp(1) = 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ old_val = a%irp(first_idx+1)
|
||||
!!$ !$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ do i=first_idx+2,last_idx+1
|
||||
!!$ !$ nxt_val = a%irp(i)
|
||||
!!$ !$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!!$ !$ old_val = nxt_val
|
||||
!!$ !$ end do
|
||||
!!$
|
||||
!!$ !$OMP END PARALLEL
|
||||
!!$ else
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
@@ -3214,7 +3214,7 @@ subroutine psb_s_mv_csr_from_coo(a,b,info)
|
||||
ip = ip + ncl
|
||||
end do
|
||||
a%irp(nr+1) = ip
|
||||
end if
|
||||
!!$ end if
|
||||
|
||||
call a%set_host()
|
||||
|
||||
|
||||
+17
-1279
File diff suppressed because it is too large
Load Diff
+146
-146
@@ -2902,78 +2902,78 @@ subroutine psb_z_cp_csr_from_coo(a,b,info)
|
||||
|
||||
a%irp(:) = 0
|
||||
|
||||
if (use_openmp) then
|
||||
!$ maxthreads = omp_get_max_threads()
|
||||
!$ allocate(sum(maxthreads+1))
|
||||
!$ sum(:) = 0
|
||||
!$ sum(1) = 1
|
||||
|
||||
!$OMP PARALLEL default(none) &
|
||||
!$OMP shared(nza,itemp,a,nthreads,sum,nr) &
|
||||
!$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
|
||||
!$OMP DO schedule(STATIC) &
|
||||
!$OMP private(k,i)
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
end do
|
||||
!$OMP END DO
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ nthreads = omp_get_num_threads()
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ ithread = omp_get_thread_num()
|
||||
|
||||
!$ work = nr/nthreads
|
||||
!$ if (ithread < MOD(nr,nthreads)) then
|
||||
!$ work = work + 1
|
||||
!$ first_idx = ithread*work + 1
|
||||
!$ else
|
||||
!$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!$ end if
|
||||
|
||||
!$ last_idx = first_idx + work - 1
|
||||
|
||||
!$ s = 0
|
||||
!$ do i=first_idx,last_idx
|
||||
!$ s = s + a%irp(i)
|
||||
!$ end do
|
||||
!$ if (work > 0) then
|
||||
!$ sum(ithread+2) = s
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ do i=2,nthreads+1
|
||||
!$ sum(i) = sum(i) + sum(i-1)
|
||||
!$ end do
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ saved_elem = a%irp(first_idx)
|
||||
!$ end if
|
||||
!$ if (ithread == 0) then
|
||||
!$ a%irp(1) = 1
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ old_val = a%irp(first_idx+1)
|
||||
!$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!$ end if
|
||||
|
||||
!$ do i=first_idx+2,last_idx+1
|
||||
!$ nxt_val = a%irp(i)
|
||||
!$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!$ old_val = nxt_val
|
||||
!$ end do
|
||||
|
||||
!$OMP END PARALLEL
|
||||
else
|
||||
!!$ if (use_openmp) then
|
||||
!!$ !$ maxthreads = omp_get_max_threads()
|
||||
!!$ !$ allocate(sum(maxthreads+1))
|
||||
!!$ !$ sum(:) = 0
|
||||
!!$ !$ sum(1) = 1
|
||||
!!$
|
||||
!!$ !$OMP PARALLEL default(none) &
|
||||
!!$ !$OMP shared(nza,itemp,a,nthreads,sum,nr) &
|
||||
!!$ !$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
!!$
|
||||
!!$ !$OMP DO schedule(STATIC) &
|
||||
!!$ !$OMP private(k,i)
|
||||
!!$ do k=1,nza
|
||||
!!$ i = itemp(k)
|
||||
!!$ a%irp(i) = a%irp(i) + 1
|
||||
!!$ end do
|
||||
!!$ !$OMP END DO
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ nthreads = omp_get_num_threads()
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ ithread = omp_get_thread_num()
|
||||
!!$
|
||||
!!$ !$ work = nr/nthreads
|
||||
!!$ !$ if (ithread < MOD(nr,nthreads)) then
|
||||
!!$ !$ work = work + 1
|
||||
!!$ !$ first_idx = ithread*work + 1
|
||||
!!$ !$ else
|
||||
!!$ !$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ last_idx = first_idx + work - 1
|
||||
!!$
|
||||
!!$ !$ s = 0
|
||||
!!$ !$ do i=first_idx,last_idx
|
||||
!!$ !$ s = s + a%irp(i)
|
||||
!!$ !$ end do
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ sum(ithread+2) = s
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ do i=2,nthreads+1
|
||||
!!$ !$ sum(i) = sum(i) + sum(i-1)
|
||||
!!$ !$ end do
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ saved_elem = a%irp(first_idx)
|
||||
!!$ !$ end if
|
||||
!!$ !$ if (ithread == 0) then
|
||||
!!$ !$ a%irp(1) = 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ old_val = a%irp(first_idx+1)
|
||||
!!$ !$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ do i=first_idx+2,last_idx+1
|
||||
!!$ !$ nxt_val = a%irp(i)
|
||||
!!$ !$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!!$ !$ old_val = nxt_val
|
||||
!!$ !$ end do
|
||||
!!$
|
||||
!!$ !$OMP END PARALLEL
|
||||
!!$ else
|
||||
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
@@ -2986,7 +2986,7 @@ subroutine psb_z_cp_csr_from_coo(a,b,info)
|
||||
ip = ip + ncl
|
||||
end do
|
||||
a%irp(nr+1) = ip
|
||||
end if
|
||||
!!$ end if
|
||||
call a%set_host()
|
||||
|
||||
|
||||
@@ -3104,10 +3104,10 @@ subroutine psb_z_mv_csr_from_coo(a,b,info)
|
||||
character(len=20) :: name='mv_from_coo'
|
||||
logical :: use_openmp = .false.
|
||||
|
||||
!$ integer(psb_ipk_), allocatable :: sum(:)
|
||||
!$ integer(psb_ipk_) :: first_idx,last_idx,work,ithread,nthreads,s
|
||||
!$ integer(psb_ipk_) :: nxt_val,old_val,saved_elem
|
||||
!$ use_openmp = .true.
|
||||
! $ integer(psb_ipk_), allocatable :: sum(:)
|
||||
! $ integer(psb_ipk_) :: first_idx,last_idx,work,ithread,nthreads,s
|
||||
! $ integer(psb_ipk_) :: nxt_val,old_val,saved_elem
|
||||
! $ use_openmp = .true.
|
||||
|
||||
|
||||
info = psb_success_
|
||||
@@ -3135,74 +3135,74 @@ subroutine psb_z_mv_csr_from_coo(a,b,info)
|
||||
|
||||
a%irp(:) = 0
|
||||
|
||||
if (use_openmp) then
|
||||
!$OMP PARALLEL default(none) &
|
||||
!$OMP shared(sum,nthreads,nr,a,itemp,nza) &
|
||||
!$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
|
||||
!$OMP DO schedule(STATIC) &
|
||||
!$OMP private(k,i)
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
end do
|
||||
!$OMP END DO
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ nthreads = omp_get_num_threads()
|
||||
!$ allocate(sum(nthreads+1))
|
||||
!$ sum(:) = 0
|
||||
!$ sum(1) = 1
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ ithread = omp_get_thread_num()
|
||||
|
||||
!$ work = nr/nthreads
|
||||
!$ if (ithread < MOD(nr,nthreads)) then
|
||||
!$ work = work + 1
|
||||
!$ first_idx = ithread*work + 1
|
||||
!$ else
|
||||
!$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!$ end if
|
||||
|
||||
!$ last_idx = first_idx + work - 1
|
||||
|
||||
!$ s = 0
|
||||
!$ do i=first_idx,last_idx
|
||||
!$ s = s + a%irp(i)
|
||||
!$ end do
|
||||
!$ if (work > 0) then
|
||||
!$ sum(ithread+2) = s
|
||||
!$ end if
|
||||
|
||||
!$OMP BARRIER
|
||||
|
||||
!$OMP SINGLE
|
||||
!$ do i=2,nthreads+1
|
||||
!$ sum(i) = sum(i) + sum(i-1)
|
||||
!$ end do
|
||||
!$OMP END SINGLE
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ saved_elem = a%irp(first_idx)
|
||||
!$ end if
|
||||
!$ if (ithread == 0) then
|
||||
!$ a%irp(1) = 1
|
||||
!$ end if
|
||||
|
||||
!$ if (work > 0) then
|
||||
!$ old_val = a%irp(first_idx+1)
|
||||
!$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!$ end if
|
||||
|
||||
!$ do i=first_idx+2,last_idx+1
|
||||
!$ nxt_val = a%irp(i)
|
||||
!$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!$ old_val = nxt_val
|
||||
!$ end do
|
||||
|
||||
!$OMP END PARALLEL
|
||||
else
|
||||
!!$ if (use_openmp) then
|
||||
!!$ !$OMP PARALLEL default(none) &
|
||||
!!$ !$OMP shared(sum,nthreads,nr,a,itemp,nza) &
|
||||
!!$ !$OMP private(ithread,work,first_idx,last_idx,s,saved_elem,nxt_val,old_val)
|
||||
!!$
|
||||
!!$ !$OMP DO schedule(STATIC) &
|
||||
!!$ !$OMP private(k,i)
|
||||
!!$ do k=1,nza
|
||||
!!$ i = itemp(k)
|
||||
!!$ a%irp(i) = a%irp(i) + 1
|
||||
!!$ end do
|
||||
!!$ !$OMP END DO
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ nthreads = omp_get_num_threads()
|
||||
!!$ !$ allocate(sum(nthreads+1))
|
||||
!!$ !$ sum(:) = 0
|
||||
!!$ !$ sum(1) = 1
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ ithread = omp_get_thread_num()
|
||||
!!$
|
||||
!!$ !$ work = nr/nthreads
|
||||
!!$ !$ if (ithread < MOD(nr,nthreads)) then
|
||||
!!$ !$ work = work + 1
|
||||
!!$ !$ first_idx = ithread*work + 1
|
||||
!!$ !$ else
|
||||
!!$ !$ first_idx = ithread*work + MOD(nr,nthreads) + 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ last_idx = first_idx + work - 1
|
||||
!!$
|
||||
!!$ !$ s = 0
|
||||
!!$ !$ do i=first_idx,last_idx
|
||||
!!$ !$ s = s + a%irp(i)
|
||||
!!$ !$ end do
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ sum(ithread+2) = s
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$OMP BARRIER
|
||||
!!$
|
||||
!!$ !$OMP SINGLE
|
||||
!!$ !$ do i=2,nthreads+1
|
||||
!!$ !$ sum(i) = sum(i) + sum(i-1)
|
||||
!!$ !$ end do
|
||||
!!$ !$OMP END SINGLE
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ saved_elem = a%irp(first_idx)
|
||||
!!$ !$ end if
|
||||
!!$ !$ if (ithread == 0) then
|
||||
!!$ !$ a%irp(1) = 1
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ if (work > 0) then
|
||||
!!$ !$ old_val = a%irp(first_idx+1)
|
||||
!!$ !$ a%irp(first_idx+1) = saved_elem + sum(ithread+1)
|
||||
!!$ !$ end if
|
||||
!!$
|
||||
!!$ !$ do i=first_idx+2,last_idx+1
|
||||
!!$ !$ nxt_val = a%irp(i)
|
||||
!!$ !$ a%irp(i) = a%irp(i-1) + old_val
|
||||
!!$ !$ old_val = nxt_val
|
||||
!!$ !$ end do
|
||||
!!$
|
||||
!!$ !$OMP END PARALLEL
|
||||
!!$ else
|
||||
do k=1,nza
|
||||
i = itemp(k)
|
||||
a%irp(i) = a%irp(i) + 1
|
||||
@@ -3214,7 +3214,7 @@ subroutine psb_z_mv_csr_from_coo(a,b,info)
|
||||
ip = ip + ncl
|
||||
end do
|
||||
a%irp(nr+1) = ip
|
||||
end if
|
||||
!!$ end if
|
||||
|
||||
call a%set_host()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user