mirror of
https://github.com/sfilippone/psblas3.git
synced 2026-10-06 22:55:08 +00:00
Updated docs
This commit is contained in:
Binary file not shown.
|
After Width: | Height: | Size: 29 KiB |
+19602
-12929
File diff suppressed because one or more lines are too long
+3
-2
@@ -86,7 +86,8 @@
|
||||
TOPFILE = userguide.tex
|
||||
HTMLFILE = userhtml.tex
|
||||
SECFILE = intro.tex commrout.tex datastruct.tex psbrout.tex toolsrout.tex\
|
||||
methods.tex precs.tex penv.tex error.tex util.tex biblio.tex
|
||||
methods.tex precs.tex penv.tex error.tex util.tex biblio.tex \
|
||||
ext-intro.tex cuda.tex
|
||||
FIGDIR = figures
|
||||
|
||||
XPDFFLAGS =
|
||||
@@ -139,7 +140,7 @@ PDF = $(join $(BASEFILE),.pdf)
|
||||
PS = $(join $(BASEFILE),.ps)
|
||||
GXS = $(join $(BASEFILE),.gxs)
|
||||
GLX = $(join $(BASEFILE),.glx)
|
||||
TARGETPDF= ../psblas-3.8.pdf
|
||||
TARGETPDF= ../psblas-3.9.pdf
|
||||
BASEHTML = $(patsubst %.tex,%,$(HTMLFILE))
|
||||
HTML = $(join $(BASEHTML),.html)
|
||||
HTMLDIR = ../html
|
||||
|
||||
+21
-4
@@ -1,9 +1,5 @@
|
||||
|
||||
\begin{thebibliography}{99}
|
||||
\bibitem{DesPat:11}
|
||||
D.~Barbieri, V.~Cardellini, S.~Filippone and D.~Rouson
|
||||
{\em Design Patterns for Scientific Computations on Sparse Matrices},
|
||||
HPSS 2011, Algorithms and Programming Tools for Next-Generation High-Performance Scientific Software, Bordeaux, Sep. 2011
|
||||
|
||||
\bibitem{PARA04FOREST}
|
||||
G.~Bella, S.~Filippone, A.~De Maio and M.~Testa,
|
||||
@@ -154,6 +150,11 @@ Lawson, C., Hanson, R., Kincaid, D. and Krogh, F.,
|
||||
{\em Fortran 95/2003 explained.}
|
||||
{Oxford University Press}, 2004.
|
||||
%
|
||||
\bibitem{MRC:11}
|
||||
{Metcalf, M., Reid, J. and Cohen, M.}
|
||||
{\em Modern Fortran explained.}
|
||||
{Oxford University Press}, 2011.
|
||||
%
|
||||
%% \bibitem{DD2}
|
||||
%% B.~Smith, P.~Bjorstad and W.~Gropp,
|
||||
%% {\em Domain Decomposition: Parallel Multilevel Methods for Elliptic
|
||||
@@ -169,4 +170,20 @@ M.~Snir, S.~Otto, S.~Huss-Lederman, D.~Walker and J.~Dongarra,
|
||||
{\em MPI: The Complete Reference. Volume 1 - The MPI Core}, second edition,
|
||||
MIT Press, 1998.
|
||||
%
|
||||
|
||||
\bibitem{DesPat:11}
|
||||
D.~Barbieri, V.~Cardellini, S.~Filippone and D.~Rouson
|
||||
{\em Design Patterns for Scientific Computations on Sparse Matrices},
|
||||
HPSS 2011, Algorithms and Programming Tools for Next-Generation High-Performance Scientific Software, Bordeaux, Sep. 2011
|
||||
|
||||
\bibitem{CaFiRo:2014}
|
||||
{ Cardellini, V.}, { Filippone, S.}, { and} { Rouson, D.} 2014,
|
||||
Design patterns for sparse-matrix computations on hybrid {CPU/GPU}
|
||||
platforms,
|
||||
{\em Scientific Programming\/}~{\em 22,\/}~1, 1--19.
|
||||
\bibitem{OurTechRep}
|
||||
D.~Barbieri, V.~Cardellini, A.~Fanfarillo, S.~Filippone, Three storage formats
|
||||
for sparse matrices on {GPGPUs}, Tech. Rep. DICII RR-15.6, Universit\`a di
|
||||
Roma Tor Vergata (February 2015).
|
||||
|
||||
\end{thebibliography}
|
||||
|
||||
@@ -0,0 +1,244 @@
|
||||
|
||||
\subsection{CUDA-class extensions}
|
||||
|
||||
For computing with CUDA we define a dual memorization strategy in
|
||||
which each variable on the CPU (``host'') side has a GPU (``device'')
|
||||
side. When a GPU-type variable is initialized, the data contained is
|
||||
(usually) the same on both sides. Each operator invoked on the
|
||||
variable may change the data so that only the host side or the device
|
||||
side are up-to-date.
|
||||
|
||||
Keeping track of the updates to data in the variables is essential: we want
|
||||
to perform most computations on the GPU, but we cannot afford the time
|
||||
needed to move data between the host memory and the device memory
|
||||
because the bandwidth of the interconnection bus would become the main
|
||||
bottleneck of the computation. Thus, each and every computational
|
||||
routine in the library is built according to the following principles:
|
||||
\begin{itemize}
|
||||
\item If the data type being handled is {GPU}-enabled, make sure that
|
||||
its device copy is up to date, perform any arithmetic operation on
|
||||
the {GPU}, and if the data has been altered as a result, mark
|
||||
the main-memory copy as outdated.
|
||||
\item The main-memory copy is never updated unless this is requested
|
||||
by the user either
|
||||
\begin{description}
|
||||
\item[explicitly] by invoking a synchronization method;
|
||||
\item[implicitly] by invoking a method that involves other data items
|
||||
that are not {GPU}-enabled, e.g., by assignment ov a vector to a
|
||||
normal array.
|
||||
\end{description}
|
||||
\end{itemize}
|
||||
In this way, data items are put on the {GPU} memory ``on demand'' and
|
||||
remain there as long as ``normal'' computations are carried out.
|
||||
As an example, the following call to a matrix-vector product
|
||||
\begin{minted}[breaklines=true,bgcolor=bg,fontsize=\small]{fortran}
|
||||
call psb_spmm(alpha,a,x,beta,y,desc_a,info)
|
||||
\end{minted}
|
||||
will transparently and automatically be performed on the {GPU} whenever
|
||||
all three data inputs \fortinline|a|, \fortinline|x| and
|
||||
\fortinline|y| are {GPU}-enabled. If a program makes many such calls
|
||||
sequentially, then
|
||||
\begin{itemize}
|
||||
\item The first kernel invocation will find the data in main memory,
|
||||
and will copy it to the {GPU} memory, thus incurring a significant
|
||||
overhead; the result is however \emph{not} copied back, and
|
||||
therefore:
|
||||
\item Subsequent kernel invocations involving the same vector will
|
||||
find the data on the {GPU} side so that they will run at full
|
||||
speed.
|
||||
\end{itemize}
|
||||
For all invocations after the first the only data that will have to be
|
||||
transferred to/from the main memory will be the scalars \fortinline|alpha|
|
||||
and \fortinline|beta|, and the return code \fortinline|info|.
|
||||
|
||||
\begin{description}
|
||||
\item[Vectors:] The data type \fortinline|psb_T_vect_gpu| provides a
|
||||
GPU-enabled extension of the inner type \fortinline|psb_T_base_vect_type|,
|
||||
and must be used together with the other inner matrix type to make
|
||||
full use of the GPU computational capabilities;
|
||||
\item[CSR:] The data type \fortinline|psb_T_csrg_sparse_mat| provides an
|
||||
interface to the GPU version of CSR available in the NVIDIA CuSPARSE
|
||||
library;
|
||||
\item[HYB:] The data type \fortinline|psb_T_hybg_sparse_mat| provides an
|
||||
interface to the HYB GPU storage available in the NVIDIA CuSPARSE
|
||||
library. The internal structure is opaque, hence the host side is
|
||||
just CSR; the HYB data format is only available up to CUDA version
|
||||
10.
|
||||
\item[ELL:] The data type \fortinline|psb_T_elg_sparse_mat| provides an
|
||||
interface to the ELLPACK implementation from SPGPU;
|
||||
|
||||
\item[HLL:] The data type \fortinline|psb_T_hlg_sparse_mat| provides an
|
||||
interface to the Hacked ELLPACK implementation from SPGPU;
|
||||
\item[HDIA:] The data type \fortinline|psb_T_hdiag_sparse_mat| provides an
|
||||
interface to the Hacked DIAgonals implementation from SPGPU;
|
||||
\end{description}
|
||||
|
||||
|
||||
\section{CUDA Environment Routines}
|
||||
\label{sec:cudaenv}
|
||||
|
||||
\subsection*{psb\_cuda\_init --- Initializes PSBLAS-CUDA
|
||||
environment}
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_init}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
call psb_cuda_init(ctxt [, device])
|
||||
\end{minted}
|
||||
|
||||
This subroutine initializes the PSBLAS-CUDA environment.
|
||||
\begin{description}
|
||||
\item[Type:] Synchronous.
|
||||
\item[\bf On Entry ]
|
||||
\item[device] ID of CUDA device to attach to.\\
|
||||
Scope: {\bf local}.\\
|
||||
Type: {\bf optional}.\\
|
||||
Intent: {\bf in}.\\
|
||||
Specified as: an integer value. \
|
||||
Default: use \fortinline|mod(iam,ngpu)| where \fortinline|iam| is the calling
|
||||
process index and \fortinline|ngpu| is the total number of CUDA devices
|
||||
available on the current node.
|
||||
\end{description}
|
||||
|
||||
|
||||
{\par\noindent\large\bfseries Notes}
|
||||
\begin{enumerate}
|
||||
\item A call to this routine must precede any other PSBLAS-CUDA call.
|
||||
\end{enumerate}
|
||||
|
||||
\subsection*{psb\_cuda\_exit --- Exit from PSBLAS-CUDA
|
||||
environment}
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_exit}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
call psb_cuda_exit(ctxt)
|
||||
\end{minted}
|
||||
|
||||
This subroutine exits from the PSBLAS CUDA context.
|
||||
\begin{description}
|
||||
\item[Type:] Synchronous.
|
||||
\item[\bf On Entry ]
|
||||
\item[ctxt] the communication context identifying the virtual
|
||||
parallel machine.\\
|
||||
Scope: {\bf global}.\\
|
||||
Type: {\bf required}.\\
|
||||
Intent: {\bf in}.\\
|
||||
Specified as: an integer variable.
|
||||
\end{description}
|
||||
|
||||
|
||||
|
||||
|
||||
\subsection*{psb\_cuda\_DeviceSync --- Synchronize CUDA device}
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_DeviceSync}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
call psb_cuda_DeviceSync()
|
||||
\end{minted}
|
||||
|
||||
This subroutine ensures that all previosly invoked kernels, i.e. all
|
||||
invocation of CUDA-side code, have completed.
|
||||
|
||||
|
||||
\subsection*{psb\_cuda\_getDeviceCount }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_getDeviceCount}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
ngpus = psb_cuda_getDeviceCount()
|
||||
\end{minted}
|
||||
|
||||
Get number of devices available on current computing node.
|
||||
|
||||
\subsection*{psb\_cuda\_getDevice }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_getDevice}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
ngpus = psb_cuda_getDevice()
|
||||
\end{minted}
|
||||
|
||||
Get device in use by current process.
|
||||
|
||||
\subsection*{psb\_cuda\_setDevice }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_setDevice}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
info = psb_cuda_setDevice(dev)
|
||||
\end{minted}
|
||||
|
||||
Set device to be used by current process.
|
||||
|
||||
\subsection*{psb\_cuda\_DeviceHasUVA }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_DeviceHasUVA}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
hasUva = psb_cuda_DeviceHasUVA()
|
||||
\end{minted}
|
||||
|
||||
Returns true if device currently in use supports UVA (Unified Virtual Addressing).
|
||||
|
||||
\subsection*{psb\_cuda\_WarpSize }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_WarpSize}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
nw = psb_cuda_WarpSize()
|
||||
\end{minted}
|
||||
|
||||
Returns the warp size.
|
||||
|
||||
|
||||
\subsection*{psb\_cuda\_MultiProcessors }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_MultiProcessors}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
nmp = psb_cuda_MultiProcessors()
|
||||
\end{minted}
|
||||
|
||||
Returns the number of multiprocessors in the CUDA device.
|
||||
|
||||
\subsection*{psb\_cuda\_MaxThreadsPerMP }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_MaxThreadsPerMP}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
nt = psb_cuda_MaxThreadsPerMP()
|
||||
\end{minted}
|
||||
|
||||
Returns the maximum number of threads per multiprocessor.
|
||||
|
||||
|
||||
\subsection*{psb\_cuda\_MaxRegistersPerBlock }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_MaxRegisterPerBlock}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
nr = psb_cuda_MaxRegistersPerBlock()
|
||||
\end{minted}
|
||||
|
||||
Returns the maximum number of register per thread block.
|
||||
|
||||
|
||||
\subsection*{psb\_cuda\_MemoryClockRate }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_MemoryClockRate}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
cl = psb_cuda_MemoryClockRate()
|
||||
\end{minted}
|
||||
|
||||
Returns the memory clock rate in KHz, as an integer.
|
||||
|
||||
\subsection*{psb\_cuda\_MemoryBusWidth }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_MemoryBusWidth}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
nb = psb_cuda_MemoryBusWidth()
|
||||
\end{minted}
|
||||
|
||||
Returns the memory bus width in bits.
|
||||
|
||||
\subsection*{psb\_cuda\_MemoryPeakBandwidth }
|
||||
\addcontentsline{toc}{subsection}{psb\_cuda\_MemoryPeakBandwidth}
|
||||
|
||||
\begin{minted}[breaklines=true]{fortran}
|
||||
bw = psb_cuda_MemoryPeakBandwidth()
|
||||
\end{minted}
|
||||
Returns the peak memory bandwidth in MB/s (real double precision).
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,412 @@
|
||||
\section{Extensions}\label{sec:ext-intro}
|
||||
|
||||
The EXT, CUDA and RSB subdirectories contains a set of extensions to the base
|
||||
library. The extensions provide additional storage formats beyond the
|
||||
ones already contained in the base library, as well as interfaces
|
||||
to:
|
||||
\begin{description}
|
||||
\item[SPGPU] a CUDA library originally published as
|
||||
\url{https://code.google.com/p/spgpu/} and now included in the
|
||||
\verb|cuda| subdir, for computations on NVIDIA GPUs;
|
||||
\item[LIBRSB] \url{http://sourceforge.net/projects/librsb/}, for
|
||||
computations on multicore parallel machines.
|
||||
\end{description}
|
||||
The infrastructure laid out in the base library to allow for these
|
||||
extensions is detailed in the references~\cite{DesPat:11,CaFiRo:2014,Sparse03};
|
||||
the CUDA-specific data formats are described in~\cite{OurTechRep}.
|
||||
|
||||
|
||||
\subsection{Using the extensions}
|
||||
\label{sec:ext-appstruct}
|
||||
A sample application using the PSBLAS extensions will contain the
|
||||
following steps:
|
||||
\begin{itemize}
|
||||
\item \verb|USE| the appropriat modules (\verb|psb_ext_mod|,
|
||||
\verb|psb_cuda_mod|);
|
||||
\item Declare a \emph{mold} variable of the necessary type
|
||||
(e.g. \verb|psb_d_ell_sparse_mat|, \verb|psb_d_hlg_sparse_mat|,
|
||||
\verb|psb_d_vect_cuda|);
|
||||
\item Pass the mold variable to the base library interface where
|
||||
needed to ensure the appropriate dynamic type.
|
||||
\end{itemize}
|
||||
Suppose you want to use the CUDA-enabled ELLPACK data structure; you
|
||||
would use a piece of code like this (and don't forget, you need
|
||||
CUDA-side vectors along with the matrices):
|
||||
\begin{minted}[breaklines=true,bgcolor=bg,fontsize=\small]{fortran}
|
||||
program my_cuda_test
|
||||
use psb_base_mod
|
||||
use psb_util_mod
|
||||
use psb_ext_mod
|
||||
use psb_cuda_mod
|
||||
type(psb_dspmat_type) :: a, agpu
|
||||
type(psb_d_vect_type) :: x, xg, bg
|
||||
|
||||
real(psb_dpk_), allocatable :: xtmp(:)
|
||||
type(psb_d_vect_cuda) :: vmold
|
||||
type(psb_d_elg_sparse_mat) :: aelg
|
||||
type(psb_ctxt_type) :: ctxt
|
||||
integer :: iam, np
|
||||
|
||||
|
||||
call psb_init(ctxt)
|
||||
call psb_info(ctxt,iam,np)
|
||||
call psb_cuda_init(ctxt, iam)
|
||||
|
||||
|
||||
! My own home-grown matrix generator
|
||||
call gen_matrix(ctxt,idim,desc_a,a,x,info)
|
||||
if (info /= 0) goto 9999
|
||||
|
||||
call a%cscnv(agpu,info,mold=aelg)
|
||||
if (info /= 0) goto 9999
|
||||
xtmp = x%get_vect()
|
||||
call xg%bld(xtmp,mold=vmold)
|
||||
call bg%bld(size(xtmp),mold=vmold)
|
||||
|
||||
! Do sparse MV
|
||||
call psb_spmm(done,agpu,xg,dzero,bg,desc_a,info)
|
||||
|
||||
|
||||
9999 continue
|
||||
if (info == 0) then
|
||||
write(*,*) '42'
|
||||
else
|
||||
write(*,*) 'Something went wrong ',info
|
||||
end if
|
||||
|
||||
|
||||
call psb_cuda_exit()
|
||||
call psb_exit(ctxt)
|
||||
stop
|
||||
end program my_cuda_test
|
||||
\end{minted}
|
||||
A full example of this strategy can be seen in the
|
||||
\texttt{test/ext/kernel} and \texttt{test/\-cuda/\-kernel} subdirectories,
|
||||
where we provide sample programs
|
||||
to test the speed of the sparse matrix-vector product with the various
|
||||
data structures included in the library.
|
||||
|
||||
|
||||
\subsection{Extensions' Data Structures}
|
||||
\label{sec:ext-datastruct}
|
||||
%\ifthenelse{\boolean{mtc}}{\minitoc}{}
|
||||
|
||||
Access to the facilities provided by the EXT library is mainly
|
||||
achieved through the data types that are provided within.
|
||||
The data classes are derived from the base classes in PSBLAS, through
|
||||
the Fortran~2003 mechanism of \emph{type extension}~\cite{MRC:11}.
|
||||
|
||||
The data classes are divided between the general purpose CPU
|
||||
extensions, the GPU interfaces and the RSB interfaces.
|
||||
In the description we will make use of the notation introduced in
|
||||
Table~\ref{tab:notation}.
|
||||
|
||||
\begin{table}[ht]
|
||||
\caption{Notation for parameters describing a sparse matrix}
|
||||
\begin{center}
|
||||
{\footnotesize
|
||||
\begin{tabular}{ll}
|
||||
\hline
|
||||
Name & Description \\
|
||||
\hline
|
||||
M & Number of rows in matrix \\
|
||||
N & Number of columns in matrix \\
|
||||
NZ & Number of nonzeros in matrix \\
|
||||
AVGNZR & Average number of nonzeros per row \\
|
||||
MAXNZR & Maximum number of nonzeros per row \\
|
||||
NDIAG & Numero of nonzero diagonals\\
|
||||
AS & Coefficients array \\
|
||||
IA & Row indices array \\
|
||||
JA & Column indices array \\
|
||||
IRP & Row start pointers array \\
|
||||
JCP & Column start pointers array \\
|
||||
NZR & Number of nonzeros per row array \\
|
||||
OFFSET & Offset for diagonals \\
|
||||
\hline
|
||||
\end{tabular}
|
||||
}
|
||||
\end{center}
|
||||
\label{tab:notation}
|
||||
\end{table}
|
||||
|
||||
\begin{figure}[ht]
|
||||
\centering
|
||||
% \includegraphics[width=5.2cm]{figures/mat.eps}
|
||||
\includegraphics[width=5.2cm]{figures/mat.pdf}
|
||||
\caption{Example of sparse matrix}
|
||||
\label{fig:dense}
|
||||
\end{figure}
|
||||
|
||||
\subsection{CPU-class extensions}
|
||||
|
||||
|
||||
\subsubsection*{ELLPACK}
|
||||
|
||||
The ELLPACK/ITPACK format (shown in Figure~\ref{fig:ell})
|
||||
comprises two 2-dimensional arrays \verb|AS| and
|
||||
\verb|JA| with \verb|M| rows and \verb|MAXNZR| columns, where
|
||||
\verb|MAXNZR| is the maximum
|
||||
number of nonzeros in any row~\cite{ELLPACK}.
|
||||
Each row of the arrays \verb|AS| and \verb|JA| contains the
|
||||
coefficients and column indices; rows shorter than
|
||||
\verb|MAXNZR| are padded with zero coefficients and appropriate column
|
||||
indices, e.g. the last valid one found in the same row.
|
||||
|
||||
\begin{figure}[ht]
|
||||
\centering
|
||||
% \includegraphics[width=8.2cm]{figures/ell.eps}
|
||||
\includegraphics[width=8.2cm]{figures/ell.pdf}
|
||||
\caption{ELLPACK compression of matrix in Figure~\ref{fig:dense}}
|
||||
\label{fig:ell}
|
||||
\end{figure}
|
||||
|
||||
|
||||
\begin{algorithm}
|
||||
\lstset{language=Fortran}
|
||||
\small
|
||||
\begin{lstlisting}
|
||||
do i=1,n
|
||||
t=0
|
||||
do j=1,maxnzr
|
||||
t = t + as(i,j)*x(ja(i,j))
|
||||
end do
|
||||
y(i) = t
|
||||
end do
|
||||
\end{lstlisting}
|
||||
\caption{\label{alg:ell} Matrix-Vector product in ELL format}
|
||||
\end{algorithm}
|
||||
The matrix-vector product $y=Ax$ can be computed with the code shown in
|
||||
Alg.~\ref{alg:ell}; it costs one memory write per outer iteration,
|
||||
plus three memory reads and two floating-point operations per inner
|
||||
iteration.
|
||||
|
||||
Unless all rows have exactly the same number of nonzeros, some of the
|
||||
coefficients in the \verb|AS| array will be zeros; therefore this
|
||||
data structure will have an overhead both in terms of memory space
|
||||
and redundant operations (multiplications by zero). The overhead can
|
||||
be acceptable if:
|
||||
\begin{enumerate}
|
||||
\item The maximum number of nonzeros per row is not much larger than
|
||||
the average;
|
||||
\item The regularity of the data structure allows for faster code,
|
||||
e.g. by allowing vectorization, thereby offsetting the additional
|
||||
storage requirements.
|
||||
\end{enumerate}
|
||||
In the extreme case where the input matrix has one full row, the
|
||||
ELLPACK structure would require more memory than the normal 2D array
|
||||
storage. The ELLPACK storage format was very popular in the vector
|
||||
computing days; in modern CPUs it is not quite as popular, but it
|
||||
is the basis for many GPU formats.
|
||||
|
||||
The relevant data type is \verb|psb_T_ell_sparse_mat|:
|
||||
\begin{minted}[breaklines=true,bgcolor=bg,fontsize=\small]{fortran}
|
||||
type, extends(psb_d_base_sparse_mat) :: psb_d_ell_sparse_mat
|
||||
!
|
||||
! ITPACK/ELL format, extended.
|
||||
!
|
||||
|
||||
integer(psb_ipk_), allocatable :: irn(:), ja(:,:), idiag(:)
|
||||
real(psb_dpk_), allocatable :: val(:,:)
|
||||
|
||||
contains
|
||||
....
|
||||
end type psb_d_ell_sparse_mat
|
||||
\end{minted}
|
||||
|
||||
|
||||
\subsubsection*{Hacked ELLPACK}
|
||||
|
||||
The \textit{hacked ELLPACK} (\textbf{HLL}) format
|
||||
alleviates the main problem of the ELLPACK format, that is,
|
||||
the amount of memory required by padding for sparse matrices in
|
||||
which the maximum row length is larger than the average.
|
||||
|
||||
The number of elements allocated to padding is $[(m*maxNR) -
|
||||
(m*avgNR) = m*(maxNR-avgNR)]$
|
||||
for both \verb|AS| and \verb|JA| arrays,
|
||||
where $m$ is equal to the number of rows of the matrix, $maxNR$ is the
|
||||
maximum number of nonzero elements
|
||||
in every row and $avgNR$ is the average number of nonzeros.
|
||||
Therefore a single densely populated row can seriously affect the
|
||||
total size of the allocation.
|
||||
|
||||
To limit this effect, in the HLL format we break the original matrix
|
||||
into equally sized groups of rows (called \textit{hacks}), and then store
|
||||
these groups as independent matrices in ELLPACK format.
|
||||
The groups can be arranged selecting rows in an arbitrarily manner;
|
||||
indeed, if the rows are sorted by decreasing number of nonzeros we
|
||||
obtain essentially the JAgged Diagonals format.
|
||||
If the rows are not in the original order, then an additional vector
|
||||
\textit{rIdx} is required, storing the actual row index for each row
|
||||
in the data structure.
|
||||
|
||||
The multiple ELLPACK-like buffers are stacked together inside a
|
||||
single, one dimensional array;
|
||||
an additional vector \textit{hackOffsets} is provided to keep track
|
||||
of the individual submatrices.
|
||||
All hacks have the same number of rows \textit{hackSize}; hence,
|
||||
the \textit{hackOffsets} vector is an array of
|
||||
$(m/hackSize)+1$ elements, each one pointing to the first index of a
|
||||
submatrix inside the stacked \textit{cM}/\textit{rP} buffers, plus an
|
||||
additional element pointing past the end of the last block, where the
|
||||
next one would begin.
|
||||
We thus have the property that
|
||||
the elements of the $k$-th \textit{hack} are stored between \verb|hackOffsets[k]| and
|
||||
\verb|hackOffsets[k+1]|, similarly to what happens in the CSR format.
|
||||
|
||||
\begin{figure}[ht]
|
||||
\centering
|
||||
% \includegraphics[width=8.2cm]{../figures/hll.eps}
|
||||
\includegraphics[width=.72\textwidth]{../figures/hll.pdf}
|
||||
\caption{Hacked ELLPACK compression of matrix in Figure~\ref{fig:dense}}
|
||||
\label{fig:hll}
|
||||
\end{figure}
|
||||
|
||||
With this data structure a very long row only affects one hack, and
|
||||
therefore the additional memory is limited to the hack in which the
|
||||
row appears.
|
||||
|
||||
The relevant data type is \verb|psb_T_hll_sparse_mat|:
|
||||
\begin{minted}[breaklines=true,bgcolor=bg,fontsize=\small]{fortran}
|
||||
type, extends(psb_d_base_sparse_mat) :: psb_d_hll_sparse_mat
|
||||
!
|
||||
! HLL format. (Hacked ELL)
|
||||
!
|
||||
integer(psb_ipk_) :: hksz
|
||||
integer(psb_ipk_), allocatable :: irn(:), ja(:), idiag(:), hkoffs(:)
|
||||
real(psb_dpk_), allocatable :: val(:)
|
||||
|
||||
contains
|
||||
....
|
||||
end type
|
||||
\end{minted}
|
||||
|
||||
\subsubsection*{Diagonal storage}
|
||||
|
||||
|
||||
The DIAgonal (DIA) format (shown in Figure~\ref{fig:dia})
|
||||
has a 2-dimensional array \verb|AS| containing in each column the
|
||||
coefficients along a diagonal of the matrix, and an integer array
|
||||
\verb|OFFSET| that determines where each diagonal starts. The
|
||||
diagonals in \verb|AS| are padded with zeros as necessary.
|
||||
|
||||
The code to compute the matrix-vector product $y=Ax$ is shown in Alg.~\ref{alg:dia};
|
||||
it costs one memory read per outer iteration,
|
||||
plus three memory reads, one memory write and two floating-point
|
||||
operations per inner iteration. The accesses to \verb|AS| and
|
||||
\verb|x| are in strict sequential order, therefore no indirect
|
||||
addressing is required.
|
||||
|
||||
\begin{figure}[ht]
|
||||
\centering
|
||||
% \includegraphics[width=8.2cm]{figures/dia.eps}
|
||||
\includegraphics[width=.72\textwidth]{figures/dia.pdf}
|
||||
\caption{DIA compression of matrix in Figure~\ref{fig:dense}}
|
||||
\label{fig:dia}
|
||||
\end{figure}
|
||||
|
||||
|
||||
\begin{algorithm}
|
||||
\begin{minted}[breaklines=true,bgcolor=bg,fontsize=\small]{fortran}
|
||||
do j=1,ndiag
|
||||
if (offset(j) > 0) then
|
||||
ir1 = 1; ir2 = m - offset(j);
|
||||
else
|
||||
ir1 = 1 - offset(j); ir2 = m;
|
||||
end if
|
||||
do i=ir1,ir2
|
||||
y(i) = y(i) + alpha*as(i,j)*x(i+offset(j))
|
||||
end do
|
||||
end do
|
||||
\end{minted}
|
||||
\caption{\label{alg:dia} Matrix-Vector product in DIA format}
|
||||
\end{algorithm}
|
||||
|
||||
|
||||
The relevant data type is \verb|psb_T_dia_sparse_mat|:
|
||||
\begin{minted}[breaklines=true,bgcolor=bg,fontsize=\small]{fortran}
|
||||
type, extends(psb_d_base_sparse_mat) :: psb_d_dia_sparse_mat
|
||||
!
|
||||
! DIA format, extended.
|
||||
!
|
||||
|
||||
integer(psb_ipk_), allocatable :: offset(:)
|
||||
integer(psb_ipk_) :: nzeros
|
||||
real(psb_dpk_), allocatable :: data(:,:)
|
||||
|
||||
end type
|
||||
\end{minted}
|
||||
|
||||
|
||||
|
||||
\subsubsection*{Hacked DIA}
|
||||
|
||||
Storage by DIAgonals is an attractive option for matrices whose
|
||||
coefficients are located on a small set of diagonals, since they do
|
||||
away with storing explicitly the indices and therefore reduce
|
||||
significantly memory traffic. However, having a few coefficients
|
||||
outside of the main set of diagonals may significantly increase the
|
||||
amount of needed padding; moreover, while the DIA code is easily
|
||||
vectorized, it does not necessarily make optimal use of the memory
|
||||
hierarchy. While processing each diagonal we are updating entries in
|
||||
the output vector \verb|y|, which is then accessed multiple times; if
|
||||
the vector \verb|y| is too large to remain in the cache memory, the
|
||||
associated cache miss penalty is paid multiple times.
|
||||
|
||||
The \textit{hacked DIA} (\textbf{HDIA}) format was designed to contain
|
||||
the amount of padding, by breaking the original matrix
|
||||
into equally sized groups of rows (\textit{hacks}), and then storing
|
||||
these groups as independent matrices in DIA format. This approach is
|
||||
similar to that of HLL, and requires using an offset vector for each
|
||||
submatrix. Again, similarly to HLL, the various submatrices are
|
||||
stacked inside a linear array to improve memory management. The fact
|
||||
that the matrix is accessed in slices helps in reducing cache misses,
|
||||
especially regarding accesses to the %output
|
||||
vector \verb|y|.
|
||||
|
||||
|
||||
An additional vector \textit{hackOffsets} is provided to complete
|
||||
the matrix format; given that \textit{hackSize} is the number of rows of each hack,
|
||||
the \textit{hackOffsets} vector is made by an array of
|
||||
$(m/hackSize)+1$ elements, pointing to the first diagonal offset of a
|
||||
submatrix inside the stacked \textit{offsets} buffers, plus an
|
||||
additional element equal to the number of nonzero diagonals in the whole matrix.
|
||||
We thus have the property that
|
||||
the number of diagonals of the $k$-th \textit{hack} is given by
|
||||
\textit{hackOffsets[k+1] - hackOffsets[k]}.
|
||||
|
||||
\begin{figure}[ht]
|
||||
\centering
|
||||
% \includegraphics[width=8.2cm]{../figures/hdia.eps}
|
||||
\includegraphics[width=.72\textwidth]{../figures/hdia.pdf}
|
||||
\caption{Hacked DIA compression of matrix in Figure~\ref{fig:dense}}
|
||||
\label{fig:hdia}
|
||||
\end{figure}
|
||||
|
||||
The relevant data type is \verb|psb_T_hdia_sparse_mat|:
|
||||
\begin{minted}[breaklines=true,bgcolor=bg,fontsize=\small]{fortran}
|
||||
type pm
|
||||
real(psb_dpk_), allocatable :: data(:,:)
|
||||
end type pm
|
||||
|
||||
type po
|
||||
integer(psb_ipk_), allocatable :: off(:)
|
||||
end type po
|
||||
|
||||
type, extends(psb_d_base_sparse_mat) :: psb_d_hdia_sparse_mat
|
||||
!
|
||||
! HDIA format, extended.
|
||||
!
|
||||
|
||||
type(pm), allocatable :: hdia(:)
|
||||
type(po), allocatable :: offset(:)
|
||||
integer(psb_ipk_) :: nblocks, nzeros
|
||||
integer(psb_ipk_) :: hack = 64
|
||||
integer(psb_long_int_k_) :: dim=0
|
||||
|
||||
contains
|
||||
....
|
||||
end type
|
||||
\end{minted}
|
||||
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
|
After Width: | Height: | Size: 29 KiB |
Symlink
+1
@@ -0,0 +1 @@
|
||||
tmp/userguide.pdf
|
||||
@@ -17,6 +17,7 @@
|
||||
\newtheorem{theorem}{Theorem}
|
||||
\newtheorem{corollary}{Corollary}
|
||||
\usepackage{listings}
|
||||
\usepackage{algorithm2e}
|
||||
\usepackage{minted}
|
||||
\usemintedstyle{friendly}
|
||||
\definecolor{bg}{rgb}{0.95,0.95,0.95}
|
||||
@@ -36,7 +37,7 @@
|
||||
\relax
|
||||
\pdfcompresslevel=0 %-- 0 = none, 9 = best
|
||||
\pdfinfo{ %-- Info dictionary of PDF output /Author (Alfredo Buttari)
|
||||
/Title (Parallel Sparse BLAS V. 3.8.0)
|
||||
/Title (Parallel Sparse BLAS V. 3.9.0)
|
||||
/Subject (Parallel Sparse Basic Linear Algebra Subroutines)
|
||||
/Keywords (Computer Science Linear Algebra Fluid Dynamics Parallel Linux MPI PSBLAS Iterative Solvers Preconditioners)
|
||||
/Creator (pdfLaTeX)
|
||||
@@ -99,7 +100,7 @@
|
||||
|
||||
\begin{document}
|
||||
{
|
||||
\pdfbookmark{PSBLAS-v3.8.0 User's Guide}{title}
|
||||
\pdfbookmark{PSBLAS-v3.9.0 User's Guide}{title}
|
||||
\lstset{language=Fortran}
|
||||
\newlength{\centeroffset}
|
||||
\setlength{\centeroffset}{-0.5\oddsidemargin}
|
||||
@@ -109,7 +110,7 @@
|
||||
\vspace*{\stretch{1}}
|
||||
\noindent\hspace*{\centeroffset}\makebox[0pt][l]{\begin{minipage}{\textwidth}
|
||||
\flushright
|
||||
{\Huge\bfseries PSBLAS 3.8.0 User's guide
|
||||
{\Huge\bfseries PSBLAS 3.9.0 User's guide
|
||||
}
|
||||
\noindent\rule[-1ex]{\textwidth}{5pt}\\[2.5ex]
|
||||
\hfill\emph{\Large A reference guide for the Parallel Sparse BLAS library}
|
||||
@@ -130,7 +131,7 @@
|
||||
{\bfseries
|
||||
by Salvatore Filippone\\
|
||||
and Alfredo Buttari}\\
|
||||
May 1st, 2022
|
||||
Aug 1st, 2024
|
||||
\end{minipage}}
|
||||
}
|
||||
%\addtolength{\textwidth}{\centeroffset}
|
||||
@@ -159,7 +160,8 @@ May 1st, 2022
|
||||
\include{util}
|
||||
\include{precs}
|
||||
\include{methods}
|
||||
|
||||
\include{ext-intro}
|
||||
\include{cuda}
|
||||
\cleardoublepage
|
||||
\input{biblio}
|
||||
|
||||
|
||||
@@ -17,6 +17,9 @@
|
||||
\newtheorem{theorem}{Theorem}
|
||||
\newtheorem{corollary}{Corollary}
|
||||
\usepackage{listings}
|
||||
\usepackage{algorithm2e}
|
||||
\usepackage{minted}
|
||||
\usemintedstyle{friendly}
|
||||
\usepackage{microtype}
|
||||
\ifpdf
|
||||
\newmintinline[fortinline]{fortran}{}
|
||||
@@ -94,9 +97,9 @@
|
||||
Alfredo Buttari } \\
|
||||
%\\[10ex]
|
||||
%\today
|
||||
Software version: 3.8.0\\
|
||||
Software version: 3.9.0\\
|
||||
%\today
|
||||
May 1st, 2022
|
||||
Aug 1st, 2024
|
||||
\cleardoublepage
|
||||
\begingroup
|
||||
\renewcommand*{\thepage}{toc}
|
||||
@@ -120,7 +123,8 @@ May 1st, 2022
|
||||
\include{util}
|
||||
\include{precs}
|
||||
\include{methods}
|
||||
|
||||
\include{ext-intro}
|
||||
\include{cuda}
|
||||
\cleardoublepage
|
||||
|
||||
\input{biblio}
|
||||
|
||||
Reference in New Issue
Block a user