From 9168d450ce68670e8448f30ead70e259af65d548 Mon Sep 17 00:00:00 2001 From: Stack-1 Date: Tue, 11 Aug 2026 10:44:53 +0200 Subject: [PATCH] [ADD] Added scripts to collect data --- samples/advanced/pdegen/amg_comm_scaling.sh | 185 ++++++++++++++++++++ samples/advanced/pdegen/amg_comm_strong.sh | 106 ----------- samples/advanced/pdegen/collect_results.sh | 169 ++++++++++++++++++ 3 files changed, 354 insertions(+), 106 deletions(-) create mode 100755 samples/advanced/pdegen/amg_comm_scaling.sh delete mode 100755 samples/advanced/pdegen/amg_comm_strong.sh create mode 100755 samples/advanced/pdegen/collect_results.sh diff --git a/samples/advanced/pdegen/amg_comm_scaling.sh b/samples/advanced/pdegen/amg_comm_scaling.sh new file mode 100755 index 00000000..c366aa14 --- /dev/null +++ b/samples/advanced/pdegen/amg_comm_scaling.sh @@ -0,0 +1,185 @@ +#!/bin/bash +# ============================================================================ +# amg_comm_strong.sbatch STRONG scaling, communication schemes on AMG +# +# Fixed total problem size; MPI ranks grow across scale points. Every scale +# point runs inside ONE reserved allocation, so the queue is paid once. +# +# The executable sweeps, in a single invocation: +# * uniform the 5 schemes applied to the whole hierarchy +# * sensitivity baseline everywhere, one level at a time moved onto another +# scheme -- the marginal value of a scheme AT a level, which +# is what a per-level policy is built on +# so there is no per-scheme or per-level loop here, and no reason to come back +# for a second allocation. +# +# Both CSVs are written per scale point; nothing needs re-running to plot. +# ============================================================================ +#SBATCH --job-name=amg_comm_strong +#SBATCH --output=amg_comm_strong_%j.out +#SBATCH --error=amg_comm_strong_%j.err +#SBATCH --nodes=16 # reserve the maximum up front +#SBATCH --ntasks-per-node=112 # full GPP node: 2 x Sapphire Rapids 8480+ +#SBATCH --cpus-per-task=1 # CPU-only: one core per rank +#SBATCH --time=02:00:00 +#SBATCH --qos=gp_ehpc # production; gp_debug caps at 2h / 3584 proc +#SBATCH --account=ehpc859 # the GPP project; ehpc580 is ACC and over budget +# +# Verify the node width before trusting RANK_POINTS below: +# sinfo -o "%P %c" +# If a GPP node is not 112 cores, ntasks-per-node, RANK_POINTS and the NNODES +# arithmetic all have to move together. + +# ============================================================================ +# USER CONFIGURATION +# ============================================================================ +# STRONG or WEAK. They answer different questions and both are worth having: +# +# strong fixed problem, growing ranks. Matches the SpMV paper, and shows the +# knee where communication starts to dominate. Caveat: as ranks grow +# the AMG hierarchy itself changes shape -- fewer rows per rank at +# every level, coarse levels degenerating -- so a scale point differs +# from the next by more than just the rank count. +# +# weak fixed rows per rank, problem grown with the ranks. The hierarchy +# keeps its shape, so the rank count is isolated from the coarsening. +# This is the cleaner design for "how do the schemes behave as the +# machine grows", which is the question a per-level policy needs. +SCALING=${SCALING:-strong} + +NREP=7 # repetitions per configuration +NLEV=5 # max levels requested of the hierarchy +ITMAX=500 + +RANK_POINTS="112 224 448 896 1792" # 1, 2, 4, 8, 16 full nodes + +# strong: one size for every scale point. +# 400^3 = 64M unknowns; at 1792 ranks that is ~36k rows/rank on the fine level +# and still ~550 on level 3, so the hierarchy stays meaningful to the end. +DIM_STRONG=400 + +# weak: idim grows as ranks^(1/3) to hold rows/rank constant at ~36.5k, +# matching the strong run at its largest point so the two are comparable there. +DIM_WEAK="160 202 254 320 403" + +# Two passes are needed and they answer different questions: +# PROFILE=false clean timings -- the numbers that compare schemes +# PROFILE=true Score-P attribution -- where the time goes, per level +# The binary is built with scorep-mpifort, so profiling is ON unless disabled: +# a "clean" run is only clean if these are set explicitly. +PROFILE=${PROFILE:-false} + +EXE=$SLURM_SUBMIT_DIR/runs/amg_d_comm_test + +# ============================================================================ +# ENVIRONMENT +# ============================================================================ +module purge +module load bsc/1.0 +module load gcc/12.3.0 +module load ucx/1.16.0-gcc +module load openmpi/5.0.5-gcc +module load openblas/0.3.27-gcc + +export OMPI_MCA_coll_hcoll_enable=0 + +if [ "$PROFILE" = "true" ]; then + export SCOREP_ENABLE_PROFILING=true + export SCOREP_ENABLE_TRACING=false + export SCOREP_TOTAL_MEMORY=128M +else + export SCOREP_ENABLE_PROFILING=false + export SCOREP_ENABLE_TRACING=false +fi + +RESDIR=$SLURM_SUBMIT_DIR/results_amg_${SCALING}_${SLURM_JOB_ID} +mkdir -p $RESDIR + +# Region filter: exclude the high-frequency tiny helpers so the profile reflects +# real MPI time rather than instrumentation of the descriptor bookkeeping. +FILTER=$RESDIR/scorep.filt +cat > $FILTER <<'FILTEOF' +SCOREP_REGION_NAMES_BEGIN + EXCLUDE + psb_indx_map_mod::* + psb_desc_mod::* + psb_error_mod::* + psb_gen_block_map_mod::* + psi_penv_mod::* + psb_hash_mod::* +SCOREP_REGION_NAMES_END +FILTEOF + +case "$SCALING" in + strong) echo "=== AMG communication schemes, STRONG scaling (CPU-only) ===" ;; + weak) echo "=== AMG communication schemes, WEAK scaling (CPU-only) ===" ;; + *) echo "FATAL: SCALING must be 'strong' or 'weak', got '$SCALING'"; exit 1 ;; +esac +echo " nrep=$NREP max_levels=$NLEV itmax=$ITMAX" +echo " PROFILE=$PROFILE" +echo " reserved_nodes=$SLURM_NNODES rank_points=[$RANK_POINTS]" +echo " exe=$EXE" +echo "============================================================" + +if [ ! -x "$EXE" ]; then + echo "FATAL: $EXE not found or not executable. Build it before submitting:" + echo " cd samples/advanced/pdegen && make amg_d_comm_test" + exit 1 +fi + +FAILED=0 + +IDX=0 +for NRANKS in $RANK_POINTS; do + IDX=$((IDX+1)) + NNODES=$(( (NRANKS + 111) / 112 )) + + if [ "$SCALING" = "weak" ]; then + DIM=$(echo $DIM_WEAK | cut -d' ' -f$IDX) + else + DIM=$DIM_STRONG + fi + + STEP_DIR=$RESDIR/${NRANKS}ranks + mkdir -p $STEP_DIR + + echo "" + echo ">>> $SCALING point: $NRANKS ranks ($NNODES nodes), dim=$DIM" + + # Name the experiment after the scale point. Left to itself Score-P writes + # scorep-, which carries no indication of the rank count and has + # to be matched back by hand afterwards. + if [ "$PROFILE" = "true" ]; then + export SCOREP_EXPERIMENT_DIRECTORY=$STEP_DIR/scorep_${NRANKS}ranks + export SCOREP_FILTERING_FILE=$FILTER + fi + + srun -N $NNODES -n $NRANKS --ntasks-per-node=112 --cpus-per-task=1 \ + $EXE $DIM $NREP $NLEV $ITMAX \ + --mode=all --csv=$STEP_DIR/amg_comm.csv \ + > $STEP_DIR/run.out 2>&1 + RC=$? + echo ">>> exit=$RC output=$STEP_DIR/run.out" + + # Fail loudly rather than silently producing a dataset nobody can trust: + # if a scheme did not reach every level, the comparison is meaningless. + if ! grep -q "SCHEME PROPAGATION: OK" $STEP_DIR/run.out; then + echo "!!! WARNING: scheme propagation not confirmed at $NRANKS ranks" + FAILED=1 + fi + # All configurations must converge identically; a differing iteration count + # means the halo exchange changed the arithmetic, not just its schedule. + NITERS=$(grep -oE "it +[0-9]+" $STEP_DIR/run.out | awk '{print $2}' | sort -u | wc -l) + if [ "$NITERS" != "1" ]; then + echo "!!! WARNING: iteration count is not constant at $NRANKS ranks ($NITERS distinct values)" + FAILED=1 + fi +done + +echo "" +if [ "$FAILED" = "0" ]; then + echo "=== AMG COMM ${SCALING} DONE, all checks passed. Results: $RESDIR ===" +else + echo "=== AMG COMM ${SCALING} DONE WITH WARNINGS. Results: $RESDIR ===" +fi +find $RESDIR -name "*.csv" -printf " %p (%s bytes)\n" diff --git a/samples/advanced/pdegen/amg_comm_strong.sh b/samples/advanced/pdegen/amg_comm_strong.sh deleted file mode 100755 index fc321b38..00000000 --- a/samples/advanced/pdegen/amg_comm_strong.sh +++ /dev/null @@ -1,106 +0,0 @@ -#!/bin/bash -# ============================================================================ -# amg_comm_strong.sbatch STRONG scaling, communication schemes on AMG -# -# Fixed total problem size; MPI ranks grow across scale points. Every scale -# point runs inside ONE reserved allocation, so the queue is paid once. -# -# The executable sweeps, in a single invocation: -# * uniform the 5 schemes applied to the whole hierarchy -# * sensitivity baseline everywhere, one level at a time moved onto another -# scheme -- the marginal value of a scheme AT a level, which -# is what a per-level policy is built on -# so there is no per-scheme or per-level loop here, and no reason to come back -# for a second allocation. -# -# Both CSVs are written per scale point; nothing needs re-running to plot. -# ============================================================================ -#SBATCH --job-name=amg_comm_strong -#SBATCH --output=amg_comm_strong_%j.out -#SBATCH --error=amg_comm_strong_%j.err -#SBATCH --nodes=8 # reserve the maximum up front -#SBATCH --ntasks-per-node=80 -#SBATCH --cpus-per-task=1 # CPU-only: one core per rank -#SBATCH --time=01:00:00 -#SBATCH --qos=acc_debug -# Account intentionally not hardcoded. Pass it at submit time: -# sbatch -A amg_comm_strong.sh -# or export SBATCH_ACCOUNT= once in your cluster profile. - -# ============================================================================ -# USER CONFIGURATION -# ============================================================================ -DIM=200 # FIXED problem size (idim^3 unknowns) -NREP=7 # repetitions per configuration -NLEV=5 # max levels requested of the hierarchy -ITMAX=500 -RANK_POINTS="80 160 320 640" # total ranks per scale point (multiples of 80) - -EXE=$SLURM_SUBMIT_DIR/runs/amg_d_comm_test - -# ============================================================================ -# ENVIRONMENT -# ============================================================================ -module purge -module load bsc/1.0 -module load gcc/12.3.0 -module load ucx/1.16.0-gcc -module load openmpi/5.0.5-gcc -module load openblas/0.3.27-gcc - -export OMPI_MCA_coll_hcoll_enable=0 - -RESDIR=$SLURM_SUBMIT_DIR/results_amg_comm_${SLURM_JOB_ID} -mkdir -p $RESDIR - -echo "=== AMG communication schemes, STRONG scaling (CPU-only) ===" -echo " fixed_dim=$DIM nrep=$NREP max_levels=$NLEV itmax=$ITMAX" -echo " reserved_nodes=$SLURM_NNODES rank_points=[$RANK_POINTS]" -echo " exe=$EXE" -echo "============================================================" - -if [ ! -x "$EXE" ]; then - echo "FATAL: $EXE not found or not executable. Build it before submitting:" - echo " cd samples/advanced/pdegen && make amg_d_comm_test" - exit 1 -fi - -FAILED=0 - -for NRANKS in $RANK_POINTS; do - NNODES=$(( (NRANKS + 79) / 80 )) - STEP_DIR=$RESDIR/${NRANKS}ranks - mkdir -p $STEP_DIR - - echo "" - echo ">>> STRONG point: $NRANKS ranks ($NNODES nodes), fixed dim=$DIM" - - srun -N $NNODES -n $NRANKS --ntasks-per-node=80 --cpus-per-task=1 \ - $EXE $DIM $NREP $NLEV $ITMAX \ - --mode=all --csv=$STEP_DIR/amg_comm.csv \ - > $STEP_DIR/run.out 2>&1 - RC=$? - echo ">>> exit=$RC output=$STEP_DIR/run.out" - - # Fail loudly rather than silently producing a dataset nobody can trust: - # if a scheme did not reach every level, the comparison is meaningless. - if ! grep -q "SCHEME PROPAGATION: OK" $STEP_DIR/run.out; then - echo "!!! WARNING: scheme propagation not confirmed at $NRANKS ranks" - FAILED=1 - fi - # All configurations must converge identically; a differing iteration count - # means the halo exchange changed the arithmetic, not just its schedule. - NITERS=$(grep -oE "it +[0-9]+" $STEP_DIR/run.out | awk '{print $2}' | sort -u | wc -l) - if [ "$NITERS" != "1" ]; then - echo "!!! WARNING: iteration count is not constant at $NRANKS ranks ($NITERS distinct values)" - FAILED=1 - fi -done - -echo "" -if [ "$FAILED" = "0" ]; then - echo "=== AMG COMM STRONG DONE, all checks passed. Results: $RESDIR ===" -else - echo "=== AMG COMM STRONG DONE WITH WARNINGS. Results: $RESDIR ===" -fi -find $RESDIR -name "*.csv" -printf " %p (%s bytes)\n" diff --git a/samples/advanced/pdegen/collect_results.sh b/samples/advanced/pdegen/collect_results.sh new file mode 100755 index 00000000..1deee8a3 --- /dev/null +++ b/samples/advanced/pdegen/collect_results.sh @@ -0,0 +1,169 @@ +#!/bin/bash +# ============================================================================ +# collect_results.sh — snapshot a result set together with its provenance +# +# A directory of CSVs is not a dataset: in a month nobody remembers which code, +# which library and which environment produced it, and the run gets repeated. +# This records all of that next to the numbers, so a result set can be trusted +# without re-running it. +# +# Score-P experiment directories are folded into the result set too. Their +# default names carry a timestamp and nothing else, so on their own nobody can +# tell which scale point a profile belongs to; here they are moved inside the +# result directory and listed with their times, in run order. +# +# USE (on the cluster, from samples/advanced/pdegen): +# ./collect_results.sh results_amg_comm_44449535 [label] [scorep-dir ...] +# +# With no scorep-dir given, every scorep-* in the current directory is taken. +# +# Produces /PROVENANCE.md and a tarball ready to transfer. +# ============================================================================ +set -u + +RESDIR=${1:?usage: collect_results.sh [label] [scorep-dir ...]} +LABEL=${2:-} +shift 2 2>/dev/null || shift $# +SCOREP_DIRS=("$@") +if [ ${#SCOREP_DIRS[@]} -eq 0 ]; then + shopt -s nullglob + SCOREP_DIRS=(scorep-*) + shopt -u nullglob +fi + +if [ ! -d "$RESDIR" ]; then + echo "No such directory: $RESDIR" >&2 + exit 1 +fi + +JOBID=$(basename "$RESDIR" | sed 's/.*_//') +OUT=$RESDIR/PROVENANCE.md + +# Fold the Score-P experiments into the result set, oldest first: the sbatch +# runs the scale points in order, so run order is the only handle available to +# match a timestamped profile to its rank count. +PROFDIR=$RESDIR/scorep +if [ ${#SCOREP_DIRS[@]} -gt 0 ]; then + mkdir -p "$PROFDIR" + echo "Folding ${#SCOREP_DIRS[@]} Score-P experiment(s) into $PROFDIR:" + while IFS= read -r d; do + [ -d "$d" ] || continue + printf ' %s %s\n' "$(date -r "$d" '+%Y-%m-%d %H:%M:%S')" "$d" + mv "$d" "$PROFDIR/" + done < <(ls -1dtr "${SCOREP_DIRS[@]}" 2>/dev/null) +fi + +{ + echo "# Result set $JOBID" + [ -n "$LABEL" ] && echo -e "\n**$LABEL**" + echo + echo "Collected: $(date -Iseconds) on $(hostname)" + echo + + echo "## Instrumentation" + echo + # The single most misread property of a result set: whether the binary was + # instrumented and whether profiling was actually on during the run. + if [ -d "$PROFDIR" ]; then + echo "Score-P experiments were collected, so profiling was ACTIVE:" + echo "absolute timings in the CSVs include instrumentation overhead and are" + echo "comparable between schemes, not against a clean build." + echo + echo "Profiles in run order (the sbatch runs the scale points in order," + echo "so the n-th profile corresponds to the n-th entry of RANK_POINTS;" + echo "check against the timestamps in the per-step run.out to be sure):" + echo + echo '```' + ls -1dtr "$PROFDIR"/*/ 2>/dev/null | while read -r d; do + printf '%s %8s %s\n' "$(date -r "$d" '+%Y-%m-%d %H:%M:%S')" \ + "$(du -sh "$d" | cut -f1)" "$(basename "$d")" + done + echo '```' + echo + echo "Read them with:" + echo '```' + echo "cube_stat -p /profile.cubex" + echo "cube_dump -m time -c all /profile.cubex" + echo '```' + else + echo "No Score-P experiments collected; profiling was probably off." + fi + echo + echo '```' + echo "SCOREP_ENABLE_PROFILING=${SCOREP_ENABLE_PROFILING:-}" + echo "SCOREP_ENABLE_TRACING=${SCOREP_ENABLE_TRACING:-}" + echo '```' + echo + echo "## MPI runtime settings" + echo + echo "These change which implementation the schemes actually run on, so they" + echo "have to be held constant across a campaign. HCOLL in particular" + echo "accelerates collectives, which the two neighborhood schemes use and the" + echo "point-to-point and one-sided ones do not: switching it changes what is" + echo "being compared, not just how fast it is." + echo + echo '```' + env | grep -E "^(OMPI_MCA_|UCX_|SLURM_CPU_BIND)" | sort || echo "(none set)" + echo '```' + echo + + echo "## Code" + echo + echo '```' + for repo in "$HOME/Desktop/scorep/amg4psblas" "$HOME/Desktop/scorep/psblas3"; do + if [ -d "$repo/.git" ]; then + printf '%-12s %s %s\n' "$(basename "$repo")" \ + "$(git -C "$repo" rev-parse --short HEAD 2>/dev/null)" \ + "$(git -C "$repo" rev-parse --abbrev-ref HEAD 2>/dev/null)" + if [ -n "$(git -C "$repo" status --porcelain 2>/dev/null)" ]; then + echo " UNCOMMITTED CHANGES PRESENT" + fi + else + printf '%-12s not a git tree\n' "$(basename "$repo")" + fi + done + echo '```' + echo + + echo "## Build configuration" + echo + echo '```' + grep -E "^PSBLASDIR|^FC=|^CC=|^FCOPT" "$HOME/Desktop/scorep/amg4psblas/Make.inc" 2>/dev/null + grep -E "^FCUDEFINES|^CUDA_DIR" \ + "$HOME/opt/psblas-cuda/include/Make.inc.psblas" 2>/dev/null + echo '```' + echo + + echo "## Job" + echo + echo '```' + sacct -j "$JOBID" --format=JobID,JobName,Partition,QOS,NNodes,NTasks,Elapsed,State 2>/dev/null \ + || echo "sacct unavailable" + echo '```' + echo + + echo "## Parameters" + echo + echo '```' + grep -E "^(DIM|NREP|NLEV|ITMAX|RANK_POINTS)=" amg_comm_strong.sh 2>/dev/null + echo '```' + echo + + echo "## Contents" + echo + echo '```' + find "$RESDIR" -type f -printf "%10s %p\n" | sort -k2 + echo '```' +} > "$OUT" + +# Keep the exact script that produced the numbers: parameters drift. +cp -p amg_comm_strong.sh "$RESDIR/amg_comm_strong.sh.used" 2>/dev/null || true + +TARBALL=${RESDIR%/}.tar.gz +tar czf "$TARBALL" "$RESDIR" + +echo "Provenance written to $OUT" +echo "Tarball: $TARBALL ($(du -h "$TARBALL" | cut -f1))" +echo +echo "Transfer with:" +echo " rsync -avz @:$(readlink -f "$TARBALL") ."