#!/usr/bin/env bash # SUM Parts - did the run come back to healthy speed? # # Healthy for pointnet at voxel_max 64000 on this box is ~0.29 s/it (benchmark) # and ~103 s/epoch. The degraded state was ~3.1 s/it and ~1100 s/epoch, caused # by the caching allocator pool overflowing VRAM into host RAM. set -uo pipefail RUNROOT="$HOME/sum-parts/runs" D="${1:-$(cat "$RUNROOT/.latest" 2>/dev/null)}" LOG="$D/train.log" echo "run: $D" [ -f "$LOG" ] || { echo " no train.log yet"; exit 1; } echo echo "=== resume confirmation ===" grep -aE 'Resume|Successful Loading|start_epoch' "$LOG" | tail -3 || echo " (none found)" echo echo "=== current epoch ===" grep -aoE 'Train Epoch \[[0-9]+/[0-9]+\]' "$LOG" | tail -1 || echo " not started" echo echo "=== recent iteration rates ===" grep -aoE '[0-9.]+(s/it|it/s)' "$LOG" | tail -12 | tr '\n' ' ' echo echo echo "=== GPU ===" nvidia-smi --query-gpu=memory.used,memory.total,utilization.gpu,power.draw --format=csv,noheader echo echo "=== verdict ===" # tqdm prints either "N.NNs/it" or "N.NNit/s" depending on which side of 1 the # rate falls, so take whichever form appeared LAST and normalise to s/it. # (Grepping only for 's/it' picks up a stale warm-up reading and misjudges a # run that has since recovered.) last=$(grep -aoE '[0-9.]+(s/it|it/s)' "$LOG" | tail -1) if [ -z "$last" ]; then echo " no rate readings yet" exit 0 fi sec=$(awk -v r="$last" 'BEGIN { if (r ~ /it\/s$/) { sub(/it\/s$/, "", r); printf "%.4f", (r > 0 ? 1.0 / r : 0) } else { sub(/s\/it$/, "", r); printf "%.4f", r } }') echo " last reading: $last -> ${sec} s/it" awk -v v="$sec" 'BEGIN { if (v < 0.8) print " HEALTHY (benchmark is 0.291 s/it)"; else if (v < 1.5) print " MARGINAL"; else print " DEGRADED -- allocator likely spilling to host RAM"; }'