System Diagnostic for Podcast Transcription
Tassos |
PRO |
06/01/26 12:51:43 PM UTC (Edited) |
0 ⭐ |
105 👁️ |
Never ⏰ |
[Script, BASH, System, diagnostic]
#!/usr/bin/env bash
# =============================================================================
# Complete System Diagnostic for Podcast Transcription
#
# ⚠️ Run with sudo or root user !
#
# Usage:
# chmod +x 12-system-evaluation.sh
# sudo ./12-system-evaluation.sh
#
# What it does:
# Examines the machine's hardware, software, and network connectivity
# and recommends the best transcription approach for this system.
#
# Compatible with: Fedora, RHEL, Debian, Ubuntu (any Linux with bash)
#
# See also:
# docs/en/12-system-evaluation.md
# docs/en/13-transcription-runbook.md
#
# Tip commands:
# dmidecode -t system | grep -E "Manufacturer|Product Name|Version|Family"
# lspci | grep -i "3d\|vga\|nvidia\|quadro"
# =============================================================================
# Fix Windows-style line endings (CRLF → LF) if present
if grep -qP '\r$' "$0" 2>/dev/null; then
sed -i 's/\r$//' "$0"
exec bash "$0" "$@"
fi
set -euo pipefail
# --- Colors ---
RED='\033[0;31m'
GREEN='\033[0;32m'
YELLOW='\033[1;33m'
CYAN='\033[0;36m'
BOLD='\033[1m'
NC='\033[0m'
header() {
echo ""
echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}"
echo -e "${BOLD} $1${NC}"
echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}"
}
section() {
echo -e "\n${YELLOW}--- $1 ---${NC}"
}
ok() { echo -e " ${GREEN}✓${NC} $1"; }
warn() { echo -e " ${YELLOW}!${NC} $1"; }
fail() { echo -e " ${RED}✗${NC} $1"; }
# =============================================================================
header "MACHINE IDENTITY"
# =============================================================================
section "Manufacturer & Model"
if [ -r /sys/devices/virtual/dmi/id/sys_vendor ]; then
VENDOR=$(cat /sys/devices/virtual/dmi/id/sys_vendor 2>/dev/null || echo "Unknown")
PRODUCT=$(cat /sys/devices/virtual/dmi/id/product_name 2>/dev/null || echo "Unknown")
VERSION=$(cat /sys/devices/virtual/dmi/id/product_version 2>/dev/null || echo "")
echo " Vendor: $VENDOR"
echo " Product: $PRODUCT"
[ -n "$VERSION" ] && echo " Version: $VERSION"
else
warn "Cannot read DMI (need sudo?)"
fi
if command -v dmidecode &>/dev/null && [ "$(id -u)" = "0" ]; then
section "Detailed System Info (dmidecode)"
dmidecode -t system 2>/dev/null | grep -E "Manufacturer|Product Name|Version|Serial|Family" | sed 's/^/ /'
echo ""
dmidecode -t baseboard 2>/dev/null | grep -E "Manufacturer|Product Name|Version" | sed 's/^/ /'
echo ""
dmidecode -t bios 2>/dev/null | grep -E "Vendor|Version|Release Date" | sed 's/^/ /'
elif [ "$(id -u)" != "0" ]; then
warn "Run with sudo for full dmidecode information"
fi
# =============================================================================
header "CPU (PROCESSOR)"
# =============================================================================
section "Model & Cores"
lscpu | grep -E "Model name|^CPU\(s\)|Thread|Core|Socket|^CPU max MHz|^CPU min MHz|Architecture" | sed 's/^/ /'
section "AVX Support"
if grep -qo 'avx512' /proc/cpuinfo 2>/dev/null; then
ok "AVX-512: Supported (excellent for inference)"
elif grep -qo 'avx2' /proc/cpuinfo 2>/dev/null; then
ok "AVX2: Supported (good for inference)"
else
fail "No AVX2 — CPU inference will be very slow"
fi
# =============================================================================
header "MEMORY (RAM)"
# =============================================================================
section "Total & Available"
free -h | head -2 | sed 's/^/ /'
AVAILABLE_MB=$(free -m | awk '/Mem:/ {print $7}')
echo ""
if [ "$AVAILABLE_MB" -ge 14000 ]; then
ok "Available RAM: ${AVAILABLE_MB} MB — enough for WhisperX large-v3 + diarization on CPU"
elif [ "$AVAILABLE_MB" -ge 11000 ]; then
ok "Available RAM: ${AVAILABLE_MB} MB — enough for Whisper large-v3 on CPU"
elif [ "$AVAILABLE_MB" -ge 6000 ]; then
warn "Available RAM: ${AVAILABLE_MB} MB — enough only for Whisper medium on CPU"
else
fail "Available RAM: ${AVAILABLE_MB} MB — insufficient for local inference, use cloud"
fi
if command -v dmidecode &>/dev/null && [ "$(id -u)" = "0" ]; then
section "RAM Details (slots, type, speed)"
dmidecode -t memory 2>/dev/null | grep -E "Size|Type:|Speed|Manufacturer|Locator" | grep -v "No Module\|Unknown\|Not Specified" | sed 's/^/ /'
fi
# =============================================================================
header "GRAPHICS CARD (GPU)"
# =============================================================================
section "GPU Detection"
GPU_FOUND=false
if lspci 2>/dev/null | grep -iE "nvidia" | grep -iv "audio" > /dev/null 2>&1; then
lspci | grep -iE "nvidia" | grep -iv "audio" | sed 's/^/ /'
GPU_FOUND=true
echo ""
if command -v nvidia-smi &>/dev/null; then
section "NVIDIA GPU Details"
nvidia-smi --query-gpu=name,memory.total,memory.free,memory.used,driver_version,compute_cap,temperature.gpu,power.draw --format=csv,noheader 2>/dev/null | while IFS=',' read -r name total free used driver compute temp power; do
echo " Name: $name"
echo " VRAM Total: $total"
echo " VRAM Free: $free"
echo " VRAM Used: $used"
echo " Driver: $driver"
echo " Compute Cap: $compute"
echo " Temperature: $temp"
echo " Power Draw: $power"
done
VRAM_MB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d ' ')
echo ""
if [ -n "$VRAM_MB" ] && [ "$VRAM_MB" -ge 10000 ] 2>/dev/null; then
ok "VRAM: ${VRAM_MB} MB — enough for Whisper large-v3 on GPU (best accuracy)"
elif [ -n "$VRAM_MB" ] && [ "$VRAM_MB" -ge 5000 ] 2>/dev/null; then
warn "VRAM: ${VRAM_MB} MB — enough for Whisper medium on GPU"
elif [ -n "$VRAM_MB" ] && [ "$VRAM_MB" -ge 2000 ] 2>/dev/null; then
warn "VRAM: ${VRAM_MB} MB — enough only for Whisper small on GPU"
fi
else
fail "nvidia-smi not found — install NVIDIA drivers"
fi
else
# Check for AMD or Intel integrated
if lspci 2>/dev/null | grep -iE "amd.*radeon|amd.*vga" > /dev/null 2>&1; then
lspci | grep -iE "amd.*radeon|amd.*vga" | sed 's/^/ /'
warn "AMD GPU — not supported for Whisper/WhisperX (requires NVIDIA CUDA)"
fi
if lspci 2>/dev/null | grep -iE "intel.*(vga|graphics)" > /dev/null 2>&1; then
lspci | grep -iE "intel.*(vga|graphics)" | sed 's/^/ /'
warn "Intel integrated graphics — not supported for AI inference"
fi
if ! $GPU_FOUND; then
fail "No NVIDIA GPU found — use cloud (Groq/Google) or CPU mode (slow)"
fi
fi
# =============================================================================
header "CUDA"
# =============================================================================
section "CUDA Toolkit"
if command -v nvcc &>/dev/null; then
nvcc --version 2>/dev/null | grep "release" | sed 's/^/ /'
ok "CUDA toolkit installed"
else
warn "CUDA toolkit not found (nvcc) — needed for WhisperX on GPU"
fi
section "CUDA via nvidia-smi"
if command -v nvidia-smi &>/dev/null; then
CUDA_VER=$(nvidia-smi 2>/dev/null | grep "CUDA Version" | awk '{print $9}')
if [ -n "$CUDA_VER" ]; then
ok "CUDA Version: $CUDA_VER (via driver)"
fi
fi
# =============================================================================
header "STORAGE"
# =============================================================================
section "Disk Space"
df -h / | tail -1 | awk '{print " Total: " $2 " Used: " $3 " Free: " $4 " Use%: " $5}'
section "Disk Model & Type"
lsblk -d -o NAME,MODEL,ROTA,SIZE,TRAN 2>/dev/null | sed 's/^/ /'
echo ""
echo " (ROTA=0: SSD, ROTA=1: HDD, TRAN: nvme/sata/usb)"
if ls /dev/nvme* &>/dev/null 2>&1; then
ok "NVMe drive detected — fast model loading"
fi
# =============================================================================
header "PYTHON & AI PACKAGES"
# =============================================================================
section "Python"
if command -v python3 &>/dev/null; then
ok "Python3: $(python3 --version 2>&1 | awk '{print $2}')"
else
fail "Python3 not found — required for all approaches"
fi
if command -v pip3 &>/dev/null; then
ok "pip3: $(pip3 --version 2>&1 | awk '{print $2}')"
else
warn "pip3 not found"
fi
section "Installed AI Packages"
if command -v pip3 &>/dev/null; then
PACKAGES=$(pip3 list 2>/dev/null | grep -iE "^(openai-whisper|whisperx|faster-whisper|torch |torchaudio|google-cloud-speech|groq |pyannote|transformers) " || true)
if [ -n "$PACKAGES" ]; then
echo "$PACKAGES" | while read -r pkg ver; do
ok "$pkg ($ver)"
done
else
warn "No relevant AI packages found — installation will be needed"
fi
fi
# =============================================================================
header "FFMPEG"
# =============================================================================
if command -v ffmpeg &>/dev/null; then
ok "FFmpeg: $(ffmpeg -version 2>/dev/null | head -1 | awk '{print $3}')"
else
fail "FFmpeg not found — required for audio conversion and chunking"
echo " Install: sudo apt install ffmpeg (Debian/Ubuntu)"
echo " sudo dnf install ffmpeg-free (Fedora/RHEL)"
fi
# =============================================================================
header "NETWORK"
# =============================================================================
section "Cloud API Connectivity"
if command -v curl &>/dev/null; then
GROQ_STATUS=$(curl -s -o /dev/null -w "%{http_code}" --max-time 5 https://api.groq.com/openai/v1/models 2>/dev/null || echo "000")
if [ "$GROQ_STATUS" != "000" ]; then
ok "Groq API: HTTP $GROQ_STATUS — reachable"
else
fail "Groq API: unreachable — Approach A will not work"
fi
GOOGLE_STATUS=$(curl -s -o /dev/null -w "%{http_code}" --max-time 5 https://speech.googleapis.com 2>/dev/null || echo "000")
if [ "$GOOGLE_STATUS" != "000" ]; then
ok "Google Cloud STT API: HTTP $GOOGLE_STATUS — reachable"
else
fail "Google Cloud STT API: unreachable — Approach C will not work"
fi
else
warn "curl not found — cannot check connectivity"
fi
# =============================================================================
header "OPERATING SYSTEM"
# =============================================================================
echo " Kernel: $(uname -r)"
echo " Arch: $(uname -m)"
if [ -f /etc/os-release ]; then
echo " Distro: $(grep PRETTY_NAME /etc/os-release | cut -d= -f2 | tr -d '"')"
fi
echo " Uptime: $(uptime -p 2>/dev/null || uptime | awk '{print $3, $4}')"
section "Power"
if command -v upower &>/dev/null; then
AC=$(upower -i /org/freedesktop/UPower/devices/line_power_AC 2>/dev/null | grep "online" | awk '{print $2}')
if [ "$AC" = "yes" ]; then
ok "On AC power — good for inference"
elif [ "$AC" = "no" ]; then
warn "On battery — PLUG IN before running inference (2x slower on battery)"
fi
fi
section "CPU Temperature"
if command -v sensors &>/dev/null; then
sensors 2>/dev/null | grep -iE "core|temp|package" | head -5 | sed 's/^/ /'
elif [ -d /sys/class/thermal ]; then
for tz in /sys/class/thermal/thermal_zone*/temp; do
TEMP=$(cat "$tz" 2>/dev/null)
if [ -n "$TEMP" ]; then
TEMP_C=$((TEMP / 1000))
ZONE=$(basename "$(dirname "$tz")")
if [ "$TEMP_C" -gt 80 ]; then
warn "$ZONE: ${TEMP_C}°C — HIGH temperature, throttling risk"
else
ok "$ZONE: ${TEMP_C}°C"
fi
fi
done
else
warn "Cannot read temperature — install lm-sensors"
fi
# =============================================================================
header "APPROACH RECOMMENDATION"
# =============================================================================
echo ""
echo -e "${BOLD}Based on this machine's hardware:${NC}"
echo ""
# Decision logic
HAS_NVIDIA_GPU=false
VRAM_SUFFICIENT_LARGE=false
VRAM_SUFFICIENT_MEDIUM=false
if command -v nvidia-smi &>/dev/null; then
VRAM_MB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d ' ')
if [ -n "$VRAM_MB" ] 2>/dev/null; then
HAS_NVIDIA_GPU=true
[ "$VRAM_MB" -ge 10000 ] 2>/dev/null && VRAM_SUFFICIENT_LARGE=true
[ "$VRAM_MB" -ge 5000 ] 2>/dev/null && VRAM_SUFFICIENT_MEDIUM=true
fi
fi
if $VRAM_SUFFICIENT_LARGE; then
echo -e " ${GREEN}★ RECOMMENDATION: Approach B — WhisperX (large-v3 on GPU)${NC}"
echo " Best accuracy (~3-4% WER), speaker diarization, free"
echo " Estimated time: ~5 hours for 4 episodes"
elif $VRAM_SUFFICIENT_MEDIUM; then
echo -e " ${GREEN}★ RECOMMENDATION: Approach B — WhisperX (medium on GPU)${NC}"
echo " Good accuracy (~4% WER), speaker diarization, free"
echo " Estimated time: ~5 hours for 4 episodes"
elif [ "$AVAILABLE_MB" -ge 14000 ] 2>/dev/null; then
echo -e " ${YELLOW}★ RECOMMENDATION: Approach A — Groq API (free, fast)${NC}"
echo " Alternative: WhisperX on CPU (large-v3) but ~50 hours!"
echo " Estimated time with Groq: ~5 minutes"
else
echo -e " ${YELLOW}★ RECOMMENDATION: Approach A — Groq API (free, fast)${NC}"
echo " Insufficient hardware for local inference"
echo " Estimated time: ~5 minutes"
fi
echo ""
echo " For proper noun accuracy (Dallaire, Habyarimana, Interahamwe):"
echo -e " ${CYAN}→ Add Approach C (Google Cloud STT, ~\$5) with custom vocabulary${NC}"
echo ""
echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}"
echo -e "${BOLD} See docs/en/13-transcription-runbook.md for full instructions${NC}"
echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}"
echo ""
Comments