Tassos icon

System Diagnostic for Podcast Transcription

Tassos | PRO | 06/01/26 12:51:43 PM UTC (Edited) | 0 ⭐ | 105 👁️ | Never ⏰ | [Script, BASH, System, diagnostic]
Bash |

14.64 KB

|

None

|

0 👍

/

0 👎

#!/usr/bin/env bash
# =============================================================================
# Complete System Diagnostic for Podcast Transcription
#
# ⚠️ Run with sudo or root user !
#
# Usage:
#   chmod +x 12-system-evaluation.sh
#   sudo ./12-system-evaluation.sh
#
# What it does:
#   Examines the machine's hardware, software, and network connectivity
#   and recommends the best transcription approach for this system.
#
# Compatible with: Fedora, RHEL, Debian, Ubuntu (any Linux with bash)
#
# See also:
#   docs/en/12-system-evaluation.md
#   docs/en/13-transcription-runbook.md
# 
# Tip commands:
# dmidecode -t system | grep -E "Manufacturer|Product Name|Version|Family"
# lspci | grep -i "3d\|vga\|nvidia\|quadro"
# =============================================================================
 
# Fix Windows-style line endings (CRLF → LF) if present
if grep -qP '\r$' "$0" 2>/dev/null; then
    sed -i 's/\r$//' "$0"
    exec bash "$0" "$@"
fi
 
set -euo pipefail
 
# --- Colors ---
RED='\033[0;31m'
GREEN='\033[0;32m'
YELLOW='\033[1;33m'
CYAN='\033[0;36m'
BOLD='\033[1m'
NC='\033[0m'
 
header() {
    echo ""
    echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}"
    echo -e "${BOLD}  $1${NC}"
    echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}"
}
 
section() {
    echo -e "\n${YELLOW}--- $1 ---${NC}"
}
 
ok()   { echo -e "  ${GREEN}✓${NC} $1"; }
warn() { echo -e "  ${YELLOW}!${NC} $1"; }
fail() { echo -e "  ${RED}✗${NC} $1"; }
 
# =============================================================================
header "MACHINE IDENTITY"
# =============================================================================
 
section "Manufacturer & Model"
if [ -r /sys/devices/virtual/dmi/id/sys_vendor ]; then
    VENDOR=$(cat /sys/devices/virtual/dmi/id/sys_vendor 2>/dev/null || echo "Unknown")
    PRODUCT=$(cat /sys/devices/virtual/dmi/id/product_name 2>/dev/null || echo "Unknown")
    VERSION=$(cat /sys/devices/virtual/dmi/id/product_version 2>/dev/null || echo "")
    echo "  Vendor:  $VENDOR"
    echo "  Product: $PRODUCT"
    [ -n "$VERSION" ] && echo "  Version: $VERSION"
else
    warn "Cannot read DMI (need sudo?)"
fi
 
if command -v dmidecode &>/dev/null && [ "$(id -u)" = "0" ]; then
    section "Detailed System Info (dmidecode)"
    dmidecode -t system 2>/dev/null | grep -E "Manufacturer|Product Name|Version|Serial|Family" | sed 's/^/  /'
    echo ""
    dmidecode -t baseboard 2>/dev/null | grep -E "Manufacturer|Product Name|Version" | sed 's/^/  /'
    echo ""
    dmidecode -t bios 2>/dev/null | grep -E "Vendor|Version|Release Date" | sed 's/^/  /'
elif [ "$(id -u)" != "0" ]; then
    warn "Run with sudo for full dmidecode information"
fi
 
# =============================================================================
header "CPU (PROCESSOR)"
# =============================================================================
 
section "Model & Cores"
lscpu | grep -E "Model name|^CPU\(s\)|Thread|Core|Socket|^CPU max MHz|^CPU min MHz|Architecture" | sed 's/^/  /'
 
section "AVX Support"
if grep -qo 'avx512' /proc/cpuinfo 2>/dev/null; then
    ok "AVX-512: Supported (excellent for inference)"
elif grep -qo 'avx2' /proc/cpuinfo 2>/dev/null; then
    ok "AVX2: Supported (good for inference)"
else
    fail "No AVX2 — CPU inference will be very slow"
fi
 
# =============================================================================
header "MEMORY (RAM)"
# =============================================================================
 
section "Total & Available"
free -h | head -2 | sed 's/^/  /'
 
AVAILABLE_MB=$(free -m | awk '/Mem:/ {print $7}')
echo ""
if [ "$AVAILABLE_MB" -ge 14000 ]; then
    ok "Available RAM: ${AVAILABLE_MB} MB — enough for WhisperX large-v3 + diarization on CPU"
elif [ "$AVAILABLE_MB" -ge 11000 ]; then
    ok "Available RAM: ${AVAILABLE_MB} MB — enough for Whisper large-v3 on CPU"
elif [ "$AVAILABLE_MB" -ge 6000 ]; then
    warn "Available RAM: ${AVAILABLE_MB} MB — enough only for Whisper medium on CPU"
else
    fail "Available RAM: ${AVAILABLE_MB} MB — insufficient for local inference, use cloud"
fi
 
if command -v dmidecode &>/dev/null && [ "$(id -u)" = "0" ]; then
    section "RAM Details (slots, type, speed)"
    dmidecode -t memory 2>/dev/null | grep -E "Size|Type:|Speed|Manufacturer|Locator" | grep -v "No Module\|Unknown\|Not Specified" | sed 's/^/  /'
fi
 
# =============================================================================
header "GRAPHICS CARD (GPU)"
# =============================================================================
 
section "GPU Detection"
GPU_FOUND=false
 
if lspci 2>/dev/null | grep -iE "nvidia" | grep -iv "audio" > /dev/null 2>&1; then
    lspci | grep -iE "nvidia" | grep -iv "audio" | sed 's/^/  /'
    GPU_FOUND=true
    echo ""
 
    if command -v nvidia-smi &>/dev/null; then
        section "NVIDIA GPU Details"
        nvidia-smi --query-gpu=name,memory.total,memory.free,memory.used,driver_version,compute_cap,temperature.gpu,power.draw --format=csv,noheader 2>/dev/null | while IFS=',' read -r name total free used driver compute temp power; do
            echo "  Name:        $name"
            echo "  VRAM Total:  $total"
            echo "  VRAM Free:   $free"
            echo "  VRAM Used:   $used"
            echo "  Driver:      $driver"
            echo "  Compute Cap: $compute"
            echo "  Temperature: $temp"
            echo "  Power Draw:  $power"
        done
 
        VRAM_MB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d ' ')
        echo ""
        if [ -n "$VRAM_MB" ] && [ "$VRAM_MB" -ge 10000 ] 2>/dev/null; then
            ok "VRAM: ${VRAM_MB} MB — enough for Whisper large-v3 on GPU (best accuracy)"
        elif [ -n "$VRAM_MB" ] && [ "$VRAM_MB" -ge 5000 ] 2>/dev/null; then
            warn "VRAM: ${VRAM_MB} MB — enough for Whisper medium on GPU"
        elif [ -n "$VRAM_MB" ] && [ "$VRAM_MB" -ge 2000 ] 2>/dev/null; then
            warn "VRAM: ${VRAM_MB} MB — enough only for Whisper small on GPU"
        fi
    else
        fail "nvidia-smi not found — install NVIDIA drivers"
    fi
else
    # Check for AMD or Intel integrated
    if lspci 2>/dev/null | grep -iE "amd.*radeon|amd.*vga" > /dev/null 2>&1; then
        lspci | grep -iE "amd.*radeon|amd.*vga" | sed 's/^/  /'
        warn "AMD GPU — not supported for Whisper/WhisperX (requires NVIDIA CUDA)"
    fi
    if lspci 2>/dev/null | grep -iE "intel.*(vga|graphics)" > /dev/null 2>&1; then
        lspci | grep -iE "intel.*(vga|graphics)" | sed 's/^/  /'
        warn "Intel integrated graphics — not supported for AI inference"
    fi
    if ! $GPU_FOUND; then
        fail "No NVIDIA GPU found — use cloud (Groq/Google) or CPU mode (slow)"
    fi
fi
 
# =============================================================================
header "CUDA"
# =============================================================================
 
section "CUDA Toolkit"
if command -v nvcc &>/dev/null; then
    nvcc --version 2>/dev/null | grep "release" | sed 's/^/  /'
    ok "CUDA toolkit installed"
else
    warn "CUDA toolkit not found (nvcc) — needed for WhisperX on GPU"
fi
 
section "CUDA via nvidia-smi"
if command -v nvidia-smi &>/dev/null; then
    CUDA_VER=$(nvidia-smi 2>/dev/null | grep "CUDA Version" | awk '{print $9}')
    if [ -n "$CUDA_VER" ]; then
        ok "CUDA Version: $CUDA_VER (via driver)"
    fi
fi
 
# =============================================================================
header "STORAGE"
# =============================================================================
 
section "Disk Space"
df -h / | tail -1 | awk '{print "  Total: " $2 "  Used: " $3 "  Free: " $4 "  Use%: " $5}'
 
section "Disk Model & Type"
lsblk -d -o NAME,MODEL,ROTA,SIZE,TRAN 2>/dev/null | sed 's/^/  /'
echo ""
echo "  (ROTA=0: SSD, ROTA=1: HDD, TRAN: nvme/sata/usb)"
 
if ls /dev/nvme* &>/dev/null 2>&1; then
    ok "NVMe drive detected — fast model loading"
fi
 
# =============================================================================
header "PYTHON & AI PACKAGES"
# =============================================================================
 
section "Python"
if command -v python3 &>/dev/null; then
    ok "Python3: $(python3 --version 2>&1 | awk '{print $2}')"
else
    fail "Python3 not found — required for all approaches"
fi
 
if command -v pip3 &>/dev/null; then
    ok "pip3: $(pip3 --version 2>&1 | awk '{print $2}')"
else
    warn "pip3 not found"
fi
 
section "Installed AI Packages"
if command -v pip3 &>/dev/null; then
    PACKAGES=$(pip3 list 2>/dev/null | grep -iE "^(openai-whisper|whisperx|faster-whisper|torch |torchaudio|google-cloud-speech|groq |pyannote|transformers) " || true)
    if [ -n "$PACKAGES" ]; then
        echo "$PACKAGES" | while read -r pkg ver; do
            ok "$pkg ($ver)"
        done
    else
        warn "No relevant AI packages found — installation will be needed"
    fi
fi
 
# =============================================================================
header "FFMPEG"
# =============================================================================
 
if command -v ffmpeg &>/dev/null; then
    ok "FFmpeg: $(ffmpeg -version 2>/dev/null | head -1 | awk '{print $3}')"
else
    fail "FFmpeg not found — required for audio conversion and chunking"
    echo "  Install: sudo apt install ffmpeg (Debian/Ubuntu)"
    echo "           sudo dnf install ffmpeg-free (Fedora/RHEL)"
fi
 
# =============================================================================
header "NETWORK"
# =============================================================================
 
section "Cloud API Connectivity"
if command -v curl &>/dev/null; then
    GROQ_STATUS=$(curl -s -o /dev/null -w "%{http_code}" --max-time 5 https://api.groq.com/openai/v1/models 2>/dev/null || echo "000")
    if [ "$GROQ_STATUS" != "000" ]; then
        ok "Groq API: HTTP $GROQ_STATUS — reachable"
    else
        fail "Groq API: unreachable — Approach A will not work"
    fi
 
    GOOGLE_STATUS=$(curl -s -o /dev/null -w "%{http_code}" --max-time 5 https://speech.googleapis.com 2>/dev/null || echo "000")
    if [ "$GOOGLE_STATUS" != "000" ]; then
        ok "Google Cloud STT API: HTTP $GOOGLE_STATUS — reachable"
    else
        fail "Google Cloud STT API: unreachable — Approach C will not work"
    fi
else
    warn "curl not found — cannot check connectivity"
fi
 
# =============================================================================
header "OPERATING SYSTEM"
# =============================================================================
 
echo "  Kernel: $(uname -r)"
echo "  Arch:   $(uname -m)"
if [ -f /etc/os-release ]; then
    echo "  Distro: $(grep PRETTY_NAME /etc/os-release | cut -d= -f2 | tr -d '"')"
fi
echo "  Uptime: $(uptime -p 2>/dev/null || uptime | awk '{print $3, $4}')"
 
section "Power"
if command -v upower &>/dev/null; then
    AC=$(upower -i /org/freedesktop/UPower/devices/line_power_AC 2>/dev/null | grep "online" | awk '{print $2}')
    if [ "$AC" = "yes" ]; then
        ok "On AC power — good for inference"
    elif [ "$AC" = "no" ]; then
        warn "On battery — PLUG IN before running inference (2x slower on battery)"
    fi
fi
 
section "CPU Temperature"
if command -v sensors &>/dev/null; then
    sensors 2>/dev/null | grep -iE "core|temp|package" | head -5 | sed 's/^/  /'
elif [ -d /sys/class/thermal ]; then
    for tz in /sys/class/thermal/thermal_zone*/temp; do
        TEMP=$(cat "$tz" 2>/dev/null)
        if [ -n "$TEMP" ]; then
            TEMP_C=$((TEMP / 1000))
            ZONE=$(basename "$(dirname "$tz")")
            if [ "$TEMP_C" -gt 80 ]; then
                warn "$ZONE: ${TEMP_C}°C — HIGH temperature, throttling risk"
            else
                ok "$ZONE: ${TEMP_C}°C"
            fi
        fi
    done
else
    warn "Cannot read temperature — install lm-sensors"
fi
 
# =============================================================================
header "APPROACH RECOMMENDATION"
# =============================================================================
 
echo ""
echo -e "${BOLD}Based on this machine's hardware:${NC}"
echo ""
 
# Decision logic
HAS_NVIDIA_GPU=false
VRAM_SUFFICIENT_LARGE=false
VRAM_SUFFICIENT_MEDIUM=false
 
if command -v nvidia-smi &>/dev/null; then
    VRAM_MB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d ' ')
    if [ -n "$VRAM_MB" ] 2>/dev/null; then
        HAS_NVIDIA_GPU=true
        [ "$VRAM_MB" -ge 10000 ] 2>/dev/null && VRAM_SUFFICIENT_LARGE=true
        [ "$VRAM_MB" -ge 5000 ] 2>/dev/null && VRAM_SUFFICIENT_MEDIUM=true
    fi
fi
 
if $VRAM_SUFFICIENT_LARGE; then
    echo -e "  ${GREEN}★ RECOMMENDATION: Approach B — WhisperX (large-v3 on GPU)${NC}"
    echo "    Best accuracy (~3-4% WER), speaker diarization, free"
    echo "    Estimated time: ~5 hours for 4 episodes"
elif $VRAM_SUFFICIENT_MEDIUM; then
    echo -e "  ${GREEN}★ RECOMMENDATION: Approach B — WhisperX (medium on GPU)${NC}"
    echo "    Good accuracy (~4% WER), speaker diarization, free"
    echo "    Estimated time: ~5 hours for 4 episodes"
elif [ "$AVAILABLE_MB" -ge 14000 ] 2>/dev/null; then
    echo -e "  ${YELLOW}★ RECOMMENDATION: Approach A — Groq API (free, fast)${NC}"
    echo "    Alternative: WhisperX on CPU (large-v3) but ~50 hours!"
    echo "    Estimated time with Groq: ~5 minutes"
else
    echo -e "  ${YELLOW}★ RECOMMENDATION: Approach A — Groq API (free, fast)${NC}"
    echo "    Insufficient hardware for local inference"
    echo "    Estimated time: ~5 minutes"
fi
 
echo ""
echo "  For proper noun accuracy (Dallaire, Habyarimana, Interahamwe):"
echo -e "  ${CYAN}→ Add Approach C (Google Cloud STT, ~\$5) with custom vocabulary${NC}"
 
echo ""
echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}"
echo -e "${BOLD}  See docs/en/13-transcription-runbook.md for full instructions${NC}"
echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}"
echo ""

Comments