Tom_Neverwinter icon

LLAMA BENCHMARK [PY] 1/04/2026

Tom_Neverwinter | PRO | 01/04/26 05:03:22 AM UTC (Edited) | 0 ⭐ | 3325 👁️ | Never ⏰ | []
text |

17.87 KB

|

None

|

0 👍

/

0 👎

import streamlit as st
import pandas as pd
import subprocess
import os
import re
import datetime
import time
 # ==========================================
# CONFIGURATION
# ==========================================
BASE_DIR = os.getcwd()
MODELS_DIR = os.path.join(BASE_DIR, "models")
LOGS_DIR = os.path.join(BASE_DIR, "bench_logs")
BENCH_EXE = os.path.join(BASE_DIR, "llama-bench.exe")
CLI_EXE = os.path.join(BASE_DIR, "llama-cli.exe")
DB_FILE = os.path.join(BASE_DIR, "model_library_benchmark.csv")
HISTORY_FILE = os.path.join(BASE_DIR, "benchmark_history.csv")
KEYWORDS = ["NSFW", "Uncensored", "Amoral", "Dark", "Fallen", "Rivermind", "Snowpiercer", "Abyss", "Erebus", "Lewd"]
 if not os.path.exists(LOGS_DIR):
    os.makedirs(LOGS_DIR)
 # --- SCHEMA ---
COLS = [
    # Identity
    "Model_Name", "Size_GB", "Quant", "Role", "B_Tag", "Params_Actual", "Backend", "Llama_Version",
    # Metrics
    "Gen_TG128", "Gen_TG128_StDev", "Ingest_PP512", "Rating_Speed", 
    # Human Eval
    "HE_Overall", "HE_Math", "HE_Science", "HE_History", "HE_Music",
    "HE_English", "HE_Coding", "HE_Logic", "HE_Creative", "HE_RP_Quality", 
    "HE_Personality", "HE_Emotes", "HE_Formatting", "HE_Instructions", 
    "HE_Censorship", "HE_Memory", "HE_Summarize", "HE_Translation", 
    "HE_JSON", "HE_RAG", "HE_Vision", "HE_Fun_Factor"
]
 st.set_page_config(page_title="Llama.cpp Librarian", layout="wide", page_icon="🦙")
 # ==========================================
# FUNCTIONS
# ==========================================
def get_hardware_info():
    try:
        gpu_info = subprocess.check_output("nvidia-smi --query-gpu=name,memory.total,memory.free --format=csv,noheader,nounits", shell=True).decode().strip()
        gpu_name, total_vram, free_vram = gpu_info.split(',')
        return f"{gpu_name} | VRAM: {free_vram.strip()}MB Free / {total_vram.strip()}MB Total"
    except:
        return "GPU Detection Failed"
 def get_llama_version():
    try:
        # Try to get version from llama-cli
        if os.path.exists(CLI_EXE):
            cmd = [CLI_EXE, "--version"]
            result = subprocess.run(cmd, capture_output=True, text=True, encoding='utf-8')
            output = result.stdout + result.stderr
            # Look for "version: 7549 (c9ced4910)"
            match = re.search(r'version:\s*(\d+)\s*\(([^)]+)\)', output)
            if match:
                return f"b{match.group(1)}-{match.group(2)[:7]}"
            return "Unknown Version"
        return "CLI Not Found"
    except:
        return "Version Check Failed"
 def parse_model_properties(filename):
    name = filename
    try:
        size_gb = round(os.path.getsize(os.path.join(MODELS_DIR, filename)) / (1024**3))
    except:
        size_gb = 0
     role = "Specialized/NSFW" if any(k.lower() in name.lower() for k in KEYWORDS) else "General"
     quant = "Unknown"
    quants = ["Q8_0", "Q6_K", "Q5_K_M", "Q5_K_S", "Q5_0", "Q4_K_M", "Q4_K_S", "Q4_0", "IQ3", "Q2_K", "F16", "F32"]
    for q in quants:
        if q.lower() in name.lower():
            quant = q
            break
     b_tag = "Other"
    b_match = re.search(r'(\d+(?:\.\d+)?)B', name, re.IGNORECASE)
    if b_match:
        b_tag = b_match.group(0).upper()
     return size_gb, role, quant, b_tag
 def run_benchmark(model_name):
    model_path = os.path.join(MODELS_DIR, model_name)
    cmd = [BENCH_EXE, "-m", model_path, "-t", "8", "-ngl", "999"]
     try:
        result = subprocess.run(cmd, capture_output=True, text=True, encoding='utf-8')
        output = result.stdout
         # Log
        timestamp = datetime.datetime.now().strftime("%Y-%m-%d_%H-%M-%S")
        log_file = os.path.join(LOGS_DIR, f"{model_name}_{timestamp}.log")
        with open(log_file, "w", encoding="utf-8") as f:
            f.write(f"COMMAND: {' '.join(cmd)}\n")
            f.write("="*40 + "\n")
            f.write(output)
            f.write("\n" + "="*40 + "\n")
            f.write(result.stderr)
         # Parse
        pp_match = re.search(r'\|\s*pp512\s*\|\s*([0-9\.]+)\s*\+/-\s*([0-9\.]+)', output)
        tg_match = re.search(r'\|\s*tg128\s*\|\s*([0-9\.]+)\s*\+/-\s*([0-9\.]+)', output)
         if not pp_match: pp_match = re.search(r'\|\s*pp512\s*\|\s*([0-9\.]+)', output)
        if not tg_match: tg_match = re.search(r'\|\s*tg128\s*\|\s*([0-9\.]+)', output)
         pp_val = float(pp_match.group(1)) if pp_match else 0.0
        tg_val = float(tg_match.group(1)) if tg_match else 0.0
        tg_std = float(tg_match.group(2)) if tg_match and len(tg_match.groups()) > 1 else 0.0
         backend = "Unknown"
        if "CUDA" in output: backend = "CUDA"
        elif "ROCm" in output: backend = "ROCm"
        elif "Metal" in output: backend = "Metal"
        elif "CPU" in output and "CUDA" not in output: backend = "CPU"
         params = "Unknown"
        param_match = re.search(r'\|\s*([0-9\.]+)\s*B\s*\|', output)
        if param_match:
            params = f"{param_match.group(1)}B"
         return {
            "pp": pp_val, "tg": tg_val, "tg_std": tg_std, 
            "backend": backend, "params": params, "raw": output,
            "version": get_llama_version()
        }
    except Exception as e:
        return {"error": str(e)}
 def load_db():
    if not os.path.exists(DB_FILE):
        return pd.DataFrame(columns=COLS)
    try:
        df = pd.read_csv(DB_FILE)
        for col in COLS:
            if col not in df.columns:
                if col.startswith("HE_"): df[col] = 0
                elif col == "Rating_Speed": df[col] = "Untested"
                elif col == "Backend": df[col] = "-"
                elif col == "Params_Actual": df[col] = "-"
                elif col == "Llama_Version": df[col] = "-"
                else: df[col] = 0.0
        return df[COLS]
    except:
        return pd.DataFrame(columns=COLS)
 def save_db(df):
    df.to_csv(DB_FILE, index=False)
 def load_history():
    if not os.path.exists(HISTORY_FILE):
        return pd.DataFrame(columns=["Timestamp", "Model_Name", "Llama_Version", "Speed", "Speed_StDev", "Backend"])
    return pd.read_csv(HISTORY_FILE)
 def append_history(model_name, version, speed, stdev, backend):
    df_hist = load_history()
    new_row = {
        "Timestamp": datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
        "Model_Name": model_name,
        "Llama_Version": version,
        "Speed": speed,
        "Speed_StDev": stdev,
        "Backend": backend
    }
    if df_hist.empty:
        df_hist = pd.DataFrame([new_row])
    else:
        df_hist = pd.concat([df_hist, pd.DataFrame([new_row])], ignore_index=True)
    df_hist.to_csv(HISTORY_FILE, index=False)
 # ==========================================
# MAIN UI
# ==========================================
st.title("🦙 Llama.cpp Command Center")
st.caption(get_hardware_info())
 if "last_bench_time" not in st.session_state:
    st.session_state.last_bench_time = 0
 df = load_db()
 # Auto-Discovery of new models
if os.path.exists(MODELS_DIR):
    disk_models = [f for f in os.listdir(MODELS_DIR) if f.endswith(".gguf") and "mmproj" not in f]
    db_models = df['Model_Name'].tolist()
    new_found = [m for m in disk_models if m not in db_models]
     if new_found:
        new_rows = []
        for f in new_found:
            s, r, q, b = parse_model_properties(f)
            row = {c: 0 for c in COLS}
            row.update({
                "Model_Name": f, "Size_GB": s, "Quant": q, "Role": r, "B_Tag": b,
                "Gen_TG128": 0.0, "Ingest_PP512": 0.0, "Rating_Speed": "Untested",
                "Backend": "-", "Params_Actual": "-", "Llama_Version": "-"
            })
            new_rows.append(row)
         if new_rows:
            df_new = pd.DataFrame(new_rows)[COLS]
            df = pd.concat([df, df_new], ignore_index=True)
            save_db(df)
            st.toast(f"Auto-indexed {len(new_rows)} new models!", icon="🆕")
 # Sidebar Actions
st.sidebar.header("Actions")
if st.sidebar.button("🔄 Scan & Index Library"):
    if not os.path.exists(MODELS_DIR):
        st.error(f"Models folder not found at: {MODELS_DIR}")
    else:
        files = [f for f in os.listdir(MODELS_DIR) if f.endswith(".gguf") and "mmproj" not in f]
        new_rows = []
        for f in files:
            existing = df[df['Model_Name'] == f]
            if not existing.empty:
                row = existing.iloc[0].to_dict()
                for c in COLS:
                    if c not in row: row[c] = 0
            else:
                s, r, q, b = parse_model_properties(f)
                row = {c: 0 for c in COLS}
                row.update({
                    "Model_Name": f, "Size_GB": s, "Quant": q, "Role": r, "B_Tag": b,
                    "Gen_TG128": 0.0, "Ingest_PP512": 0.0, "Rating_Speed": "Untested",
                    "Backend": "-", "Params_Actual": "-", "Llama_Version": "-"
                })
            new_rows.append(row)
        df = pd.DataFrame(new_rows)[COLS]
        save_db(df)
        st.sidebar.success(f"Indexed {len(new_rows)} models!")
        st.rerun()
 # --- TABS ---
tab1, tab2, tab3, tab4, tab5 = st.tabs(["📊 Speed", "📝 Human Ratings", "⚖️ Analysis & Winners", "🆚 Versions", "🚀 Benchmarking"])
 with tab1:
    st.subheader("Speed Leaderboard")
    if not df.empty and df['Gen_TG128'].max() > 0:
        clean = df[df['Gen_TG128'] > 0].sort_values(by="Gen_TG128", ascending=False).head(15)
        st.bar_chart(clean, x="Model_Name", y="Gen_TG128", color="Quant")
           display_cols = ["Model_Name", "Params_Actual", "Backend", "Llama_Version", "Quant", "Gen_TG128", "Gen_TG128_StDev", "Rating_Speed"]
        st.dataframe(clean[display_cols], hide_index=True, width="stretch")
    else:
        st.info("No benchmarks run yet.")
 with tab2:
    st.subheader("Human Ratings Matrix")
    st.info("Edit your scores (0-10) below. 'Overall' is auto-calculated when you Save.")
     if not df.empty:
        # Determine columns to edit (everything starting with HE_)
        he_cols = [c for c in COLS if c.startswith("HE_")]
         # Prepare Dataframe for Editor (Index = Model_Name to lock it)
        df_for_editing = df[["Model_Name"] + he_cols].set_index("Model_Name")
         edited_df = st.data_editor(
            df_for_editing, 
            key="human_eval_editor", 
            width="stretch",
            column_config={
                # Lock the Overall column so it's read-only
                "HE_Overall": st.column_config.NumberColumn("⭐ Overall (Auto)", format="%.1f", disabled=True),
            }
        )
         if st.button("💾 Save & Calculate Averages"):
            # Update master DF from the edited version
            for model_name, row in edited_df.iterrows():
                master_idx = df.index[df['Model_Name'] == model_name]
                 if not master_idx.empty:
                    # 1. Update the specific category columns
                    # We skip HE_Overall here because we calculate it below
                    for col in he_cols:
                        if col != "HE_Overall":
                            df.loc[master_idx, col] = row[col]
                     # 2. AUTO-TABULATE LOGIC
                    # Extract the scores for this model
                    # We only want he_cols excluding overall
                    cat_scores = [row[c] for c in he_cols if c != "HE_Overall"]
                     # Filter: Only count scores > 0 (Ignore "Unrated")
                    valid_scores = [s for s in cat_scores if s > 0]
                     if valid_scores:
                        new_avg = sum(valid_scores) / len(valid_scores)
                    else:
                        new_avg = 0
                     # 3. Write the calculated average
                    df.loc[master_idx, "HE_Overall"] = round(new_avg, 1)
             save_db(df)
            st.success("Saved! Overall scores have been updated.")
            time.sleep(1) # Brief pause for visual feedback
            st.rerun()    # Refresh page to show new numbers
 with tab3:
    st.subheader("⚖️ The Value Matrix (Speed vs. Smarts)")
    if df.empty or df['Gen_TG128'].max() == 0:
        st.warning("Please run benchmarks and add some human ratings to unlock this tab.")
    else:
        col_ctrl1, col_ctrl2 = st.columns([1, 2])
        with col_ctrl1:
            st.markdown("#### 🎚️ Your Priorities")
            weight = st.slider("Preference Balance:", 0, 100, 50, format="%d%% Quality")
            w_qual = weight / 100.0
            w_speed = 1.0 - w_qual
            st.caption(f"Weights: Speed {int(w_speed*100)}% | Quality {int(w_qual*100)}%")
         adf = df.copy()
        max_speed = adf['Gen_TG128'].max()
        if max_speed == 0: max_speed = 1
         adf['Norm_Speed'] = adf['Gen_TG128'] / max_speed
        adf['Norm_Qual'] = adf['HE_Overall'] / 10.0
        adf['Win_Score'] = (adf['Norm_Speed'] * w_speed * 100) + (adf['Norm_Qual'] * w_qual * 100)
        winners = adf.sort_values(by="Win_Score", ascending=False)
         st.markdown(f"### 🏆 Top Models based on your {weight}% Quality Preference")
        winners['Win_Score'] = winners['Win_Score'].round(1)
        disp_cols = ["Model_Name", "Win_Score", "Gen_TG128", "HE_Overall", "Quant", "Size_GB"]
         st.dataframe(
            winners[disp_cols].head(15), 
            hide_index=True, 
            width="stretch",
            column_config={
                "Win_Score": st.column_config.ProgressColumn("Composite Score", format="%.1f", min_value=0, max_value=100),
                "Gen_TG128": st.column_config.NumberColumn("Speed (t/s)", format="%.1f"),
                "HE_Overall": st.column_config.NumberColumn("Quality", format="%.1f"),
            }
        )
   with tab4:
    st.subheader("🆚 Version Comparison")
    st.info("Compare how different versions of Llama.cpp perform on the same model.")
     df_hist = load_history()
    if df_hist.empty:
        st.warning("No history found. Run some benchmarks first!")
    else:
        # Filter logic
        models_in_hist = df_hist['Model_Name'].unique()
        selected_model = st.selectbox("Select Model to Compare", models_in_hist)
         if selected_model:
            model_data = df_hist[df_hist['Model_Name'] == selected_model].sort_values(by="Timestamp")
             if len(model_data) < 2:
                st.warning("Need at least 2 benchmark runs for this model to compare.")
                st.dataframe(model_data)
            else:
                # Chart
                st.line_chart(model_data, x="Llama_Version", y="Speed")
                 # Pivot for easier reading
                st.markdown("### 📋 History Log")
                st.dataframe(
                    model_data[["Llama_Version", "Speed", "Speed_StDev", "Timestamp", "Backend"]].style.highlight_max(axis=0, subset=["Speed"]),
                    use_container_width=True,
                    hide_index=True
                )
                 # Calcs
                best = model_data.loc[model_data['Speed'].idxmax()]
                worst = model_data.loc[model_data['Speed'].idxmin()]
                diff = best['Speed'] - worst['Speed']
                pct_diff = (diff / worst['Speed']) * 100 if worst['Speed'] > 0 else 0
                 st.metric(
                    label=f"Best Version ({best['Llama_Version']})", 
                    value=f"{best['Speed']} t/s",
                    delta=f"{pct_diff:.1f}% vs {worst['Llama_Version']}"
                )
 with tab5:
    st.subheader("Benchmarking Station")
    if not df.empty:
        model = st.selectbox("Select Model", df['Model_Name'].unique())
         if st.button(f"🔥 Run Benchmark: {model}"):
            now = time.time()
            if now - st.session_state.last_bench_time < 2.0:
                st.warning("⏳ Too fast!")
            else:
                st.session_state.last_bench_time = now
                with st.spinner("Running llama-bench..."):
                    data = run_benchmark(model)
                 if "error" in data:
                    st.error(f"Failed: {data['error']}")
                else:
                    tg = data['tg']
                    rating = "SLOW 🔴"
                    if tg > 10: rating = "USABLE 🟡"
                    if tg > 25: rating = "SMOOTH 🟢"
                    if tg > 50: rating = "GODLIKE 🚀"
                     ver_str = data['version']
                    st.success(f"Speed: {tg} t/s (±{data['tg_std']}) | Ver: {ver_str}")
                     idx = df.index[df['Model_Name'] == model].tolist()[0]
                    df.at[idx, 'Gen_TG128'] = tg
                    df.at[idx, 'Gen_TG128_StDev'] = data['tg_std']
                    df.at[idx, 'Ingest_PP512'] = data['pp']
                    df.at[idx, 'Rating_Speed'] = rating
                    df.at[idx, 'Backend'] = data['backend']
                    df.at[idx, 'Params_Actual'] = data['params']
                    df.at[idx, 'Params_Actual'] = data['params']
                    df.at[idx, 'Llama_Version'] = ver_str
                    save_db(df)
                     # Append to History
                    append_history(model, ver_str, tg, data['tg_std'], data['backend'])
                     with st.expander("📄 View Raw Benchmark Output", expanded=True):
                        st.code(data['raw'], language="text")
                    st.caption(f"Log saved to: {LOGS_DIR}")
    else:
        st.warning("Library is empty.")

Comments