import streamlit as st import pandas as pd import subprocess import os import re import datetime import time # ========================================== # CONFIGURATION # ========================================== BASE_DIR = os.getcwd() MODELS_DIR = os.path.join(BASE_DIR, "models") LOGS_DIR = os.path.join(BASE_DIR, "bench_logs") BENCH_EXE = os.path.join(BASE_DIR, "llama-bench.exe") CLI_EXE = os.path.join(BASE_DIR, "llama-cli.exe") DB_FILE = os.path.join(BASE_DIR, "model_library_benchmark.csv") HISTORY_FILE = os.path.join(BASE_DIR, "benchmark_history.csv") KEYWORDS = ["NSFW", "Uncensored", "Amoral", "Dark", "Fallen", "Rivermind", "Snowpiercer", "Abyss", "Erebus", "Lewd"] if not os.path.exists(LOGS_DIR): os.makedirs(LOGS_DIR) # --- SCHEMA --- COLS = [ # Identity "Model_Name", "Size_GB", "Quant", "Role", "B_Tag", "Params_Actual", "Backend", "Llama_Version", # Metrics "Gen_TG128", "Gen_TG128_StDev", "Ingest_PP512", "Rating_Speed", # Human Eval "HE_Overall", "HE_Math", "HE_Science", "HE_History", "HE_Music", "HE_English", "HE_Coding", "HE_Logic", "HE_Creative", "HE_RP_Quality", "HE_Personality", "HE_Emotes", "HE_Formatting", "HE_Instructions", "HE_Censorship", "HE_Memory", "HE_Summarize", "HE_Translation", "HE_JSON", "HE_RAG", "HE_Vision", "HE_Fun_Factor" ] st.set_page_config(page_title="Llama.cpp Librarian", layout="wide", page_icon="🦙") # ========================================== # FUNCTIONS # ========================================== def get_hardware_info(): try: gpu_info = subprocess.check_output("nvidia-smi --query-gpu=name,memory.total,memory.free --format=csv,noheader,nounits", shell=True).decode().strip() gpu_name, total_vram, free_vram = gpu_info.split(',') return f"{gpu_name} | VRAM: {free_vram.strip()}MB Free / {total_vram.strip()}MB Total" except: return "GPU Detection Failed" def get_llama_version(): try: # Try to get version from llama-cli if os.path.exists(CLI_EXE): cmd = [CLI_EXE, "--version"] result = subprocess.run(cmd, capture_output=True, text=True, encoding='utf-8') output = result.stdout + result.stderr # Look for "version: 7549 (c9ced4910)" match = re.search(r'version:\s*(\d+)\s*\(([^)]+)\)', output) if match: return f"b{match.group(1)}-{match.group(2)[:7]}" return "Unknown Version" return "CLI Not Found" except: return "Version Check Failed" def parse_model_properties(filename): name = filename try: size_gb = round(os.path.getsize(os.path.join(MODELS_DIR, filename)) / (1024**3)) except: size_gb = 0 role = "Specialized/NSFW" if any(k.lower() in name.lower() for k in KEYWORDS) else "General" quant = "Unknown" quants = ["Q8_0", "Q6_K", "Q5_K_M", "Q5_K_S", "Q5_0", "Q4_K_M", "Q4_K_S", "Q4_0", "IQ3", "Q2_K", "F16", "F32"] for q in quants: if q.lower() in name.lower(): quant = q break b_tag = "Other" b_match = re.search(r'(\d+(?:\.\d+)?)B', name, re.IGNORECASE) if b_match: b_tag = b_match.group(0).upper() return size_gb, role, quant, b_tag def run_benchmark(model_name): model_path = os.path.join(MODELS_DIR, model_name) cmd = [BENCH_EXE, "-m", model_path, "-t", "8", "-ngl", "999"] try: result = subprocess.run(cmd, capture_output=True, text=True, encoding='utf-8') output = result.stdout # Log timestamp = datetime.datetime.now().strftime("%Y-%m-%d_%H-%M-%S") log_file = os.path.join(LOGS_DIR, f"{model_name}_{timestamp}.log") with open(log_file, "w", encoding="utf-8") as f: f.write(f"COMMAND: {' '.join(cmd)}\n") f.write("="*40 + "\n") f.write(output) f.write("\n" + "="*40 + "\n") f.write(result.stderr) # Parse pp_match = re.search(r'\|\s*pp512\s*\|\s*([0-9\.]+)\s*\+/-\s*([0-9\.]+)', output) tg_match = re.search(r'\|\s*tg128\s*\|\s*([0-9\.]+)\s*\+/-\s*([0-9\.]+)', output) if not pp_match: pp_match = re.search(r'\|\s*pp512\s*\|\s*([0-9\.]+)', output) if not tg_match: tg_match = re.search(r'\|\s*tg128\s*\|\s*([0-9\.]+)', output) pp_val = float(pp_match.group(1)) if pp_match else 0.0 tg_val = float(tg_match.group(1)) if tg_match else 0.0 tg_std = float(tg_match.group(2)) if tg_match and len(tg_match.groups()) > 1 else 0.0 backend = "Unknown" if "CUDA" in output: backend = "CUDA" elif "ROCm" in output: backend = "ROCm" elif "Metal" in output: backend = "Metal" elif "CPU" in output and "CUDA" not in output: backend = "CPU" params = "Unknown" param_match = re.search(r'\|\s*([0-9\.]+)\s*B\s*\|', output) if param_match: params = f"{param_match.group(1)}B" return { "pp": pp_val, "tg": tg_val, "tg_std": tg_std, "backend": backend, "params": params, "raw": output, "version": get_llama_version() } except Exception as e: return {"error": str(e)} def load_db(): if not os.path.exists(DB_FILE): return pd.DataFrame(columns=COLS) try: df = pd.read_csv(DB_FILE) for col in COLS: if col not in df.columns: if col.startswith("HE_"): df[col] = 0 elif col == "Rating_Speed": df[col] = "Untested" elif col == "Backend": df[col] = "-" elif col == "Params_Actual": df[col] = "-" elif col == "Llama_Version": df[col] = "-" else: df[col] = 0.0 return df[COLS] except: return pd.DataFrame(columns=COLS) def save_db(df): df.to_csv(DB_FILE, index=False) def load_history(): if not os.path.exists(HISTORY_FILE): return pd.DataFrame(columns=["Timestamp", "Model_Name", "Llama_Version", "Speed", "Speed_StDev", "Backend"]) return pd.read_csv(HISTORY_FILE) def append_history(model_name, version, speed, stdev, backend): df_hist = load_history() new_row = { "Timestamp": datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S"), "Model_Name": model_name, "Llama_Version": version, "Speed": speed, "Speed_StDev": stdev, "Backend": backend } if df_hist.empty: df_hist = pd.DataFrame([new_row]) else: df_hist = pd.concat([df_hist, pd.DataFrame([new_row])], ignore_index=True) df_hist.to_csv(HISTORY_FILE, index=False) # ========================================== # MAIN UI # ========================================== st.title("🦙 Llama.cpp Command Center") st.caption(get_hardware_info()) if "last_bench_time" not in st.session_state: st.session_state.last_bench_time = 0 df = load_db() # Auto-Discovery of new models if os.path.exists(MODELS_DIR): disk_models = [f for f in os.listdir(MODELS_DIR) if f.endswith(".gguf") and "mmproj" not in f] db_models = df['Model_Name'].tolist() new_found = [m for m in disk_models if m not in db_models] if new_found: new_rows = [] for f in new_found: s, r, q, b = parse_model_properties(f) row = {c: 0 for c in COLS} row.update({ "Model_Name": f, "Size_GB": s, "Quant": q, "Role": r, "B_Tag": b, "Gen_TG128": 0.0, "Ingest_PP512": 0.0, "Rating_Speed": "Untested", "Backend": "-", "Params_Actual": "-", "Llama_Version": "-" }) new_rows.append(row) if new_rows: df_new = pd.DataFrame(new_rows)[COLS] df = pd.concat([df, df_new], ignore_index=True) save_db(df) st.toast(f"Auto-indexed {len(new_rows)} new models!", icon="🆕") # Sidebar Actions st.sidebar.header("Actions") if st.sidebar.button("🔄 Scan & Index Library"): if not os.path.exists(MODELS_DIR): st.error(f"Models folder not found at: {MODELS_DIR}") else: files = [f for f in os.listdir(MODELS_DIR) if f.endswith(".gguf") and "mmproj" not in f] new_rows = [] for f in files: existing = df[df['Model_Name'] == f] if not existing.empty: row = existing.iloc[0].to_dict() for c in COLS: if c not in row: row[c] = 0 else: s, r, q, b = parse_model_properties(f) row = {c: 0 for c in COLS} row.update({ "Model_Name": f, "Size_GB": s, "Quant": q, "Role": r, "B_Tag": b, "Gen_TG128": 0.0, "Ingest_PP512": 0.0, "Rating_Speed": "Untested", "Backend": "-", "Params_Actual": "-", "Llama_Version": "-" }) new_rows.append(row) df = pd.DataFrame(new_rows)[COLS] save_db(df) st.sidebar.success(f"Indexed {len(new_rows)} models!") st.rerun() # --- TABS --- tab1, tab2, tab3, tab4, tab5 = st.tabs(["📊 Speed", "📝 Human Ratings", "⚖️ Analysis & Winners", "🆚 Versions", "🚀 Benchmarking"]) with tab1: st.subheader("Speed Leaderboard") if not df.empty and df['Gen_TG128'].max() > 0: clean = df[df['Gen_TG128'] > 0].sort_values(by="Gen_TG128", ascending=False).head(15) st.bar_chart(clean, x="Model_Name", y="Gen_TG128", color="Quant") display_cols = ["Model_Name", "Params_Actual", "Backend", "Llama_Version", "Quant", "Gen_TG128", "Gen_TG128_StDev", "Rating_Speed"] st.dataframe(clean[display_cols], hide_index=True, width="stretch") else: st.info("No benchmarks run yet.") with tab2: st.subheader("Human Ratings Matrix") st.info("Edit your scores (0-10) below. 'Overall' is auto-calculated when you Save.") if not df.empty: # Determine columns to edit (everything starting with HE_) he_cols = [c for c in COLS if c.startswith("HE_")] # Prepare Dataframe for Editor (Index = Model_Name to lock it) df_for_editing = df[["Model_Name"] + he_cols].set_index("Model_Name") edited_df = st.data_editor( df_for_editing, key="human_eval_editor", width="stretch", column_config={ # Lock the Overall column so it's read-only "HE_Overall": st.column_config.NumberColumn("⭐ Overall (Auto)", format="%.1f", disabled=True), } ) if st.button("💾 Save & Calculate Averages"): # Update master DF from the edited version for model_name, row in edited_df.iterrows(): master_idx = df.index[df['Model_Name'] == model_name] if not master_idx.empty: # 1. Update the specific category columns # We skip HE_Overall here because we calculate it below for col in he_cols: if col != "HE_Overall": df.loc[master_idx, col] = row[col] # 2. AUTO-TABULATE LOGIC # Extract the scores for this model # We only want he_cols excluding overall cat_scores = [row[c] for c in he_cols if c != "HE_Overall"] # Filter: Only count scores > 0 (Ignore "Unrated") valid_scores = [s for s in cat_scores if s > 0] if valid_scores: new_avg = sum(valid_scores) / len(valid_scores) else: new_avg = 0 # 3. Write the calculated average df.loc[master_idx, "HE_Overall"] = round(new_avg, 1) save_db(df) st.success("Saved! Overall scores have been updated.") time.sleep(1) # Brief pause for visual feedback st.rerun() # Refresh page to show new numbers with tab3: st.subheader("⚖️ The Value Matrix (Speed vs. Smarts)") if df.empty or df['Gen_TG128'].max() == 0: st.warning("Please run benchmarks and add some human ratings to unlock this tab.") else: col_ctrl1, col_ctrl2 = st.columns([1, 2]) with col_ctrl1: st.markdown("#### 🎚️ Your Priorities") weight = st.slider("Preference Balance:", 0, 100, 50, format="%d%% Quality") w_qual = weight / 100.0 w_speed = 1.0 - w_qual st.caption(f"Weights: Speed {int(w_speed*100)}% | Quality {int(w_qual*100)}%") adf = df.copy() max_speed = adf['Gen_TG128'].max() if max_speed == 0: max_speed = 1 adf['Norm_Speed'] = adf['Gen_TG128'] / max_speed adf['Norm_Qual'] = adf['HE_Overall'] / 10.0 adf['Win_Score'] = (adf['Norm_Speed'] * w_speed * 100) + (adf['Norm_Qual'] * w_qual * 100) winners = adf.sort_values(by="Win_Score", ascending=False) st.markdown(f"### 🏆 Top Models based on your {weight}% Quality Preference") winners['Win_Score'] = winners['Win_Score'].round(1) disp_cols = ["Model_Name", "Win_Score", "Gen_TG128", "HE_Overall", "Quant", "Size_GB"] st.dataframe( winners[disp_cols].head(15), hide_index=True, width="stretch", column_config={ "Win_Score": st.column_config.ProgressColumn("Composite Score", format="%.1f", min_value=0, max_value=100), "Gen_TG128": st.column_config.NumberColumn("Speed (t/s)", format="%.1f"), "HE_Overall": st.column_config.NumberColumn("Quality", format="%.1f"), } ) with tab4: st.subheader("🆚 Version Comparison") st.info("Compare how different versions of Llama.cpp perform on the same model.") df_hist = load_history() if df_hist.empty: st.warning("No history found. Run some benchmarks first!") else: # Filter logic models_in_hist = df_hist['Model_Name'].unique() selected_model = st.selectbox("Select Model to Compare", models_in_hist) if selected_model: model_data = df_hist[df_hist['Model_Name'] == selected_model].sort_values(by="Timestamp") if len(model_data) < 2: st.warning("Need at least 2 benchmark runs for this model to compare.") st.dataframe(model_data) else: # Chart st.line_chart(model_data, x="Llama_Version", y="Speed") # Pivot for easier reading st.markdown("### 📋 History Log") st.dataframe( model_data[["Llama_Version", "Speed", "Speed_StDev", "Timestamp", "Backend"]].style.highlight_max(axis=0, subset=["Speed"]), use_container_width=True, hide_index=True ) # Calcs best = model_data.loc[model_data['Speed'].idxmax()] worst = model_data.loc[model_data['Speed'].idxmin()] diff = best['Speed'] - worst['Speed'] pct_diff = (diff / worst['Speed']) * 100 if worst['Speed'] > 0 else 0 st.metric( label=f"Best Version ({best['Llama_Version']})", value=f"{best['Speed']} t/s", delta=f"{pct_diff:.1f}% vs {worst['Llama_Version']}" ) with tab5: st.subheader("Benchmarking Station") if not df.empty: model = st.selectbox("Select Model", df['Model_Name'].unique()) if st.button(f"🔥 Run Benchmark: {model}"): now = time.time() if now - st.session_state.last_bench_time < 2.0: st.warning("⏳ Too fast!") else: st.session_state.last_bench_time = now with st.spinner("Running llama-bench..."): data = run_benchmark(model) if "error" in data: st.error(f"Failed: {data['error']}") else: tg = data['tg'] rating = "SLOW 🔴" if tg > 10: rating = "USABLE 🟡" if tg > 25: rating = "SMOOTH 🟢" if tg > 50: rating = "GODLIKE 🚀" ver_str = data['version'] st.success(f"Speed: {tg} t/s (±{data['tg_std']}) | Ver: {ver_str}") idx = df.index[df['Model_Name'] == model].tolist()[0] df.at[idx, 'Gen_TG128'] = tg df.at[idx, 'Gen_TG128_StDev'] = data['tg_std'] df.at[idx, 'Ingest_PP512'] = data['pp'] df.at[idx, 'Rating_Speed'] = rating df.at[idx, 'Backend'] = data['backend'] df.at[idx, 'Params_Actual'] = data['params'] df.at[idx, 'Params_Actual'] = data['params'] df.at[idx, 'Llama_Version'] = ver_str save_db(df) # Append to History append_history(model, ver_str, tg, data['tg_std'], data['backend']) with st.expander("📄 View Raw Benchmark Output", expanded=True): st.code(data['raw'], language="text") st.caption(f"Log saved to: {LOGS_DIR}") else: st.warning("Library is empty.")