import streamlit as st
import pandas as pd
import subprocess
import os
import re
import datetime
import time
# ==========================================
# CONFIGURATION
# ==========================================
BASE_DIR = os.getcwd()
MODELS_DIR = os.path.join(BASE_DIR, "models")
LOGS_DIR = os.path.join(BASE_DIR, "bench_logs")
BENCH_EXE = os.path.join(BASE_DIR, "llama-bench.exe")
CLI_EXE = os.path.join(BASE_DIR, "llama-cli.exe")
DB_FILE = os.path.join(BASE_DIR, "model_library_benchmark.csv")
HISTORY_FILE = os.path.join(BASE_DIR, "benchmark_history.csv")
KEYWORDS = ["NSFW", "Uncensored", "Amoral", "Dark", "Fallen", "Rivermind", "Snowpiercer", "Abyss", "Erebus", "Lewd"]
if not os.path.exists(LOGS_DIR):
os.makedirs(LOGS_DIR)
# --- SCHEMA ---
COLS = [
# Identity
"Model_Name", "Size_GB", "Quant", "Role", "B_Tag", "Params_Actual", "Backend", "Llama_Version",
# Metrics
"Gen_TG128", "Gen_TG128_StDev", "Ingest_PP512", "Rating_Speed",
# Human Eval
"HE_Overall", "HE_Math", "HE_Science", "HE_History", "HE_Music",
"HE_English", "HE_Coding", "HE_Logic", "HE_Creative", "HE_RP_Quality",
"HE_Personality", "HE_Emotes", "HE_Formatting", "HE_Instructions",
"HE_Censorship", "HE_Memory", "HE_Summarize", "HE_Translation",
"HE_JSON", "HE_RAG", "HE_Vision", "HE_Fun_Factor"
]
st.set_page_config(page_title="Llama.cpp Librarian", layout="wide", page_icon="🦙")
# ==========================================
# FUNCTIONS
# ==========================================
def get_hardware_info():
try:
gpu_info = subprocess.check_output("nvidia-smi --query-gpu=name,memory.total,memory.free --format=csv,noheader,nounits", shell=True).decode().strip()
gpu_name, total_vram, free_vram = gpu_info.split(',')
return f"{gpu_name} | VRAM: {free_vram.strip()}MB Free / {total_vram.strip()}MB Total"
except:
return "GPU Detection Failed"
def get_llama_version():
try:
# Try to get version from llama-cli
if os.path.exists(CLI_EXE):
cmd = [CLI_EXE, "--version"]
result = subprocess.run(cmd, capture_output=True, text=True, encoding='utf-8')
output = result.stdout + result.stderr
# Look for "version: 7549 (c9ced4910)"
match = re.search(r'version:\s*(\d+)\s*\(([^)]+)\)', output)
if match:
return f"b{match.group(1)}-{match.group(2)[:7]}"
return "Unknown Version"
return "CLI Not Found"
except:
return "Version Check Failed"
def parse_model_properties(filename):
name = filename
try:
size_gb = round(os.path.getsize(os.path.join(MODELS_DIR, filename)) / (1024**3))
except:
size_gb = 0
role = "Specialized/NSFW" if any(k.lower() in name.lower() for k in KEYWORDS) else "General"
quant = "Unknown"
quants = ["Q8_0", "Q6_K", "Q5_K_M", "Q5_K_S", "Q5_0", "Q4_K_M", "Q4_K_S", "Q4_0", "IQ3", "Q2_K", "F16", "F32"]
for q in quants:
if q.lower() in name.lower():
quant = q
break
b_tag = "Other"
b_match = re.search(r'(\d+(?:\.\d+)?)B', name, re.IGNORECASE)
if b_match:
b_tag = b_match.group(0).upper()
return size_gb, role, quant, b_tag
def run_benchmark(model_name):
model_path = os.path.join(MODELS_DIR, model_name)
cmd = [BENCH_EXE, "-m", model_path, "-t", "8", "-ngl", "999"]
try:
result = subprocess.run(cmd, capture_output=True, text=True, encoding='utf-8')
output = result.stdout
# Log
timestamp = datetime.datetime.now().strftime("%Y-%m-%d_%H-%M-%S")
log_file = os.path.join(LOGS_DIR, f"{model_name}_{timestamp}.log")
with open(log_file, "w", encoding="utf-8") as f:
f.write(f"COMMAND: {' '.join(cmd)}\n")
f.write("="*40 + "\n")
f.write(output)
f.write("\n" + "="*40 + "\n")
f.write(result.stderr)
# Parse
pp_match = re.search(r'\|\s*pp512\s*\|\s*([0-9\.]+)\s*\+/-\s*([0-9\.]+)', output)
tg_match = re.search(r'\|\s*tg128\s*\|\s*([0-9\.]+)\s*\+/-\s*([0-9\.]+)', output)
if not pp_match: pp_match = re.search(r'\|\s*pp512\s*\|\s*([0-9\.]+)', output)
if not tg_match: tg_match = re.search(r'\|\s*tg128\s*\|\s*([0-9\.]+)', output)
pp_val = float(pp_match.group(1)) if pp_match else 0.0
tg_val = float(tg_match.group(1)) if tg_match else 0.0
tg_std = float(tg_match.group(2)) if tg_match and len(tg_match.groups()) > 1 else 0.0
backend = "Unknown"
if "CUDA" in output: backend = "CUDA"
elif "ROCm" in output: backend = "ROCm"
elif "Metal" in output: backend = "Metal"
elif "CPU" in output and "CUDA" not in output: backend = "CPU"
params = "Unknown"
param_match = re.search(r'\|\s*([0-9\.]+)\s*B\s*\|', output)
if param_match:
params = f"{param_match.group(1)}B"
return {
"pp": pp_val, "tg": tg_val, "tg_std": tg_std,
"backend": backend, "params": params, "raw": output,
"version": get_llama_version()
}
except Exception as e:
return {"error": str(e)}
def load_db():
if not os.path.exists(DB_FILE):
return pd.DataFrame(columns=COLS)
try:
df = pd.read_csv(DB_FILE)
for col in COLS:
if col not in df.columns:
if col.startswith("HE_"): df[col] = 0
elif col == "Rating_Speed": df[col] = "Untested"
elif col == "Backend": df[col] = "-"
elif col == "Params_Actual": df[col] = "-"
elif col == "Llama_Version": df[col] = "-"
else: df[col] = 0.0
return df[COLS]
except:
return pd.DataFrame(columns=COLS)
def save_db(df):
df.to_csv(DB_FILE, index=False)
def load_history():
if not os.path.exists(HISTORY_FILE):
return pd.DataFrame(columns=["Timestamp", "Model_Name", "Llama_Version", "Speed", "Speed_StDev", "Backend"])
return pd.read_csv(HISTORY_FILE)
def append_history(model_name, version, speed, stdev, backend):
df_hist = load_history()
new_row = {
"Timestamp": datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
"Model_Name": model_name,
"Llama_Version": version,
"Speed": speed,
"Speed_StDev": stdev,
"Backend": backend
}
if df_hist.empty:
df_hist = pd.DataFrame([new_row])
else:
df_hist = pd.concat([df_hist, pd.DataFrame([new_row])], ignore_index=True)
df_hist.to_csv(HISTORY_FILE, index=False)
# ==========================================
# MAIN UI
# ==========================================
st.title("🦙 Llama.cpp Command Center")
st.caption(get_hardware_info())
if "last_bench_time" not in st.session_state:
st.session_state.last_bench_time = 0
df = load_db()
# Auto-Discovery of new models
if os.path.exists(MODELS_DIR):
disk_models = [f for f in os.listdir(MODELS_DIR) if f.endswith(".gguf") and "mmproj" not in f]
db_models = df['Model_Name'].tolist()
new_found = [m for m in disk_models if m not in db_models]
if new_found:
new_rows = []
for f in new_found:
s, r, q, b = parse_model_properties(f)
row = {c: 0 for c in COLS}
row.update({
"Model_Name": f, "Size_GB": s, "Quant": q, "Role": r, "B_Tag": b,
"Gen_TG128": 0.0, "Ingest_PP512": 0.0, "Rating_Speed": "Untested",
"Backend": "-", "Params_Actual": "-", "Llama_Version": "-"
})
new_rows.append(row)
if new_rows:
df_new = pd.DataFrame(new_rows)[COLS]
df = pd.concat([df, df_new], ignore_index=True)
save_db(df)
st.toast(f"Auto-indexed {len(new_rows)} new models!", icon="🆕")
# Sidebar Actions
st.sidebar.header("Actions")
if st.sidebar.button("🔄 Scan & Index Library"):
if not os.path.exists(MODELS_DIR):
st.error(f"Models folder not found at: {MODELS_DIR}")
else:
files = [f for f in os.listdir(MODELS_DIR) if f.endswith(".gguf") and "mmproj" not in f]
new_rows = []
for f in files:
existing = df[df['Model_Name'] == f]
if not existing.empty:
row = existing.iloc[0].to_dict()
for c in COLS:
if c not in row: row[c] = 0
else:
s, r, q, b = parse_model_properties(f)
row = {c: 0 for c in COLS}
row.update({
"Model_Name": f, "Size_GB": s, "Quant": q, "Role": r, "B_Tag": b,
"Gen_TG128": 0.0, "Ingest_PP512": 0.0, "Rating_Speed": "Untested",
"Backend": "-", "Params_Actual": "-", "Llama_Version": "-"
})
new_rows.append(row)
df = pd.DataFrame(new_rows)[COLS]
save_db(df)
st.sidebar.success(f"Indexed {len(new_rows)} models!")
st.rerun()
# --- TABS ---
tab1, tab2, tab3, tab4, tab5 = st.tabs(["📊 Speed", "📝 Human Ratings", "⚖️ Analysis & Winners", "🆚 Versions", "🚀 Benchmarking"])
with tab1:
st.subheader("Speed Leaderboard")
if not df.empty and df['Gen_TG128'].max() > 0:
clean = df[df['Gen_TG128'] > 0].sort_values(by="Gen_TG128", ascending=False).head(15)
st.bar_chart(clean, x="Model_Name", y="Gen_TG128", color="Quant")
display_cols = ["Model_Name", "Params_Actual", "Backend", "Llama_Version", "Quant", "Gen_TG128", "Gen_TG128_StDev", "Rating_Speed"]
st.dataframe(clean[display_cols], hide_index=True, width="stretch")
else:
st.info("No benchmarks run yet.")
with tab2:
st.subheader("Human Ratings Matrix")
st.info("Edit your scores (0-10) below. 'Overall' is auto-calculated when you Save.")
if not df.empty:
# Determine columns to edit (everything starting with HE_)
he_cols = [c for c in COLS if c.startswith("HE_")]
# Prepare Dataframe for Editor (Index = Model_Name to lock it)
df_for_editing = df[["Model_Name"] + he_cols].set_index("Model_Name")
edited_df = st.data_editor(
df_for_editing,
key="human_eval_editor",
width="stretch",
column_config={
# Lock the Overall column so it's read-only
"HE_Overall": st.column_config.NumberColumn("⭐ Overall (Auto)", format="%.1f", disabled=True),
}
)
if st.button("💾 Save & Calculate Averages"):
# Update master DF from the edited version
for model_name, row in edited_df.iterrows():
master_idx = df.index[df['Model_Name'] == model_name]
if not master_idx.empty:
# 1. Update the specific category columns
# We skip HE_Overall here because we calculate it below
for col in he_cols:
if col != "HE_Overall":
df.loc[master_idx, col] = row[col]
# 2. AUTO-TABULATE LOGIC
# Extract the scores for this model
# We only want he_cols excluding overall
cat_scores = [row[c] for c in he_cols if c != "HE_Overall"]
# Filter: Only count scores > 0 (Ignore "Unrated")
valid_scores = [s for s in cat_scores if s > 0]
if valid_scores:
new_avg = sum(valid_scores) / len(valid_scores)
else:
new_avg = 0
# 3. Write the calculated average
df.loc[master_idx, "HE_Overall"] = round(new_avg, 1)
save_db(df)
st.success("Saved! Overall scores have been updated.")
time.sleep(1) # Brief pause for visual feedback
st.rerun() # Refresh page to show new numbers
with tab3:
st.subheader("⚖️ The Value Matrix (Speed vs. Smarts)")
if df.empty or df['Gen_TG128'].max() == 0:
st.warning("Please run benchmarks and add some human ratings to unlock this tab.")
else:
col_ctrl1, col_ctrl2 = st.columns([1, 2])
with col_ctrl1:
st.markdown("#### 🎚️ Your Priorities")
weight = st.slider("Preference Balance:", 0, 100, 50, format="%d%% Quality")
w_qual = weight / 100.0
w_speed = 1.0 - w_qual
st.caption(f"Weights: Speed {int(w_speed*100)}% | Quality {int(w_qual*100)}%")
adf = df.copy()
max_speed = adf['Gen_TG128'].max()
if max_speed == 0: max_speed = 1
adf['Norm_Speed'] = adf['Gen_TG128'] / max_speed
adf['Norm_Qual'] = adf['HE_Overall'] / 10.0
adf['Win_Score'] = (adf['Norm_Speed'] * w_speed * 100) + (adf['Norm_Qual'] * w_qual * 100)
winners = adf.sort_values(by="Win_Score", ascending=False)
st.markdown(f"### 🏆 Top Models based on your {weight}% Quality Preference")
winners['Win_Score'] = winners['Win_Score'].round(1)
disp_cols = ["Model_Name", "Win_Score", "Gen_TG128", "HE_Overall", "Quant", "Size_GB"]
st.dataframe(
winners[disp_cols].head(15),
hide_index=True,
width="stretch",
column_config={
"Win_Score": st.column_config.ProgressColumn("Composite Score", format="%.1f", min_value=0, max_value=100),
"Gen_TG128": st.column_config.NumberColumn("Speed (t/s)", format="%.1f"),
"HE_Overall": st.column_config.NumberColumn("Quality", format="%.1f"),
}
)
with tab4:
st.subheader("🆚 Version Comparison")
st.info("Compare how different versions of Llama.cpp perform on the same model.")
df_hist = load_history()
if df_hist.empty:
st.warning("No history found. Run some benchmarks first!")
else:
# Filter logic
models_in_hist = df_hist['Model_Name'].unique()
selected_model = st.selectbox("Select Model to Compare", models_in_hist)
if selected_model:
model_data = df_hist[df_hist['Model_Name'] == selected_model].sort_values(by="Timestamp")
if len(model_data) < 2:
st.warning("Need at least 2 benchmark runs for this model to compare.")
st.dataframe(model_data)
else:
# Chart
st.line_chart(model_data, x="Llama_Version", y="Speed")
# Pivot for easier reading
st.markdown("### 📋 History Log")
st.dataframe(
model_data[["Llama_Version", "Speed", "Speed_StDev", "Timestamp", "Backend"]].style.highlight_max(axis=0, subset=["Speed"]),
use_container_width=True,
hide_index=True
)
# Calcs
best = model_data.loc[model_data['Speed'].idxmax()]
worst = model_data.loc[model_data['Speed'].idxmin()]
diff = best['Speed'] - worst['Speed']
pct_diff = (diff / worst['Speed']) * 100 if worst['Speed'] > 0 else 0
st.metric(
label=f"Best Version ({best['Llama_Version']})",
value=f"{best['Speed']} t/s",
delta=f"{pct_diff:.1f}% vs {worst['Llama_Version']}"
)
with tab5:
st.subheader("Benchmarking Station")
if not df.empty:
model = st.selectbox("Select Model", df['Model_Name'].unique())
if st.button(f"🔥 Run Benchmark: {model}"):
now = time.time()
if now - st.session_state.last_bench_time < 2.0:
st.warning("⏳ Too fast!")
else:
st.session_state.last_bench_time = now
with st.spinner("Running llama-bench..."):
data = run_benchmark(model)
if "error" in data:
st.error(f"Failed: {data['error']}")
else:
tg = data['tg']
rating = "SLOW 🔴"
if tg > 10: rating = "USABLE 🟡"
if tg > 25: rating = "SMOOTH 🟢"
if tg > 50: rating = "GODLIKE 🚀"
ver_str = data['version']
st.success(f"Speed: {tg} t/s (±{data['tg_std']}) | Ver: {ver_str}")
idx = df.index[df['Model_Name'] == model].tolist()[0]
df.at[idx, 'Gen_TG128'] = tg
df.at[idx, 'Gen_TG128_StDev'] = data['tg_std']
df.at[idx, 'Ingest_PP512'] = data['pp']
df.at[idx, 'Rating_Speed'] = rating
df.at[idx, 'Backend'] = data['backend']
df.at[idx, 'Params_Actual'] = data['params']
df.at[idx, 'Params_Actual'] = data['params']
df.at[idx, 'Llama_Version'] = ver_str
save_db(df)
# Append to History
append_history(model, ver_str, tg, data['tg_std'], data['backend'])
with st.expander("📄 View Raw Benchmark Output", expanded=True):
st.code(data['raw'], language="text")
st.caption(f"Log saved to: {LOGS_DIR}")
else:
st.warning("Library is empty.")
Comments