runtime data

This commit is contained in:
lachtan
2026-07-22 12:32:23 +02:00
parent 8e66d6b92a
commit 94dccb3758
11 changed files with 1475 additions and 194 deletions

290
scripts/benchmark_ollama.py Normal file
View File

@@ -0,0 +1,290 @@
#!/usr/bin/env python3
"""Benchmark Ollama cloud models: TTFT, throughput, thinking vs response phases.
Measures:
- TTFT-thinking: time to first thinking token
- TTFT-response: time to first response token (visible output)
- Thinking phase: duration and tok/s of reasoning
- Response phase: duration and tok/s of visible output
- Total wall time and overall tok/s
- Token counts (prompt, thinking, response, total eval)
Usage:
python3 benchmark_ollama.py [--host nvidia.hell] [--port 11434]
python3 benchmark_ollama.py --models glm-5.1,glm-5.2,kimi-k2.6 --runs 3
python3 benchmark_ollama.py --list
"""
import argparse
import json
import time
import sys
import urllib.request
import urllib.error
DEFAULT_PROMPT = "Write a detailed 500-word essay about the history of computing, from Babbage to modern AI. Include key milestones, people, and technologies."
DEFAULT_HOST = "nvidia.hell"
DEFAULT_PORT = 11434
DEFAULT_NUM_PREDICT = 1000
def api_get(host, port, path):
url = f"http://{host}:{port}{path}"
req = urllib.request.Request(url, method="GET")
with urllib.request.urlopen(req, timeout=10) as resp:
return json.loads(resp.read())
def api_post_stream(host, port, path, payload):
"""POST with streaming response. Yields (line_dict, elapsed_since_start)."""
url = f"http://{host}:{port}{path}"
data = json.dumps(payload).encode("utf-8")
req = urllib.request.Request(
url, data=data, method="POST",
headers={"Content-Type": "application/json"},
)
t0 = time.monotonic()
resp = urllib.request.urlopen(req, timeout=300)
buffer = b""
while True:
chunk = resp.read(1)
if not chunk:
break
buffer += chunk
if chunk == b"\n":
line = buffer.strip()
buffer = b""
if not line:
continue
try:
obj = json.loads(line)
yield obj, time.monotonic() - t0
except json.JSONDecodeError:
continue
def list_models(host, port):
try:
data = api_get(host, port, "/api/tags")
return [m["name"] for m in data.get("models", [])]
except Exception as e:
print(f"Error listing models: {e}", file=sys.stderr)
return []
def benchmark_model(host, port, model, prompt, num_predict):
"""Run a single benchmark against one model."""
payload = {
"model": model,
"prompt": prompt,
"stream": True,
"options": {"temperature": 0.0, "num_predict": num_predict},
}
# Track timing for thinking and response phases
first_thinking_time = None
last_thinking_time = None
first_response_time = None
last_response_time = None
thinking_chars = 0
response_chars = 0
# Stats from final chunk
prompt_eval_count = 0
eval_count = 0
total_duration_ns = 0
done_reason = ""
try:
for obj, elapsed in api_post_stream(host, port, "/api/generate", payload):
if obj.get("error"):
return {"model": model, "error": obj["error"]}
thinking = obj.get("thinking", "") or ""
response = obj.get("response", "") or ""
if thinking:
if first_thinking_time is None:
first_thinking_time = elapsed
last_thinking_time = elapsed
thinking_chars += len(thinking)
if response:
if first_response_time is None:
first_response_time = elapsed
last_response_time = elapsed
response_chars += len(response)
if obj.get("done"):
prompt_eval_count = obj.get("prompt_eval_count", 0)
eval_count = obj.get("eval_count", 0)
total_duration_ns = obj.get("total_duration", 0)
done_reason = obj.get("done_reason", "")
break
except urllib.error.HTTPError as e:
body = e.read().decode("utf-8", errors="replace")[:200]
return {"model": model, "error": f"HTTP {e.code}: {body}"}
except Exception as e:
return {"model": model, "error": str(e)}
# Calculate metrics
# Estimate token split: thinking vs response by char ratio
total_gen_chars = thinking_chars + response_chars
if total_gen_chars > 0:
thinking_tokens_est = round(eval_count * thinking_chars / total_gen_chars)
response_tokens_est = round(eval_count * response_chars / total_gen_chars)
else:
thinking_tokens_est = 0
response_tokens_est = 0
# Phase durations
thinking_duration = 0
if first_thinking_time and last_thinking_time:
thinking_duration = last_thinking_time - first_thinking_time
response_duration = 0
if first_response_time and last_response_time:
response_duration = last_response_time - first_response_time
# Total generation time (from first token to last token, whether thinking or response)
first_token_time = None
last_token_time = None
if first_thinking_time is not None or first_response_time is not None:
first_token_time = min(
t for t in [first_thinking_time, first_response_time] if t is not None
)
if last_thinking_time is not None or last_response_time is not None:
last_token_time = max(
t for t in [last_thinking_time, last_response_time] if t is not None
)
total_gen_time = 0
if first_token_time and last_token_time:
total_gen_time = last_token_time - first_token_time
# Throughput calculations
# Overall: eval_count / total_gen_time (all tokens including thinking)
overall_tps = (eval_count / total_gen_time) if total_gen_time > 0 else 0
# Thinking phase throughput
thinking_tps = (thinking_tokens_est / thinking_duration) if thinking_duration > 0 else 0
# Response phase throughput
response_tps = (response_tokens_est / response_duration) if response_duration > 0 else 0
# Wall-clock total (from request start to last token)
wall_total = last_token_time or 0
# TTFT metrics
ttft_thinking = first_thinking_time
ttft_response = first_response_time
return {
"model": model,
"done_reason": done_reason,
"ttft_thinking_s": round(ttft_thinking, 3) if ttft_thinking else None,
"ttft_response_s": round(ttft_response, 3) if ttft_response else None,
"wall_total_s": round(wall_total, 3),
"prompt_tokens": prompt_eval_count,
"eval_count": eval_count,
"thinking_tokens_est": thinking_tokens_est,
"response_tokens_est": response_tokens_est,
"thinking_duration_s": round(thinking_duration, 3),
"response_duration_s": round(response_duration, 3),
"thinking_tps": round(thinking_tps, 1),
"response_tps": round(response_tps, 1),
"overall_tps": round(overall_tps, 1),
"thinking_chars": thinking_chars,
"response_chars": response_chars,
}
def main():
parser = argparse.ArgumentParser(description="Benchmark Ollama models (thinking + response)")
parser.add_argument("--host", default=DEFAULT_HOST)
parser.add_argument("--port", type=int, default=DEFAULT_PORT)
parser.add_argument("--prompt", default=DEFAULT_PROMPT)
parser.add_argument("--models", help="Comma-separated model list (default: all cloud)")
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
parser.add_argument("--runs", type=int, default=1)
parser.add_argument("--list", action="store_true")
args = parser.parse_args()
print(f"Connecting to http://{args.host}:{args.port} ...")
models = list_models(args.host, args.port)
if not models:
print("No models found or connection failed.", file=sys.stderr)
sys.exit(1)
# Filter to cloud models by default for relevance
cloud_models = [m for m in models if ":cloud" in m]
all_models = models
if args.list:
print(f"Available models ({len(models)}):")
for m in all_models:
print(f" - {m}")
return
# Select models
if args.models:
wanted = [m.strip() for m in args.models.split(",")]
selected = []
for w in wanted:
matches = [m for m in all_models if w.lower() in m.lower()]
if matches:
selected.extend(matches)
else:
print(f" Warning: '{w}' not found", file=sys.stderr)
seen = set()
selected = [m for m in selected if not (m in seen or seen.add(m))]
else:
selected = cloud_models
if not selected:
print("No matching models selected.", file=sys.stderr)
sys.exit(1)
print(f"\nBenchmarking {len(selected)} model(s), {args.runs} run(s) each, max {args.num_predict} tokens")
print(f"Prompt: \"{args.prompt[:80]}...\"")
print(f"{'='*120}")
results = []
for model in selected:
for run in range(args.runs):
run_label = f"run {run+1}/{args.runs}" if args.runs > 1 else ""
print(f"\n[{model}] {run_label}")
r = benchmark_model(args.host, args.port, model, args.prompt, args.num_predict)
results.append(r)
if "error" in r:
print(f" ERROR: {r['error']}")
continue
print(f" Done reason: {r['done_reason']}")
tt = f"{r['ttft_thinking_s']:.3f}s" if r['ttft_thinking_s'] else "N/A"
tr = f"{r['ttft_response_s']:.3f}s" if r['ttft_response_s'] else "N/A"
print(f" TTFT thinking: {tt}")
print(f" TTFT response: {tr}")
print(f" Wall total: {r['wall_total_s']:.3f}s")
print(f" Tokens: prompt={r['prompt_tokens']} total_gen={r['eval_count']} (think~{r['thinking_tokens_est']} resp~{r['response_tokens_est']})")
print(f" Think phase: {r['thinking_duration_s']:.3f}s @ {r['thinking_tps']:.1f} tok/s")
print(f" Resp phase: {r['response_duration_s']:.3f}s @ {r['response_tps']:.1f} tok/s")
print(f" Overall tps: {r['overall_tps']:.1f} tok/s ({r['eval_count']} tok in {r['wall_total_s']:.1f}s)")
# Summary table
print(f"\n{'='*120}")
print("SUMMARY")
print(f"{'Model':<22} {'Reason':>8} {'TTFT-t':>7} {'TTFT-r':>7} {'Wall s':>7} {'Tokens':>7} {'Think t':>7} {'Think tps':>9} {'Resp t':>7} {'Resp tps':>9} {'Overall':>8}")
print("-" * 120)
for r in results:
if "error" in r:
print(f"{r['model']:<22} ERROR: {r['error'][:50]}")
continue
tt = f"{r['ttft_thinking_s']:.3f}" if r['ttft_thinking_s'] else " N/A"
tr = f"{r['ttft_response_s']:.3f}" if r['ttft_response_s'] else " N/A"
print(f"{r['model']:<22} {r['done_reason']:>8} {tt:>7} {tr:>7} {r['wall_total_s']:>7.3f} {r['eval_count']:>7} {r['thinking_duration_s']:>7.3f} {r['thinking_tps']:>9.1f} {r['response_duration_s']:>7.3f} {r['response_tps']:>9.1f} {r['overall_tps']:>8.1f}")
if __name__ == "__main__":
main()

71
scripts/remind_add.py Normal file
View File

@@ -0,0 +1,71 @@
#!/usr/bin/env python3
"""One-off helper to add a reminder directly to the SQLite DB."""
import sqlite3
import sys
from datetime import datetime, timezone
from pathlib import Path
WORKSPACE = Path(__file__).resolve().parent.parent
DB_PATH = WORKSPACE / "db" / "reminders.sqlite"
def main() -> int:
if len(sys.argv) < 3:
print("usage: remind_add.py <at_iso> <text>", file=sys.stderr)
return 1
at_str = sys.argv[1]
text = sys.argv[2]
try:
datetime.fromisoformat(at_str)
except ValueError as exc:
print(f"invalid --at datetime: {exc}", file=sys.stderr)
return 1
DB_PATH.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(DB_PATH)
conn.row_factory = sqlite3.Row
conn.executescript(
"""
PRAGMA journal_mode = WAL;
PRAGMA foreign_keys = ON;
CREATE TABLE IF NOT EXISTS reminders (
id INTEGER PRIMARY KEY AUTOINCREMENT,
text TEXT NOT NULL CHECK(text <> ''),
enabled INTEGER NOT NULL DEFAULT 1 CHECK(enabled IN (0, 1)),
timezone TEXT NOT NULL DEFAULT 'Europe/Prague',
created_at TEXT NOT NULL,
updated_at TEXT NOT NULL,
deleted_at TEXT
);
CREATE TABLE IF NOT EXISTS schedule_at (
id INTEGER PRIMARY KEY AUTOINCREMENT,
reminder_id INTEGER NOT NULL REFERENCES reminders(id) ON DELETE CASCADE,
at_datetime TEXT NOT NULL
);
CREATE INDEX IF NOT EXISTS idx_reminders_text ON reminders(text);
CREATE INDEX IF NOT EXISTS idx_at_datetime ON schedule_at(reminder_id, at_datetime);
"""
)
now = datetime.now(timezone.utc).isoformat(timespec="seconds")
cur = conn.execute(
"INSERT INTO reminders (text, enabled, timezone, created_at, updated_at) VALUES (?, 1, 'Europe/Prague', ?, ?)",
(text, now, now),
)
reminder_id = cur.lastrowid
conn.execute(
"INSERT INTO schedule_at (reminder_id, at_datetime) VALUES (?, ?)",
(reminder_id, at_str),
)
conn.commit()
conn.close()
print(f"added reminder #{reminder_id} at {at_str}: {text}")
return 0
if __name__ == "__main__":
sys.exit(main())