nanobot: 2026-09-13 20:05:41 - toolcall test vysledky 18 modelu + skript
This commit is contained in:
373
projects/ai/artifacts/ollama-toolcall-test.py
Normal file
373
projects/ai/artifacts/ollama-toolcall-test.py
Normal file
@@ -0,0 +1,373 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Functional tool-calling test suite for local Ollama models (<10 GB, tools capability).
|
||||
|
||||
10 scenarios per model, increasing difficulty, English-only prompts.
|
||||
|
||||
S1 direct call explicit "use the tool", simple city arg
|
||||
S2 implied call no explicit tool instruction, model must infer
|
||||
S3 tool selection 3 tools available, must pick get_weather
|
||||
S4 context resolution city referenced indirectly ("there") from context
|
||||
S5 multiple parameters city + unit, unit must be extracted from request
|
||||
S6 no-tool restraint answerable without tools -> must NOT call
|
||||
S7 multi-call two cities compared -> both calls expected
|
||||
S8 result reasoning use result to answer a yes/no derived question
|
||||
S9 argument fidelity city with diacritics must be preserved
|
||||
S10 distractor selection 5 tools, must pick the non-obvious get_stock_price
|
||||
|
||||
Per scenario: PHASE 1 (emit valid tool call) + PHASE 2 (use the returned result).
|
||||
Verdicts: PASS = 1, PARTIAL = 0.5, FAIL = 0. Prints each scenario result
|
||||
immediately (unbuffered). Usage:
|
||||
|
||||
PYTHONUNBUFFERED=1 uv run ollama-toolcall-test.py <model-name>
|
||||
"""
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
import unicodedata
|
||||
import urllib.request
|
||||
|
||||
BASE = "http://nvidia.hell:11434"
|
||||
TIMEOUT = 120 # per request; abort the model run if exceeded
|
||||
|
||||
MODELS = [
|
||||
"lfm2.5-thinking", # 0.7
|
||||
"deepseek-v3", # 2.0
|
||||
"llama3.2", # 2.0
|
||||
"granite4", # 2.1
|
||||
"phi4-mini", # 2.5
|
||||
"nemotron-mini", # 2.7
|
||||
"phi4-mini-reasoning", # 3.2
|
||||
"mistral", # 4.1
|
||||
"granite4:tiny-h", # 4.2
|
||||
"granite3.3", # 4.9
|
||||
"aya-expanse", # 5.1
|
||||
"qwen3", # 5.2
|
||||
"granite4.1:8b", # 5.3
|
||||
"gemma4-uncensored", # 5.3
|
||||
"ornith", # 5.6
|
||||
"qwen3.5-uncensored", # 5.6
|
||||
"ministral-3", # 6.0
|
||||
"qwen3.5:9b", # 6.6
|
||||
"qwen3.5", # 6.6
|
||||
"mistral-nemo", # 7.1
|
||||
]
|
||||
|
||||
|
||||
def strip_accents(s):
|
||||
return "".join(c for c in unicodedata.normalize("NFD", s) if unicodedata.category(c) != "Mn")
|
||||
|
||||
|
||||
def trunc(s, n=200):
|
||||
s = (s or "").replace("\n", " ").strip()
|
||||
return s[:n] + ("…" if len(s) > n else "")
|
||||
|
||||
|
||||
LAST_TIMING = {}
|
||||
|
||||
|
||||
def chat(payload, timeout=TIMEOUT):
|
||||
req = urllib.request.Request(
|
||||
f"{BASE}/api/chat",
|
||||
data=json.dumps(payload).encode(),
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r:
|
||||
return json.load(r)
|
||||
|
||||
|
||||
def fdef(name, description, properties, required):
|
||||
return {"type": "function", "function": {
|
||||
"name": name, "description": description,
|
||||
"parameters": {"type": "object", "properties": properties, "required": required},
|
||||
}}
|
||||
|
||||
|
||||
W_CITY = {"city": {"type": "string", "description": "City name"}}
|
||||
|
||||
|
||||
def weather_tool(unit=False):
|
||||
props = dict(W_CITY)
|
||||
if unit:
|
||||
props["unit"] = {"type": "string", "enum": ["celsius", "fahrenheit"],
|
||||
"description": "Temperature unit for the result"}
|
||||
return fdef("get_weather", "Get the current weather for a city", props, ["city"])
|
||||
|
||||
|
||||
def get_time_tool():
|
||||
return fdef("get_time", "Get the current local time in a timezone",
|
||||
{"timezone": {"type": "string", "description": "IANA timezone, e.g. Europe/Prague"}}, ["timezone"])
|
||||
|
||||
|
||||
def get_currency_tool():
|
||||
return fdef("get_currency_rate", "Get the current exchange rate between two currencies",
|
||||
{"from": {"type": "string"}, "to": {"type": "string"}}, ["from", "to"])
|
||||
|
||||
|
||||
def get_stock_tool():
|
||||
return fdef("get_stock_price", "Get the current stock price for a company",
|
||||
{"symbol": {"type": "string", "description": "Stock ticker symbol, e.g. GOOG"}}, ["symbol"])
|
||||
|
||||
|
||||
def send_email_tool():
|
||||
return fdef("send_email", "Send an email to someone",
|
||||
{"to": {"type": "string", "description": "Recipient email"},
|
||||
"subject": {"type": "string"}, "body": {"type": "string"}}, ["to"])
|
||||
|
||||
|
||||
def translate_tool():
|
||||
return fdef("translate_text", "Translate a text to another language",
|
||||
{"text": {"type": "string"}, "target_language": {"type": "string"}}, ["text"])
|
||||
|
||||
|
||||
def tool_result(city, **kw):
|
||||
d = {"city": city}
|
||||
d.update(kw)
|
||||
return json.dumps(d)
|
||||
|
||||
|
||||
SCENARIOS = [
|
||||
{ # 1 easy: explicit instruction, trivial extraction
|
||||
"id": "S1", "name": "direct call",
|
||||
"tools": [weather_tool()],
|
||||
"messages": [{"role": "user", "content": "What is the current weather in Brno right now? Use the tool."}],
|
||||
"expect_tool": "get_weather", "expect_args": {"city": "brno"},
|
||||
"phase2": True,
|
||||
"result": tool_result("Brno", temperature_c=18, condition="partly cloudy", wind_kmh=7),
|
||||
"answer_any": ["18"],
|
||||
},
|
||||
{ # 2: no "use the tool" hint at all
|
||||
"id": "S2", "name": "implied call",
|
||||
"tools": [weather_tool()],
|
||||
"messages": [{"role": "user", "content": "I'm about to head out. Can you check what the weather's like in Prague?"}],
|
||||
"expect_tool": "get_weather", "expect_args": {"city": "prague"},
|
||||
"phase2": True,
|
||||
"result": tool_result("Prague", temperature_c=14, condition="overcast", wind_kmh=10),
|
||||
"answer_any": ["14"],
|
||||
},
|
||||
{ # 3: pick the right tool out of 3
|
||||
"id": "S3", "name": "tool selection",
|
||||
"tools": [weather_tool(), get_time_tool(), get_currency_tool()],
|
||||
"messages": [{"role": "user", "content": "Should I take an umbrella when going to work in Ostrava today?"}],
|
||||
"expect_tool": "get_weather", "expect_args": {"city": "ostrava"},
|
||||
"phase2": True,
|
||||
"result": tool_result("Ostrava", temperature_c=12, condition="rain", wind_kmh=20, precipitation_prob="85%"),
|
||||
"answer_any": ["umbrella", "rain"],
|
||||
},
|
||||
{ # 4: city only implied by conversation context
|
||||
"id": "S4", "name": "context resolution",
|
||||
"tools": [weather_tool()],
|
||||
"messages": [
|
||||
{"role": "user", "content": "My sister lives in Pardubice."},
|
||||
{"role": "assistant", "content": "Nice! Pardubice is a lovely city. Is there anything I can help you with?"},
|
||||
{"role": "user", "content": "What's the weather like there right now?"},
|
||||
],
|
||||
"expect_tool": "get_weather", "expect_args": {"city": "pardubice"},
|
||||
"phase2": True,
|
||||
"result": tool_result("Pardubice", temperature_c=21, condition="clear", wind_kmh=5),
|
||||
"answer_any": ["21"],
|
||||
},
|
||||
{ # 5: two parameters, second one from wording
|
||||
"id": "S5", "name": "multiple parameters",
|
||||
"tools": [weather_tool(unit=True)],
|
||||
"messages": [{"role": "user", "content": "What's the weather in Vienna? I'd like it in Fahrenheit."}],
|
||||
"expect_tool": "get_weather", "expect_args": {"city": "vienna", "unit": "fahr"},
|
||||
"phase2": True,
|
||||
"result": json.dumps({"city": "Vienna", "temperature_f": 64, "condition": "sunny", "humidity": "35%"}),
|
||||
"answer_any": ["64"],
|
||||
},
|
||||
{ # 6: must NOT call any tool
|
||||
"id": "S6", "name": "no-tool restraint",
|
||||
"tools": [weather_tool()],
|
||||
"messages": [{"role": "user", "content": "What is the capital of Germany?"}],
|
||||
"no_call": True, "phase2": False,
|
||||
"answer_any": ["berlin"],
|
||||
},
|
||||
{ # 7: two tool calls in one answer
|
||||
"id": "S7", "name": "multi-call",
|
||||
"tools": [weather_tool()],
|
||||
"messages": [{"role": "user", "content": "Compare the current weather in Brno and Olomouc."}],
|
||||
"multi_call": ["brno", "olomouc"],
|
||||
"phase2": False,
|
||||
},
|
||||
{ # 8: answer a derived yes/no question from the result
|
||||
"id": "S8", "name": "result reasoning",
|
||||
"tools": [weather_tool()],
|
||||
"messages": [{"role": "user", "content": "Is it too hot for a run in Plzen right now?"}],
|
||||
"expect_tool": "get_weather", "expect_args": {"city": "plzen"},
|
||||
"phase2": True,
|
||||
"result": tool_result("Plzen", temperature_c=31, condition="sunny", humidity="40%"),
|
||||
"answer_any": ["31"],
|
||||
},
|
||||
{ # 9: diacritics must survive argument extraction
|
||||
"id": "S9", "name": "argument fidelity",
|
||||
"tools": [weather_tool()],
|
||||
"messages": [{"role": "user", "content": "What's the current weather in České Budějovice? Use the tool."}],
|
||||
"expect_tool": "get_weather", "expect_args": {"city": "budejovice"},
|
||||
"phase2": True,
|
||||
"result": tool_result("Ceske Budejovice", temperature_c=23, condition="cloudy", wind_kmh=8),
|
||||
"answer_any": ["23"],
|
||||
},
|
||||
{ # 10 hardest: non-obvious tool among 5, arg is a ticker not a city
|
||||
"id": "S10", "name": "distractor selection",
|
||||
"tools": [weather_tool(), get_time_tool(), get_stock_tool(), send_email_tool(), translate_tool()],
|
||||
"messages": [{"role": "user", "content": "How are Google shares doing today?"}],
|
||||
"expect_tool": "get_stock_price", "expect_args": {"symbol": "goog"},
|
||||
"phase2": True,
|
||||
"result": json.dumps({"symbol": "GOOG", "price_usd": 172.5, "currency": "USD", "change_pct": 1.2}),
|
||||
"answer_any": ["172", "goog"],
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def parse_args(tc):
|
||||
"""Return (name, args_dict) from a tool_call entry, tolerating str/dict args."""
|
||||
fn = tc.get("function", {})
|
||||
name = fn.get("name", "")
|
||||
raw = fn.get("arguments", {})
|
||||
if raw is None:
|
||||
raw = {}
|
||||
if isinstance(raw, str):
|
||||
try:
|
||||
raw = json.loads(raw)
|
||||
except json.JSONDecodeError:
|
||||
raw = None
|
||||
return name, (raw if isinstance(raw, dict) else None)
|
||||
|
||||
|
||||
def args_ok(args, expected):
|
||||
if not isinstance(args, dict):
|
||||
return False, "arguments not a valid object"
|
||||
for key, substr in expected.items():
|
||||
val = str(args.get(key, ""))
|
||||
if substr.lower() not in val.lower() or (substr.lower() not in strip_accents(val.lower()) and False):
|
||||
if substr.lower() not in strip_accents(val.lower()):
|
||||
return False, f"{key}={args.get(key)!r} (want substring {substr!r})"
|
||||
return True, ""
|
||||
|
||||
|
||||
def answer_ok(answer, needles):
|
||||
low = strip_accents(answer.lower())
|
||||
return any(n.lower() in low for n in needles)
|
||||
|
||||
|
||||
def run_scenario(model, sc):
|
||||
global LAST_TIMING
|
||||
msgs = list(sc["messages"])
|
||||
try:
|
||||
r1 = chat({"model": model, "messages": msgs, "tools": sc["tools"], "stream": False})
|
||||
except Exception as e:
|
||||
print(f"PHASE1 ERROR: {e}")
|
||||
return "ERROR"
|
||||
LAST_TIMING = {
|
||||
"load_ms": r1.get("load_duration") or 0,
|
||||
"prompt_ms": r1.get("prompt_eval_duration") or 0,
|
||||
"eval_ms": r1.get("eval_duration") or 0,
|
||||
}
|
||||
m1 = r1.get("message", {})
|
||||
calls = m1.get("tool_calls") or []
|
||||
|
||||
# --- no-call scenario ---
|
||||
if sc.get("no_call"):
|
||||
if calls:
|
||||
print(f" called {calls[0].get('function', {}).get('name', '?')} despite no tool needed -> FAIL")
|
||||
return "FAIL"
|
||||
content = m1.get("content", "") or ""
|
||||
ok = answer_ok(content, sc["answer_any"])
|
||||
print(f" no tool call: ok; answer: {trunc(content, 150)}")
|
||||
return "PASS" if ok else "FAIL"
|
||||
|
||||
# --- multi-call scenario ---
|
||||
if sc.get("multi_call"):
|
||||
found = set()
|
||||
names_ok = True
|
||||
for tc in calls:
|
||||
name, args = parse_args(tc)
|
||||
if name != "get_weather":
|
||||
names_ok = False
|
||||
if isinstance(args, dict):
|
||||
found.add(strip_accents(str(args.get("city", "")).lower()))
|
||||
want = set(sc["multi_call"])
|
||||
hit = want & found
|
||||
verdict = "FAIL" if not calls else ("PASS" if hit == want else "PARTIAL")
|
||||
print(f" calls: {len(calls)}; cities: {sorted(found) or 'none'}; names_ok={names_ok}")
|
||||
return verdict
|
||||
|
||||
# --- standard call scenario ---
|
||||
if not calls:
|
||||
print(f" no tool_calls; content: {trunc(m1.get('content', ''), 180)}")
|
||||
return "FAIL"
|
||||
name, args = parse_args(calls[0])
|
||||
print(f" call: {name} {json.dumps(args, ensure_ascii=False) if args else calls[0]}")
|
||||
p1 = "PASS"
|
||||
if name != sc["expect_tool"]:
|
||||
p1 = "PARTIAL"
|
||||
else:
|
||||
ok, why = args_ok(args, sc.get("expect_args", {}))
|
||||
if not ok:
|
||||
p1 = "PARTIAL"
|
||||
print(f" arg mismatch: {why}")
|
||||
print(f" PHASE1 -> {p1}")
|
||||
|
||||
if not sc.get("phase2"):
|
||||
return p1
|
||||
|
||||
# --- phase 2: feed result back ---
|
||||
msgs.append(m1)
|
||||
msgs.append({"role": "tool", "name": sc["expect_tool"], "content": sc["result"]})
|
||||
try:
|
||||
r2 = chat({"model": model, "messages": msgs, "tools": sc["tools"], "stream": False})
|
||||
except Exception as e:
|
||||
print(f"PHASE2 ERROR: {e}")
|
||||
return "ERROR"
|
||||
answer = (r2.get("message", {}) or {}).get("content", "") or ""
|
||||
p2ok = answer_ok(answer, sc["answer_any"])
|
||||
print(f" answer: {trunc(answer, 200)}")
|
||||
print(f" PHASE2 -> {'PASS' if p2ok else 'FAIL'}")
|
||||
|
||||
if p1 == "PASS":
|
||||
return "PASS" if p2ok else "PARTIAL"
|
||||
return p1 if p1 == "FAIL" else "PARTIAL"
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print(f"usage: {sys.argv[0]} <model-name>\nmodels (smallest first): {', '.join(MODELS)}")
|
||||
sys.exit(2)
|
||||
model = sys.argv[1]
|
||||
if model not in MODELS:
|
||||
print(f"unknown model {model!r}; known: {', '.join(MODELS)}")
|
||||
sys.exit(2)
|
||||
|
||||
print(f"MODEL: {model} — {len(SCENARIOS)} scenarios, English prompts\n" + "=" * 70)
|
||||
score, verdicts, t0 = 0.0, {}, time.monotonic()
|
||||
aborted = False
|
||||
for sc in SCENARIOS:
|
||||
print(f"\n{sc['id']} {sc['name']}")
|
||||
try:
|
||||
v = run_scenario(model, sc)
|
||||
except KeyboardInterrupt:
|
||||
v = "ERROR"
|
||||
if v == "ERROR":
|
||||
print(" -> aborting remaining scenarios for this model (no response in reasonable time)")
|
||||
aborted = True
|
||||
verdicts[sc["id"]] = v
|
||||
score += {"PASS": 1.0, "PARTIAL": 0.5}.get(v, 0.0)
|
||||
if v != "ERROR":
|
||||
print(f" => {v}")
|
||||
if aborted:
|
||||
break
|
||||
|
||||
dt = time.monotonic() - t0
|
||||
print(f"\n{'=' * 70}\nSUMMARY {model}")
|
||||
line = ", ".join(f"{sid}:{v}" for sid, v in verdicts.items())
|
||||
print(line)
|
||||
n = sum(1 for v in verdicts.values() if v != "ERROR")
|
||||
print(f"SCORE: {score:.1f}/{len(SCENARIOS)} ({n}/{len(SCENARIOS)} scenarios answered)")
|
||||
t = LAST_TIMING
|
||||
if t:
|
||||
load = t.get("load_ms", 0) / 1e9
|
||||
print(f"timing of last request: load {load:.1f}s, prompt {t.get('prompt_ms', 0)/1e6:.0f}ms, eval {t.get('eval_ms', 0)/1e6:.0f}ms (load only on first run of a model)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -63,3 +63,27 @@ https://www.librechat.ai/docs/local/docker
|
||||
**Zjištěné výsledky (2026-09-12):** 57 modelů na instanci; lokálních s `tools` pod 10 GB cca 24 (nejmenší `lfm2.5-thinking` 0.7 GB, největší `gemma4`/`gemma4:e4b` 9.6 GB). `:cloud` modely mají size 0 GB (pointer na Ollama Cloud, neběží lokálně — do kritéria velikosti na disku nepatří). `gemini-3-flash-preview:cloud` vrací `POST /api/show` HTTP 410 Gone — mrtvý pointer, Ollama Cloud model stáhla.
|
||||
|
||||
**Otevřené:** funkční test tool callingu (reálný `tools` request na každý kandidát a ověření validního `tool_calls` v odpovědi) ještě neproveden — nabídka visí.
|
||||
- 2026-09-13: Funkční test tool callingu lokálních Ollama modelů na `nvidia.hell` dokončen (18 modelů, 10 scénářů, anglické prompty).
|
||||
|
||||
**Skript:** `projects/ai/artifacts/ollama-toolcall-test.py` — 10 scénářů rostoucí obtížnosti (S1 direct call, S2 implied call, S3 tool selection ze 3, S4 context resolution, S5 více parametrů, S6 no-tool restraint — NESMÍ volat, S7 multi-call dvě volání, S8 result reasoning, S9 diakritika v argumentu, S10 distractor selection z 5 nástrojů). Každý scénář 2 fáze: vyslat správný tool call + zpracovat vrácený výsledek. Spouští se per model: `PYTHONUNBUFFERED=1 uv run ollama-toolcall-test.py <model>`. Přísný parser (odmítl vnořené arguments u nemotronu — reálný klient unwrap nedělá).
|
||||
|
||||
**Čisté 10/10:** granite4.1:8b (5.3 GB, 26s, nejlepší balance), lfm2.5-thinking (0.7 GB, 27s — pozor, 1B model, slabá sémantika, ale mechanika solidní), ministral-3 (6.0 GB, 27.5s, nejrychlejší eval ~332ms), ornith (5.6 GB, 46s), qwen3.5:9b (6.6 GB, 73s), qwen3 (5.2 GB, 89s — pomalý, thinking model).
|
||||
|
||||
**Selhání podle vzorců:**
|
||||
- Over-eager calling (S6 fail — volá i když nemá): deepseek-v3, llama3.2 (jinak 9/10).
|
||||
- S7 multi-call bug (jen jedno volání z dvou): granite4 (9.5/10), granite4:tiny-h (8/10).
|
||||
- Prázdné odpovědi/args: granite4:tiny-h (S3), granite3.3 (S7.5/10 — S6 odpověděl prázdný text, S3 prázdné arguments).
|
||||
- Iniciativa (umí volat jen na explicitní pokyn, jinak čeká/odpovídá textem): mistral-nemo (7/10 — S4/S8 „Shall I proceed?", S8 vrátil giphy GIF link), mistral (5/10 — při implied call vypíše JSON volání jako text a **halucinuje výsledek**, S6 odpověděl voláním get_weather s city=Berlin).
|
||||
- Template mismatch (tool call jako plaintext, syrové speciální tokeny): phi4-mini (1.5/10), phi4-mini-reasoning (netestovatelný — reasoning CoT tolik zpomalí, že ~3 min bez odpovědi).
|
||||
- Vnořené arguments (`{"type":..., "arguments":{...}}` místo přímých parametrů — sémanticky správně, formálně rozbité pro standardního klienta): nemotron-mini (5.5/10), dřív i aya-expanse.
|
||||
- Sejmuly server (memory watchdog na nvidia.hell zabil llama-server — `Remote end closed connection`): aya-expanse (5.1 GB, hned na S1), qwen3.5 (6.6 GB, na S10 s 5 nástroji; do S9 byl 9/9 PASS, qwen3.5:9b S10 ale zvládl — rozdíl mezi tagy nebo náhoda v načítání).
|
||||
- Vynechané dle zadání uživatele: gemma4-uncensored, qwen3.5-uncensored; gemma4/gemma4:e4b (9.6 GB) dřív sejmul server, scope ukončen u mistral-nemo (7.1 GB).
|
||||
|
||||
**Poznatky pro budoucí testy:**
|
||||
1. Jeden prompt nestačí — s českým promptem lfm2.5-thinking failoval na S1 (city=„Berlin"), s anglickým 10/10. Malé modely rozhoduje jazyk promptu, ne schopnost.
|
||||
2. S6 no-tool restraint je filtr, který vypadlo víc modelů než kterákoli jiná kategorie.
|
||||
3. Více nástrojů v promptu = vyšší paměťové nároky = watchdog kill. Testovat vícenástrojové scénáře jen u finálních kandidátů.
|
||||
4. Rodinné vzorce: granite = disciplína (S6 umí), qwen = spolehlivý tool calling, phi4 = rozbitý, mistral rodina = problém s iniciativou.
|
||||
5. Čas v summary skriptu je wall-clock celého běhu včetně prvního načtení modelu; `load_duration` z API se vypisuje zvlášť (load>0 jen při prvním requestu).
|
||||
|
||||
**Doporučení pro agenta / tool use:** granite4.1:8b (nejlepší balance), ministral-3 (nejrychlejší), qwen3 (spolehlivost, pomalejší).
|
||||
|
||||
Reference in New Issue
Block a user