Files
llm-router/benchmarks/agent_bench.py
T
Joseph CostaandClaude Opus 4.8 9938d46a67 Initial commit: llm-router — smart OpenAI/Anthropic/Ollama router
A stdlib pre-router in front of Ollama with LiteLLM backend:
- auto model selection by content/tools/modality, with fallbacks
- OpenAI /v1, Anthropic /v1/messages, and Ollama-native /api/* endpoints
- Whisper-shaped /v1/audio/transcriptions + in-chat audio
- key-based fleet policies (e.g. force a client onto uncensored models)
- optional Bearer auth; launchd/systemd service install
- benchmark harnesses (speed, quality, agentic tool use) with sample results

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-05 02:05:16 -05:00

160 lines
7.0 KiB
Python

#!/usr/bin/env python3
"""
Agentic tool-use benchmark for the local fleet, via Ollama's native tools API.
Scores the real agent failure modes (each 0/1, objective):
A invoke + args : calls calculator for a math Q, expression evaluates correctly
B tool selection : picks get_weather (not calc/search) for a weather Q, right city
E multi-arg : get_weather with correct city AND unit
C use-result : calls tool, we return a value, final answer uses it
D abstain : does NOT call a tool for a plain greeting
overall = mean of the 6 checks. think disabled for a fast, direct baseline.
Run after models are downloaded: ./.venv/bin/python agent_bench.py
Writes agent_report.md and agent_results.json.
"""
import json, re, datetime, urllib.request
import benchmark as b
OLLAMA = "http://127.0.0.1:11434"
CALC = {"type": "function", "function": {
"name": "calculator",
"description": "Evaluate an arithmetic expression and return the number.",
"parameters": {"type": "object",
"properties": {"expression": {"type": "string",
"description": "e.g. '3*4+1'"}},
"required": ["expression"]}}}
WEATHER = {"type": "function", "function": {
"name": "get_weather",
"description": "Get the current weather for a city.",
"parameters": {"type": "object",
"properties": {"city": {"type": "string"},
"unit": {"type": "string",
"enum": ["celsius", "fahrenheit"]}},
"required": ["city", "unit"]}}}
SEARCH = {"type": "function", "function": {
"name": "search_kb",
"description": "Search the internal company knowledge base for a query.",
"parameters": {"type": "object",
"properties": {"query": {"type": "string"}},
"required": ["query"]}}}
def chat_tools(model, messages, tools, num_predict=512, think=False):
body = {"model": model, "messages": messages, "tools": tools, "stream": False,
"keep_alive": "30m", "options": {"temperature": 0, "num_predict": num_predict}}
if think is not None:
body["think"] = think
req = urllib.request.Request(OLLAMA + "/api/chat", data=json.dumps(body).encode(),
headers={"content-type": "application/json"})
try:
with urllib.request.urlopen(req, timeout=600) as r:
return json.loads(r.read()).get("message", {}) or {}
except Exception:
# some models reject think param with tools; retry without
body.pop("think", None)
req = urllib.request.Request(OLLAMA + "/api/chat", data=json.dumps(body).encode(),
headers={"content-type": "application/json"})
with urllib.request.urlopen(req, timeout=600) as r:
return json.loads(r.read()).get("message", {}) or {}
def first_call(msg):
tcs = msg.get("tool_calls") or []
if not tcs:
return None, {}
fn = tcs[0].get("function", {}) or {}
args = fn.get("arguments")
if isinstance(args, str):
try:
args = json.loads(args)
except Exception:
args = {}
return fn.get("name"), (args or {})
def safe_eval(expr):
if not re.fullmatch(r"[0-9+\-*/(). ]+", expr or ""):
raise ValueError("unsafe")
return eval(expr, {"__builtins__": {}}, {})
def run_agent(model):
c = {}
# A: invoke calculator + correct expression
name, args = first_call(chat_tools(model, [b.user("What is 3847 * 293? Use the calculator tool.")], [CALC]))
c["invoke"] = 1 if name == "calculator" else 0
try:
c["args"] = 1 if c["invoke"] and abs(safe_eval(str(args.get("expression", ""))) - 3847 * 293) < 1e-6 else 0
except Exception:
c["args"] = 0
# B: tool selection among three + right city
name, args = first_call(chat_tools(model, [b.user("What's the weather in Tokyo, in celsius?")],
[CALC, WEATHER, SEARCH]))
c["select"] = 1 if name == "get_weather" and "tokyo" in str(args.get("city", "")).lower() else 0
# E: multi-arg correctness
name, args = first_call(chat_tools(model, [b.user("Get the weather in Paris in fahrenheit.")],
[CALC, WEATHER, SEARCH]))
c["multiarg"] = 1 if (name == "get_weather" and "paris" in str(args.get("city", "")).lower()
and "f" in str(args.get("unit", "")).lower()) else 0
# C: multi-turn, use the returned result
msgs = [b.user("I have 17 boxes with 24 apples each. How many apples total? Use the calculator.")]
msg = chat_tools(model, msgs, [CALC])
name, _ = first_call(msg)
c["useresult"] = 0
if name == "calculator":
msgs.append({"role": "assistant", "content": "", "tool_calls": msg.get("tool_calls")})
msgs.append({"role": "tool", "content": "408"})
final = chat_tools(model, msgs, [CALC], num_predict=256)
c["useresult"] = 1 if "408" in (final.get("content", "") or "") else 0
# D: abstain when no tool is needed
name, _ = first_call(chat_tools(model, [b.user("Say hello in exactly one word.")],
[CALC, WEATHER, SEARCH]))
c["abstain"] = 1 if name is None else 0
c["overall"] = sum(c[k] for k in ("invoke", "args", "select", "multiarg", "useresult", "abstain")) / 6
return c
def main():
models = b.list_models()
results = {}
for i, m in enumerate(models, 1):
print(f"[{i}/{len(models)}] {m} ...", flush=True)
try:
results[m] = run_agent(m)
s = results[m]
print(f" overall {s['overall']*100:4.0f}% invoke {s['invoke']} args {s['args']} "
f"select {s['select']} multiarg {s['multiarg']} useresult {s['useresult']} "
f"abstain {s['abstain']}", flush=True)
except Exception as e:
print(f" ERROR: {e}", flush=True)
results[m] = {"error": str(e)}
with open("agent_results.json", "w") as f:
json.dump(results, f, indent=2)
ts = datetime.datetime.now().strftime("%Y-%m-%d %H:%M")
ranked = sorted([(m, s) for m, s in results.items() if "overall" in s],
key=lambda x: x[1]["overall"], reverse=True)
with open("agent_report.md", "w") as f:
f.write(f"# Agentic tool-use benchmark — {ts}\n\n")
f.write("Via Ollama tools API. Each column 0/1 (or %). abstain = correctly did NOT "
"call a tool when none was needed.\n\n")
f.write("| Model | Overall | Invoke | Args | Select | MultiArg | UseResult | Abstain |\n")
f.write("|---|--:|:--:|:--:|:--:|:--:|:--:|:--:|\n")
for m, s in ranked:
f.write(f"| `{m}` | **{s['overall']*100:.0f}%** | {s['invoke']} | {s['args']} | "
f"{s['select']} | {s['multiarg']} | {s['useresult']} | {s['abstain']} |\n")
print("\nWrote agent_report.md and agent_results.json")
if __name__ == "__main__":
main()