Document Ornith 1.5 A/B benchmark
This commit is contained in:
Executable
+79
@@ -0,0 +1,79 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Identical, read-only A/B prompts against one OpenAI-compatible endpoint."""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
|
||||
def request(base, key, payload, timeout=1800):
|
||||
headers = {"Content-Type": "application/json"}
|
||||
if key:
|
||||
headers["Authorization"] = "Bearer " + key
|
||||
req = urllib.request.Request(base.rstrip("/") + "/v1/chat/completions",
|
||||
data=json.dumps(payload).encode(), headers=headers)
|
||||
start = time.monotonic()
|
||||
with urllib.request.urlopen(req, timeout=timeout) as response:
|
||||
result = json.load(response)
|
||||
elapsed = time.monotonic() - start
|
||||
choice = (result.get("choices") or [{}])[0]
|
||||
msg = choice.get("message") or {}
|
||||
return {
|
||||
"wall_seconds": round(elapsed, 3), "usage": result.get("usage", {}),
|
||||
"timings": result.get("timings", {}), "finish_reason": choice.get("finish_reason"),
|
||||
"content": msg.get("content", ""), "reasoning_content": msg.get("reasoning_content", ""),
|
||||
"tool_calls": msg.get("tool_calls", []),
|
||||
}
|
||||
|
||||
|
||||
def chat(base, key, model, prompt, max_tokens=512, tools=None, effort="medium"):
|
||||
payload = {"model": model, "stream": False, "temperature": 0.2,
|
||||
"seed": 42, "reasoning_effort": effort, "max_tokens": max_tokens,
|
||||
"messages": [{"role": "user", "content": prompt}]}
|
||||
if tools:
|
||||
payload["tools"] = tools
|
||||
payload["tool_choice"] = "auto"
|
||||
return request(base, key, payload)
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("--label", required=True)
|
||||
p.add_argument("--base", required=True)
|
||||
p.add_argument("--model", required=True)
|
||||
p.add_argument("--key-env", default="BENCH_API_KEY")
|
||||
p.add_argument("--tasks", type=pathlib.Path, default=pathlib.Path(__file__).with_name("tasks.json"))
|
||||
p.add_argument("--output", type=pathlib.Path, required=True)
|
||||
p.add_argument("--go", action="store_true")
|
||||
args = p.parse_args()
|
||||
if not args.go:
|
||||
p.error("No inference before explicit --go")
|
||||
key = os.environ.get(args.key_env, "")
|
||||
tasks = json.loads(args.tasks.read_text())
|
||||
report = {"label": args.label, "model": args.model, "started": time.time(), "tasks": []}
|
||||
for task in tasks:
|
||||
answer = chat(args.base, key, args.model, task["prompt"], task["max_tokens"])
|
||||
report["tasks"].append({"id": task["id"], **answer})
|
||||
print(task["id"], answer["wall_seconds"], flush=True)
|
||||
# Same prompt twice reveals uncached prefill and prompt-cache reuse.
|
||||
prompt = ("In one sentence, explain why a 10 mm through-hole in a 40 mm cube "
|
||||
"does not change its external dimensions. " * 1000) + "Answer now."
|
||||
report["prefill_first"] = chat(args.base, key, args.model, prompt, 128)
|
||||
report["prefill_repeat"] = chat(args.base, key, args.model, prompt, 128)
|
||||
report["decode"] = chat(args.base, key, args.model,
|
||||
"Write a numbered list of exactly 100 distinct workshop safety tips.", 2048)
|
||||
report["tool"] = chat(args.base, key, args.model,
|
||||
"What is the current temperature in Rastatt? Use get_weather once; do not invent the result.",
|
||||
256, [{"type": "function", "function": {"name": "get_weather",
|
||||
"description": "Get current weather for a city", "parameters": {"type": "object",
|
||||
"properties": {"city": {"type": "string"}}, "required": ["city"]}}}], effort="none")
|
||||
report["finished"] = time.time()
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n")
|
||||
print(args.output)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user