Document Ornith 1.5 A/B benchmark

This commit is contained in:
Mikei386
2026-09-19 19:31:20 +02:00
parent 9978e5b7e6
commit 9af719674b
27 changed files with 3503 additions and 1 deletions
+79
View File
@@ -0,0 +1,79 @@
#!/usr/bin/env python3
"""Identical, read-only A/B prompts against one OpenAI-compatible endpoint."""
import argparse
import json
import os
import pathlib
import time
import urllib.request
def request(base, key, payload, timeout=1800):
headers = {"Content-Type": "application/json"}
if key:
headers["Authorization"] = "Bearer " + key
req = urllib.request.Request(base.rstrip("/") + "/v1/chat/completions",
data=json.dumps(payload).encode(), headers=headers)
start = time.monotonic()
with urllib.request.urlopen(req, timeout=timeout) as response:
result = json.load(response)
elapsed = time.monotonic() - start
choice = (result.get("choices") or [{}])[0]
msg = choice.get("message") or {}
return {
"wall_seconds": round(elapsed, 3), "usage": result.get("usage", {}),
"timings": result.get("timings", {}), "finish_reason": choice.get("finish_reason"),
"content": msg.get("content", ""), "reasoning_content": msg.get("reasoning_content", ""),
"tool_calls": msg.get("tool_calls", []),
}
def chat(base, key, model, prompt, max_tokens=512, tools=None, effort="medium"):
payload = {"model": model, "stream": False, "temperature": 0.2,
"seed": 42, "reasoning_effort": effort, "max_tokens": max_tokens,
"messages": [{"role": "user", "content": prompt}]}
if tools:
payload["tools"] = tools
payload["tool_choice"] = "auto"
return request(base, key, payload)
def main():
p = argparse.ArgumentParser()
p.add_argument("--label", required=True)
p.add_argument("--base", required=True)
p.add_argument("--model", required=True)
p.add_argument("--key-env", default="BENCH_API_KEY")
p.add_argument("--tasks", type=pathlib.Path, default=pathlib.Path(__file__).with_name("tasks.json"))
p.add_argument("--output", type=pathlib.Path, required=True)
p.add_argument("--go", action="store_true")
args = p.parse_args()
if not args.go:
p.error("No inference before explicit --go")
key = os.environ.get(args.key_env, "")
tasks = json.loads(args.tasks.read_text())
report = {"label": args.label, "model": args.model, "started": time.time(), "tasks": []}
for task in tasks:
answer = chat(args.base, key, args.model, task["prompt"], task["max_tokens"])
report["tasks"].append({"id": task["id"], **answer})
print(task["id"], answer["wall_seconds"], flush=True)
# Same prompt twice reveals uncached prefill and prompt-cache reuse.
prompt = ("In one sentence, explain why a 10 mm through-hole in a 40 mm cube "
"does not change its external dimensions. " * 1000) + "Answer now."
report["prefill_first"] = chat(args.base, key, args.model, prompt, 128)
report["prefill_repeat"] = chat(args.base, key, args.model, prompt, 128)
report["decode"] = chat(args.base, key, args.model,
"Write a numbered list of exactly 100 distinct workshop safety tips.", 2048)
report["tool"] = chat(args.base, key, args.model,
"What is the current temperature in Rastatt? Use get_weather once; do not invent the result.",
256, [{"type": "function", "function": {"name": "get_weather",
"description": "Get current weather for a city", "parameters": {"type": "object",
"properties": {"city": {"type": "string"}}, "required": ["city"]}}}], effort="none")
report["finished"] = time.time()
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n")
print(args.output)
if __name__ == "__main__":
main()