Benchmark Dirk Qwen3.8 against production

This commit is contained in:
Mikei386
2026-09-01 08:44:55 +02:00
parent 78096c9027
commit 548f6643fd
7 changed files with 326 additions and 18 deletions
+96
View File
@@ -0,0 +1,96 @@
#!/usr/bin/env python3
"""Run the fixed Qwen acceptance prompts and a native tool-call probe."""
from __future__ import annotations
import argparse
import json
import pathlib
import time
import urllib.request
def post(base: str, payload: dict, timeout: int = 1800) -> tuple[dict, float]:
request = urllib.request.Request(
base + "/v1/chat/completions",
data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json"},
)
started = time.monotonic()
with urllib.request.urlopen(request, timeout=timeout) as response:
return json.load(response), time.monotonic() - started
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("label")
parser.add_argument("--base", default="http://127.0.0.1:5004")
parser.add_argument("--tasks", default="/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json")
parser.add_argument("--output", default="/data/benchmarks/dirk-qwen38")
args = parser.parse_args()
tasks = json.loads(pathlib.Path(args.tasks).read_text())
results = []
for task in tasks:
response, wall = post(args.base, {
"model": "benchmark",
"temperature": 0.2,
"max_tokens": task["max_tokens"],
"reasoning_effort": "medium",
"messages": [{"role": "user", "content": task["prompt"]}],
})
message = (response.get("choices") or [{}])[0].get("message") or {}
results.append({
"id": task["id"],
"finish_reason": (response.get("choices") or [{}])[0].get("finish_reason"),
"wall_seconds": round(wall, 3),
"content": message.get("content", ""),
"reasoning_content": message.get("reasoning_content", ""),
"usage": response.get("usage", {}),
"timings": response.get("timings", {}),
})
tool_response, tool_wall = post(args.base, {
"model": "benchmark",
"temperature": 0.2,
"max_tokens": 500,
"reasoning_effort": "none",
"messages": [{
"role": "user",
"content": "Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.",
}],
"tools": [{
"type": "function",
"function": {
"name": "read_server_status",
"description": "Read-only server status lookup",
"parameters": {
"type": "object",
"properties": {"server": {"type": "string"}},
"required": ["server"],
"additionalProperties": False,
},
},
}],
"tool_choice": "auto",
})
tool_message = (tool_response.get("choices") or [{}])[0].get("message") or {}
report = {
"label": args.label,
"results": results,
"tool_probe": {
"wall_seconds": round(tool_wall, 3),
"message": tool_message,
"timings": tool_response.get("timings", {}),
},
}
output = pathlib.Path(args.output)
output.mkdir(parents=True, exist_ok=True)
target = output / f"quality-{args.label}.json"
target.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n")
print(target)
return 0
if __name__ == "__main__":
raise SystemExit(main())