Benchmark Dirk Qwen3.8 against production
This commit is contained in:
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run the fixed Qwen acceptance prompts and a native tool-call probe."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import pathlib
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
|
||||
def post(base: str, payload: dict, timeout: int = 1800) -> tuple[dict, float]:
|
||||
request = urllib.request.Request(
|
||||
base + "/v1/chat/completions",
|
||||
data=json.dumps(payload).encode(),
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
started = time.monotonic()
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
return json.load(response), time.monotonic() - started
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("label")
|
||||
parser.add_argument("--base", default="http://127.0.0.1:5004")
|
||||
parser.add_argument("--tasks", default="/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json")
|
||||
parser.add_argument("--output", default="/data/benchmarks/dirk-qwen38")
|
||||
args = parser.parse_args()
|
||||
|
||||
tasks = json.loads(pathlib.Path(args.tasks).read_text())
|
||||
results = []
|
||||
for task in tasks:
|
||||
response, wall = post(args.base, {
|
||||
"model": "benchmark",
|
||||
"temperature": 0.2,
|
||||
"max_tokens": task["max_tokens"],
|
||||
"reasoning_effort": "medium",
|
||||
"messages": [{"role": "user", "content": task["prompt"]}],
|
||||
})
|
||||
message = (response.get("choices") or [{}])[0].get("message") or {}
|
||||
results.append({
|
||||
"id": task["id"],
|
||||
"finish_reason": (response.get("choices") or [{}])[0].get("finish_reason"),
|
||||
"wall_seconds": round(wall, 3),
|
||||
"content": message.get("content", ""),
|
||||
"reasoning_content": message.get("reasoning_content", ""),
|
||||
"usage": response.get("usage", {}),
|
||||
"timings": response.get("timings", {}),
|
||||
})
|
||||
|
||||
tool_response, tool_wall = post(args.base, {
|
||||
"model": "benchmark",
|
||||
"temperature": 0.2,
|
||||
"max_tokens": 500,
|
||||
"reasoning_effort": "none",
|
||||
"messages": [{
|
||||
"role": "user",
|
||||
"content": "Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.",
|
||||
}],
|
||||
"tools": [{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "read_server_status",
|
||||
"description": "Read-only server status lookup",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"server": {"type": "string"}},
|
||||
"required": ["server"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
},
|
||||
}],
|
||||
"tool_choice": "auto",
|
||||
})
|
||||
tool_message = (tool_response.get("choices") or [{}])[0].get("message") or {}
|
||||
report = {
|
||||
"label": args.label,
|
||||
"results": results,
|
||||
"tool_probe": {
|
||||
"wall_seconds": round(tool_wall, 3),
|
||||
"message": tool_message,
|
||||
"timings": tool_response.get("timings", {}),
|
||||
},
|
||||
}
|
||||
output = pathlib.Path(args.output)
|
||||
output.mkdir(parents=True, exist_ok=True)
|
||||
target = output / f"quality-{args.label}.json"
|
||||
target.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n")
|
||||
print(target)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user