97 lines
3.3 KiB
Python
97 lines
3.3 KiB
Python
#!/usr/bin/env python3
|
|
"""Run the fixed Qwen acceptance prompts and a native tool-call probe."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import pathlib
|
|
import time
|
|
import urllib.request
|
|
|
|
|
|
def post(base: str, payload: dict, timeout: int = 1800) -> tuple[dict, float]:
|
|
request = urllib.request.Request(
|
|
base + "/v1/chat/completions",
|
|
data=json.dumps(payload).encode(),
|
|
headers={"Content-Type": "application/json"},
|
|
)
|
|
started = time.monotonic()
|
|
with urllib.request.urlopen(request, timeout=timeout) as response:
|
|
return json.load(response), time.monotonic() - started
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser()
|
|
parser.add_argument("label")
|
|
parser.add_argument("--base", default="http://127.0.0.1:5004")
|
|
parser.add_argument("--tasks", default="/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json")
|
|
parser.add_argument("--output", default="/data/benchmarks/dirk-qwen38")
|
|
args = parser.parse_args()
|
|
|
|
tasks = json.loads(pathlib.Path(args.tasks).read_text())
|
|
results = []
|
|
for task in tasks:
|
|
response, wall = post(args.base, {
|
|
"model": "benchmark",
|
|
"temperature": 0.2,
|
|
"max_tokens": task["max_tokens"],
|
|
"reasoning_effort": "medium",
|
|
"messages": [{"role": "user", "content": task["prompt"]}],
|
|
})
|
|
message = (response.get("choices") or [{}])[0].get("message") or {}
|
|
results.append({
|
|
"id": task["id"],
|
|
"finish_reason": (response.get("choices") or [{}])[0].get("finish_reason"),
|
|
"wall_seconds": round(wall, 3),
|
|
"content": message.get("content", ""),
|
|
"reasoning_content": message.get("reasoning_content", ""),
|
|
"usage": response.get("usage", {}),
|
|
"timings": response.get("timings", {}),
|
|
})
|
|
|
|
tool_response, tool_wall = post(args.base, {
|
|
"model": "benchmark",
|
|
"temperature": 0.2,
|
|
"max_tokens": 500,
|
|
"reasoning_effort": "none",
|
|
"messages": [{
|
|
"role": "user",
|
|
"content": "Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.",
|
|
}],
|
|
"tools": [{
|
|
"type": "function",
|
|
"function": {
|
|
"name": "read_server_status",
|
|
"description": "Read-only server status lookup",
|
|
"parameters": {
|
|
"type": "object",
|
|
"properties": {"server": {"type": "string"}},
|
|
"required": ["server"],
|
|
"additionalProperties": False,
|
|
},
|
|
},
|
|
}],
|
|
"tool_choice": "auto",
|
|
})
|
|
tool_message = (tool_response.get("choices") or [{}])[0].get("message") or {}
|
|
report = {
|
|
"label": args.label,
|
|
"results": results,
|
|
"tool_probe": {
|
|
"wall_seconds": round(tool_wall, 3),
|
|
"message": tool_message,
|
|
"timings": tool_response.get("timings", {}),
|
|
},
|
|
}
|
|
output = pathlib.Path(args.output)
|
|
output.mkdir(parents=True, exist_ok=True)
|
|
target = output / f"quality-{args.label}.json"
|
|
target.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n")
|
|
print(target)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|