#!/usr/bin/env python3 """Run a private, deterministic llama.cpp vision latency smoke test.""" from __future__ import annotations import argparse import base64 import json import struct import time import urllib.request import zlib def chunk(kind: bytes, payload: bytes) -> bytes: return ( struct.pack(">I", len(payload)) + kind + payload + struct.pack(">I", zlib.crc32(kind + payload) & 0xFFFFFFFF) ) def synthetic_png(width: int = 1024, height: int = 768) -> bytes: rows = [] for y in range(height): row = bytearray([0]) for x in range(width): row.extend(((x * 255) // width, (y * 255) // height, ((x // 64 + y // 64) % 2) * 210)) rows.append(bytes(row)) payload = zlib.compress(b"".join(rows), level=6) return ( b"\x89PNG\r\n\x1a\n" + chunk(b"IHDR", struct.pack(">IIBBBBB", width, height, 8, 2, 0, 0, 0)) + chunk(b"IDAT", payload) + chunk(b"IEND", b"") ) def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--url", required=True) parser.add_argument("--model", default="qwen-medium") args = parser.parse_args() image = base64.b64encode(synthetic_png()).decode("ascii") body = { "model": args.model, "messages": [ { "role": "user", "content": [ { "type": "image_url", "image_url": {"url": f"data:image/png;base64,{image}"}, }, { "type": "text", "text": "Beschreibe dieses synthetische Testbild knapp auf Deutsch.", }, ], } ], "temperature": 0, "max_tokens": 500, "reasoning_effort": "none", "cache_prompt": False, } request = urllib.request.Request( args.url.rstrip("/") + "/v1/chat/completions", data=json.dumps(body).encode("utf-8"), headers={"Content-Type": "application/json"}, ) started = time.perf_counter() with urllib.request.urlopen(request, timeout=600) as response: result = json.load(response) elapsed = time.perf_counter() - started print( json.dumps( { "wall_seconds": round(elapsed, 3), "usage": result.get("usage"), "timings": result.get("timings"), "answer": result["choices"][0]["message"].get("content", "")[:240], }, ensure_ascii=False, indent=2, ) ) if __name__ == "__main__": main()