From 548f6643fdbd2f15625889b04fda922a1c851dea Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Tue, 1 Sep 2026 08:44:55 +0200 Subject: [PATCH] Benchmark Dirk Qwen3.8 against production --- dev/benchmark_vision.py | 3 +- docs/DIRK_QWEN38_AB_20260901.md | 80 +++++++++++++++++ experiments/dirk-qwen38/README.md | 15 ++-- experiments/dirk-qwen38/bench-case.py | 122 ++++++++++++++++++++++++++ experiments/dirk-qwen38/quality-ab.py | 96 ++++++++++++++++++++ experiments/dirk-qwen38/run-case.sh | 22 +++-- experiments/dirk-qwen38/wait-ready.sh | 6 +- 7 files changed, 326 insertions(+), 18 deletions(-) create mode 100644 docs/DIRK_QWEN38_AB_20260901.md create mode 100644 experiments/dirk-qwen38/bench-case.py create mode 100644 experiments/dirk-qwen38/quality-ab.py diff --git a/dev/benchmark_vision.py b/dev/benchmark_vision.py index 8467966..1a37b2a 100644 --- a/dev/benchmark_vision.py +++ b/dev/benchmark_vision.py @@ -62,7 +62,8 @@ def main() -> None: } ], "temperature": 0, - "max_tokens": 80, + "max_tokens": 500, + "reasoning_effort": "none", "cache_prompt": False, } request = urllib.request.Request( diff --git a/docs/DIRK_QWEN38_AB_20260901.md b/docs/DIRK_QWEN38_AB_20260901.md new file mode 100644 index 0000000..9d065c8 --- /dev/null +++ b/docs/DIRK_QWEN38_AB_20260901.md @@ -0,0 +1,80 @@ +# Dirk Qwen3.8-27B A/B benchmark (2026-09-01) + +Candidate: `peculiar-ragdoll/Dirk-Qwen3.8-27B-GGUF`, pinned revision +`12362f2b3d7dc11044e99c9e7e99fb9f530528c0`, quant +`Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf`. + +Reference: the production pure Qwen model +`qwen3.8-27b-IQ4_XS-pure.gguf`. + +Both sides used the same local llama.cpp image and production-style runtime +settings: one slot, Q4_0 KV, unified KV, 24 GiB prompt cache, batch 2048, +ubatch 128 and embedded MTP with draft length 3. The container-visible GPU +order was CUDA0 = RTX 5080 and CUDA1 = RTX 3060. A separate Python process +occupied about 1.9 GiB on the RTX 3060 throughout the run and was deliberately +not disturbed. + +## Stable candidate matrix + +| Context | Stable split (5080:3060) | Short prefill | Short decode | Long tested prompt | Long prefill | Long decode | Recall | +|---:|---:|---:|---:|---:|---:|---:|---:| +| 80k | 87:13 | 1,381 t/s | 68.8 t/s | 56.2k | 1,127 t/s | 51.8 t/s | 3/3 | +| 160k | 80:20 | 1,268 t/s | 64.2 t/s | 112.3k | 854 t/s | 40.1 t/s | 3/3 | +| 192k | 72:28 | 1,124 t/s | 60.2 t/s | 134.7k | 677 t/s | 35.3 t/s | 3/3 | +| 262,144 | 70:30 | 1,076 t/s | 59.6 t/s | 183.8k | 539 t/s | 29.9 t/s | 3/3 | + +At 80k, 88:12 loaded but failed on the first real prefill; 87:13 was the +maximum practical split. At 262k, 68:32 exhausted the RTX 3060 during KV +allocation and 72:28 exhausted the RTX 5080 during compute-buffer allocation. +70:30 was the only tested midpoint that loaded and completed the 183.8k-token +recall probe. + +## Fair 160k comparison + +| Model | Split | Short prefill | Short decode | 112k prefill | 112k decode | Recall | +|---|---:|---:|---:|---:|---:|---:| +| Pure IQ4_XS | 85:15 | 1,479 t/s | 104.4 t/s | 954 t/s | 56.4 t/s | 3/3 | +| Dirk Q4_K_XL | 80:20 | 1,268 t/s | 64.2 t/s | 854 t/s | 40.1 t/s | 3/3 | + +Dirk was 14% slower on short prefill, 11% slower on the long prefill, 38% +slower on short decode and 29% slower on long decode. + +## Quality and tool use + +The fixed acceptance set covered logic, evidence-based diagnosis, concurrent +Python, capacity planning, prompt-injection resistance, configuration versus +runtime state, safe read-only diagnostics and evidence boundaries. Both models +emitted the requested native function call with the correct argument. + +| Model | Completion tokens | Total task wall time | Mean decode | Completed final answers | +|---|---:|---:|---:|---:| +| Pure IQ4_XS | 17,542 | 224.9 s | 77.2 t/s | 7/9 before limit | +| Dirk Q4_K_XL | 12,178 | 260.2 s | 47.5 t/s | 9/9 | + +Dirk used about 31% fewer completion tokens and was more concise. It produced +the cleaner proof for the impossible live-migration task. Pure reached the +right conclusion but its state proof contained a source-host accounting error +and hit the output limit. Both concurrent-Python answers had a subtle remaining +edge case: simultaneously completed failing tasks outside the returned task +were not all gathered, so neither answer was perfect. + +Despite producing fewer tokens, Dirk needed about 16% more wall time for the +whole quality set because decode was much slower. + +## Vision + +The supplied F16 projector loaded at 160k with the 80:20 split. With reasoning +disabled, the private synthetic image was described correctly. Image prefill +was 106.7 t/s, decode was 39.7 t/s, and end-to-end latency was 11.0 seconds. + +## Decision + +Do not replace the production Pure Qwen medium profile with Dirk. Pure is the +clear speed winner and retained the same long-context recall and tool-call +ability. Dirk is useful only as an optional high-context/concise profile: it +can provide a verified 262k configured context on both GPUs and tends to spend +fewer output tokens, but it is slower in real elapsed time. + +The test container was removed after the run. Production `mike-ai-llama-medium` +and `mike-ai-llama-review` were restarted and verified healthy. + diff --git a/experiments/dirk-qwen38/README.md b/experiments/dirk-qwen38/README.md index 0fb49e7..9ab2bb5 100644 --- a/experiments/dirk-qwen38/README.md +++ b/experiments/dirk-qwen38/README.md @@ -23,24 +23,22 @@ therefore the mandatory A/B reference. container is running. It does not stop production itself. The model server is bound to loopback only and cannot be reached from the LAN. -## Prepared matrix +## Measured matrix Run the following only after the GPUs have explicitly been declared free: ```sh -./run-case.sh 80000 90,10 text -./run-case.sh 160000 90,10 text -./run-case.sh 160000 85,15 text +./run-case.sh 80000 87,13 text ./run-case.sh 160000 80,20 text -./run-case.sh 192000 85,15 text -./run-case.sh 262144 80,20 text +./run-case.sh 192000 72,28 text +./run-case.sh 262144 70,30 text ``` The largest stable context is determined first. Vision is checked only after a text winner exists: ```sh -./run-case.sh 160000 85,15 vision +./run-case.sh 160000 80,20 vision ``` For every case, record uncached prefill, cached prefill, decode throughput, @@ -50,3 +48,6 @@ technical correctness and improves real Hermes task completion. Expected SHA-256 checksums are stored in `MODEL_ARTIFACTS.sha256`. +The completed A/B result is documented in +`docs/DIRK_QWEN38_AB_20260901.md`. The candidate did not replace the production +Pure Qwen profile. diff --git a/experiments/dirk-qwen38/bench-case.py b/experiments/dirk-qwen38/bench-case.py new file mode 100644 index 0000000..7708000 --- /dev/null +++ b/experiments/dirk-qwen38/bench-case.py @@ -0,0 +1,122 @@ +#!/usr/bin/env python3 +"""Small, dependency-free llama.cpp performance and context probe.""" + +from __future__ import annotations + +import argparse +import json +import pathlib +import time +import urllib.request + + +def post(base: str, path: str, payload: dict, timeout: int = 1800) -> tuple[dict, float]: + request = urllib.request.Request( + base + path, + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + started = time.monotonic() + with urllib.request.urlopen(request, timeout=timeout) as response: + result = json.load(response) + return result, time.monotonic() - started + + +def make_text(lines: int) -> str: + return "\n".join( + f"Record {n:06d}: cobalt lantern maple orbit quartz river silver tango." for n in range(lines) + ) + + +def count_tokens(base: str, text: str) -> int: + result, _ = post(base, "/tokenize", {"content": text, "add_special": False}) + return len(result.get("tokens", [])) + + +def sized_text(base: str, target: int) -> tuple[str, int]: + # One probe establishes the tokenizer-specific tokens per synthetic line. + sample = make_text(100) + per_line = max(1.0, count_tokens(base, sample) / 100) + lines = max(1, int(target / per_line)) + text = make_text(lines) + actual = count_tokens(base, text) + if actual < target * 0.95: + lines = int(lines * target / max(1, actual)) + text = make_text(lines) + actual = count_tokens(base, text) + return text, actual + + +def chat(base: str, prompt: str, max_tokens: int, temperature: float = 0.2) -> dict: + result, wall = post(base, "/v1/chat/completions", { + "model": "benchmark", + "temperature": temperature, + "max_tokens": max_tokens, + "reasoning_effort": "none", + "messages": [{"role": "user", "content": prompt}], + }) + message = (result.get("choices") or [{}])[0].get("message") or {} + return { + "wall_seconds": round(wall, 3), + "timings": result.get("timings", {}), + "usage": result.get("usage", {}), + "content": message.get("content", ""), + "reasoning_content": message.get("reasoning_content", ""), + } + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("label") + parser.add_argument("context", type=int) + parser.add_argument("--base", default="http://127.0.0.1:5004") + parser.add_argument("--output", default="/data/benchmarks/dirk-qwen38") + args = parser.parse_args() + + result: dict = {"label": args.label, "context": args.context, "started": time.time()} + short, short_n = sized_text(args.base, min(16000, max(4000, args.context // 10))) + prompt = short + "\nReply with exactly: PREFILL-OK" + result["prompt_tokens_synthetic"] = short_n + result["uncached"] = chat(args.base, prompt, 32) + result["cached"] = chat(args.base, prompt, 32) + + output_prompt = ( + "Return exactly 256 comma-separated integers beginning at 1 and ending at 256. " + "Do not explain and do not omit any integer." + ) + result["decode"] = chat(args.base, output_prompt, 768) + + recall_target = int(args.context * 0.70) + long_text, long_n = sized_text(args.base, recall_target) + marks = [ + (len(long_text) // 8, "NEEDLE_ALPHA=RAVEN-417"), + (len(long_text) // 2, "NEEDLE_BETA=CEDAR-928"), + (len(long_text) * 7 // 8, "NEEDLE_GAMMA=ORBIT-563"), + ] + for position, needle in reversed(marks): + long_text = long_text[:position] + "\n" + needle + "\n" + long_text[position:] + recall_prompt = long_text + ( + "\nReturn only a JSON object with keys alpha, beta, gamma and their exact values " + "from the three NEEDLE lines." + ) + recall = chat(args.base, recall_prompt, 256) + recall["synthetic_tokens"] = long_n + content = recall.get("content", "") + recall["needles_found"] = { + "alpha": "RAVEN-417" in content, + "beta": "CEDAR-928" in content, + "gamma": "ORBIT-563" in content, + } + result["recall"] = recall + result["finished"] = time.time() + + output = pathlib.Path(args.output) + output.mkdir(parents=True, exist_ok=True) + target = output / f"{args.label}.json" + target.write_text(json.dumps(result, indent=2, ensure_ascii=False) + "\n") + print(json.dumps(result, indent=2, ensure_ascii=False)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/dirk-qwen38/quality-ab.py b/experiments/dirk-qwen38/quality-ab.py new file mode 100644 index 0000000..0b79251 --- /dev/null +++ b/experiments/dirk-qwen38/quality-ab.py @@ -0,0 +1,96 @@ +#!/usr/bin/env python3 +"""Run the fixed Qwen acceptance prompts and a native tool-call probe.""" + +from __future__ import annotations + +import argparse +import json +import pathlib +import time +import urllib.request + + +def post(base: str, payload: dict, timeout: int = 1800) -> tuple[dict, float]: + request = urllib.request.Request( + base + "/v1/chat/completions", + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + started = time.monotonic() + with urllib.request.urlopen(request, timeout=timeout) as response: + return json.load(response), time.monotonic() - started + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("label") + parser.add_argument("--base", default="http://127.0.0.1:5004") + parser.add_argument("--tasks", default="/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json") + parser.add_argument("--output", default="/data/benchmarks/dirk-qwen38") + args = parser.parse_args() + + tasks = json.loads(pathlib.Path(args.tasks).read_text()) + results = [] + for task in tasks: + response, wall = post(args.base, { + "model": "benchmark", + "temperature": 0.2, + "max_tokens": task["max_tokens"], + "reasoning_effort": "medium", + "messages": [{"role": "user", "content": task["prompt"]}], + }) + message = (response.get("choices") or [{}])[0].get("message") or {} + results.append({ + "id": task["id"], + "finish_reason": (response.get("choices") or [{}])[0].get("finish_reason"), + "wall_seconds": round(wall, 3), + "content": message.get("content", ""), + "reasoning_content": message.get("reasoning_content", ""), + "usage": response.get("usage", {}), + "timings": response.get("timings", {}), + }) + + tool_response, tool_wall = post(args.base, { + "model": "benchmark", + "temperature": 0.2, + "max_tokens": 500, + "reasoning_effort": "none", + "messages": [{ + "role": "user", + "content": "Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.", + }], + "tools": [{ + "type": "function", + "function": { + "name": "read_server_status", + "description": "Read-only server status lookup", + "parameters": { + "type": "object", + "properties": {"server": {"type": "string"}}, + "required": ["server"], + "additionalProperties": False, + }, + }, + }], + "tool_choice": "auto", + }) + tool_message = (tool_response.get("choices") or [{}])[0].get("message") or {} + report = { + "label": args.label, + "results": results, + "tool_probe": { + "wall_seconds": round(tool_wall, 3), + "message": tool_message, + "timings": tool_response.get("timings", {}), + }, + } + output = pathlib.Path(args.output) + output.mkdir(parents=True, exist_ok=True) + target = output / f"quality-{args.label}.json" + target.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n") + print(target) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/dirk-qwen38/run-case.sh b/experiments/dirk-qwen38/run-case.sh index a943aa1..cdf784a 100755 --- a/experiments/dirk-qwen38/run-case.sh +++ b/experiments/dirk-qwen38/run-case.sh @@ -15,14 +15,16 @@ if [[ $MODE != text && $MODE != vision ]]; then fi MODEL_DIR=${MODEL_DIR:-/data/models/qwen3.8-27b-dirk} -MODEL=/models/qwen3.8-27b-dirk/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf +MODEL=${MODEL_PATH:-/models/qwen3.8-27b-dirk/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf} +HOST_MODEL_FILE=${HOST_MODEL_FILE:-$MODEL_DIR/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf} +MODEL_ALIAS=${MODEL_ALIAS:-qwen-dirk-test} PROJECTOR=/models/qwen3.8-27b-dirk/mmproj-F16.gguf IMAGE=${LLAMA_IMAGE:-mike-ai/llama.cpp:local} NAME=mike-ai-llama-dirk-test RESULT_DIR=${RESULT_DIR:-/data/benchmarks/dirk-qwen38} -[[ -s "$MODEL_DIR/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf" ]] || { - echo "Dirk model is missing; run download-model.sh first" >&2 +[[ -s "$HOST_MODEL_FILE" ]] || { + echo "Model file is missing: $HOST_MODEL_FILE" >&2 exit 1 } if [[ $MODE == vision && ! -s "$MODEL_DIR/mmproj-F16.gguf" ]]; then @@ -45,21 +47,23 @@ install -d -m 0755 "$RESULT_DIR" args=( --model "$MODEL" - --alias qwen-dirk-test + --alias "$MODEL_ALIAS" --ctx-size "$CONTEXT" --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt - --cache-ram 8192 + --cache-ram 24576 --threads 6 --threads-batch 6 - --batch-size 64 - --ubatch-size 32 + --batch-size 2048 + --ubatch-size 128 --parallel 1 + --kv-unified --jinja --reasoning auto --reasoning-budget 8192 + --reasoning-preserve --host 127.0.0.1 --port 5004 --metrics @@ -77,6 +81,7 @@ args=( --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 + --spec-draft-p-min 0.05 ) if [[ $MODE == vision ]]; then @@ -84,7 +89,7 @@ if [[ $MODE == vision ]]; then fi label="ctx${CONTEXT}-split${SPLIT//,/-}-${MODE}" -docker run -d --rm \ +docker run -d \ --name "$NAME" \ --gpus all \ --network host \ @@ -102,4 +107,3 @@ docker run -d --rm \ "$IMAGE" "${args[@]}" printf 'Started isolated case %s on http://127.0.0.1:5004\n' "$label" - diff --git a/experiments/dirk-qwen38/wait-ready.sh b/experiments/dirk-qwen38/wait-ready.sh index 95d1f42..18757e4 100755 --- a/experiments/dirk-qwen38/wait-ready.sh +++ b/experiments/dirk-qwen38/wait-ready.sh @@ -3,6 +3,11 @@ set -Eeuo pipefail deadline=$((SECONDS + ${READY_TIMEOUT:-900})) until curl -fsS --max-time 3 http://127.0.0.1:5004/health >/dev/null; do + state=$(docker inspect -f '{{.State.Running}}' mike-ai-llama-dirk-test 2>/dev/null || true) + if [[ $state != true ]]; then + docker logs --tail 100 mike-ai-llama-dirk-test >&2 || true + exit 1 + fi if ((SECONDS >= deadline)); then docker logs --tail 100 mike-ai-llama-dirk-test >&2 || true exit 1 @@ -11,4 +16,3 @@ until curl -fsS --max-time 3 http://127.0.0.1:5004/health >/dev/null; do done curl -fsS http://127.0.0.1:5004/props printf '\nDirk test server is ready.\n' -