Benchmark Dirk Qwen3.8 against production

This commit is contained in:
Mikei386
2026-09-01 08:44:55 +02:00
parent 78096c9027
commit 548f6643fd
7 changed files with 326 additions and 18 deletions
+8 -7
View File
@@ -23,24 +23,22 @@ therefore the mandatory A/B reference.
container is running. It does not stop production itself. The model server is
bound to loopback only and cannot be reached from the LAN.
## Prepared matrix
## Measured matrix
Run the following only after the GPUs have explicitly been declared free:
```sh
./run-case.sh 80000 90,10 text
./run-case.sh 160000 90,10 text
./run-case.sh 160000 85,15 text
./run-case.sh 80000 87,13 text
./run-case.sh 160000 80,20 text
./run-case.sh 192000 85,15 text
./run-case.sh 262144 80,20 text
./run-case.sh 192000 72,28 text
./run-case.sh 262144 70,30 text
```
The largest stable context is determined first. Vision is checked only after a
text winner exists:
```sh
./run-case.sh 160000 85,15 vision
./run-case.sh 160000 80,20 vision
```
For every case, record uncached prefill, cached prefill, decode throughput,
@@ -50,3 +48,6 @@ technical correctness and improves real Hermes task completion.
Expected SHA-256 checksums are stored in `MODEL_ARTIFACTS.sha256`.
The completed A/B result is documented in
`docs/DIRK_QWEN38_AB_20260901.md`. The candidate did not replace the production
Pure Qwen profile.
+122
View File
@@ -0,0 +1,122 @@
#!/usr/bin/env python3
"""Small, dependency-free llama.cpp performance and context probe."""
from __future__ import annotations
import argparse
import json
import pathlib
import time
import urllib.request
def post(base: str, path: str, payload: dict, timeout: int = 1800) -> tuple[dict, float]:
request = urllib.request.Request(
base + path,
data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json"},
)
started = time.monotonic()
with urllib.request.urlopen(request, timeout=timeout) as response:
result = json.load(response)
return result, time.monotonic() - started
def make_text(lines: int) -> str:
return "\n".join(
f"Record {n:06d}: cobalt lantern maple orbit quartz river silver tango." for n in range(lines)
)
def count_tokens(base: str, text: str) -> int:
result, _ = post(base, "/tokenize", {"content": text, "add_special": False})
return len(result.get("tokens", []))
def sized_text(base: str, target: int) -> tuple[str, int]:
# One probe establishes the tokenizer-specific tokens per synthetic line.
sample = make_text(100)
per_line = max(1.0, count_tokens(base, sample) / 100)
lines = max(1, int(target / per_line))
text = make_text(lines)
actual = count_tokens(base, text)
if actual < target * 0.95:
lines = int(lines * target / max(1, actual))
text = make_text(lines)
actual = count_tokens(base, text)
return text, actual
def chat(base: str, prompt: str, max_tokens: int, temperature: float = 0.2) -> dict:
result, wall = post(base, "/v1/chat/completions", {
"model": "benchmark",
"temperature": temperature,
"max_tokens": max_tokens,
"reasoning_effort": "none",
"messages": [{"role": "user", "content": prompt}],
})
message = (result.get("choices") or [{}])[0].get("message") or {}
return {
"wall_seconds": round(wall, 3),
"timings": result.get("timings", {}),
"usage": result.get("usage", {}),
"content": message.get("content", ""),
"reasoning_content": message.get("reasoning_content", ""),
}
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("label")
parser.add_argument("context", type=int)
parser.add_argument("--base", default="http://127.0.0.1:5004")
parser.add_argument("--output", default="/data/benchmarks/dirk-qwen38")
args = parser.parse_args()
result: dict = {"label": args.label, "context": args.context, "started": time.time()}
short, short_n = sized_text(args.base, min(16000, max(4000, args.context // 10)))
prompt = short + "\nReply with exactly: PREFILL-OK"
result["prompt_tokens_synthetic"] = short_n
result["uncached"] = chat(args.base, prompt, 32)
result["cached"] = chat(args.base, prompt, 32)
output_prompt = (
"Return exactly 256 comma-separated integers beginning at 1 and ending at 256. "
"Do not explain and do not omit any integer."
)
result["decode"] = chat(args.base, output_prompt, 768)
recall_target = int(args.context * 0.70)
long_text, long_n = sized_text(args.base, recall_target)
marks = [
(len(long_text) // 8, "NEEDLE_ALPHA=RAVEN-417"),
(len(long_text) // 2, "NEEDLE_BETA=CEDAR-928"),
(len(long_text) * 7 // 8, "NEEDLE_GAMMA=ORBIT-563"),
]
for position, needle in reversed(marks):
long_text = long_text[:position] + "\n" + needle + "\n" + long_text[position:]
recall_prompt = long_text + (
"\nReturn only a JSON object with keys alpha, beta, gamma and their exact values "
"from the three NEEDLE lines."
)
recall = chat(args.base, recall_prompt, 256)
recall["synthetic_tokens"] = long_n
content = recall.get("content", "")
recall["needles_found"] = {
"alpha": "RAVEN-417" in content,
"beta": "CEDAR-928" in content,
"gamma": "ORBIT-563" in content,
}
result["recall"] = recall
result["finished"] = time.time()
output = pathlib.Path(args.output)
output.mkdir(parents=True, exist_ok=True)
target = output / f"{args.label}.json"
target.write_text(json.dumps(result, indent=2, ensure_ascii=False) + "\n")
print(json.dumps(result, indent=2, ensure_ascii=False))
return 0
if __name__ == "__main__":
raise SystemExit(main())
+96
View File
@@ -0,0 +1,96 @@
#!/usr/bin/env python3
"""Run the fixed Qwen acceptance prompts and a native tool-call probe."""
from __future__ import annotations
import argparse
import json
import pathlib
import time
import urllib.request
def post(base: str, payload: dict, timeout: int = 1800) -> tuple[dict, float]:
request = urllib.request.Request(
base + "/v1/chat/completions",
data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json"},
)
started = time.monotonic()
with urllib.request.urlopen(request, timeout=timeout) as response:
return json.load(response), time.monotonic() - started
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("label")
parser.add_argument("--base", default="http://127.0.0.1:5004")
parser.add_argument("--tasks", default="/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json")
parser.add_argument("--output", default="/data/benchmarks/dirk-qwen38")
args = parser.parse_args()
tasks = json.loads(pathlib.Path(args.tasks).read_text())
results = []
for task in tasks:
response, wall = post(args.base, {
"model": "benchmark",
"temperature": 0.2,
"max_tokens": task["max_tokens"],
"reasoning_effort": "medium",
"messages": [{"role": "user", "content": task["prompt"]}],
})
message = (response.get("choices") or [{}])[0].get("message") or {}
results.append({
"id": task["id"],
"finish_reason": (response.get("choices") or [{}])[0].get("finish_reason"),
"wall_seconds": round(wall, 3),
"content": message.get("content", ""),
"reasoning_content": message.get("reasoning_content", ""),
"usage": response.get("usage", {}),
"timings": response.get("timings", {}),
})
tool_response, tool_wall = post(args.base, {
"model": "benchmark",
"temperature": 0.2,
"max_tokens": 500,
"reasoning_effort": "none",
"messages": [{
"role": "user",
"content": "Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.",
}],
"tools": [{
"type": "function",
"function": {
"name": "read_server_status",
"description": "Read-only server status lookup",
"parameters": {
"type": "object",
"properties": {"server": {"type": "string"}},
"required": ["server"],
"additionalProperties": False,
},
},
}],
"tool_choice": "auto",
})
tool_message = (tool_response.get("choices") or [{}])[0].get("message") or {}
report = {
"label": args.label,
"results": results,
"tool_probe": {
"wall_seconds": round(tool_wall, 3),
"message": tool_message,
"timings": tool_response.get("timings", {}),
},
}
output = pathlib.Path(args.output)
output.mkdir(parents=True, exist_ok=True)
target = output / f"quality-{args.label}.json"
target.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n")
print(target)
return 0
if __name__ == "__main__":
raise SystemExit(main())
+13 -9
View File
@@ -15,14 +15,16 @@ if [[ $MODE != text && $MODE != vision ]]; then
fi
MODEL_DIR=${MODEL_DIR:-/data/models/qwen3.8-27b-dirk}
MODEL=/models/qwen3.8-27b-dirk/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf
MODEL=${MODEL_PATH:-/models/qwen3.8-27b-dirk/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf}
HOST_MODEL_FILE=${HOST_MODEL_FILE:-$MODEL_DIR/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf}
MODEL_ALIAS=${MODEL_ALIAS:-qwen-dirk-test}
PROJECTOR=/models/qwen3.8-27b-dirk/mmproj-F16.gguf
IMAGE=${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
NAME=mike-ai-llama-dirk-test
RESULT_DIR=${RESULT_DIR:-/data/benchmarks/dirk-qwen38}
[[ -s "$MODEL_DIR/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf" ]] || {
echo "Dirk model is missing; run download-model.sh first" >&2
[[ -s "$HOST_MODEL_FILE" ]] || {
echo "Model file is missing: $HOST_MODEL_FILE" >&2
exit 1
}
if [[ $MODE == vision && ! -s "$MODEL_DIR/mmproj-F16.gguf" ]]; then
@@ -45,21 +47,23 @@ install -d -m 0755 "$RESULT_DIR"
args=(
--model "$MODEL"
--alias qwen-dirk-test
--alias "$MODEL_ALIAS"
--ctx-size "$CONTEXT"
--flash-attn on
--cache-type-k q4_0
--cache-type-v q4_0
--cache-prompt
--cache-ram 8192
--cache-ram 24576
--threads 6
--threads-batch 6
--batch-size 64
--ubatch-size 32
--batch-size 2048
--ubatch-size 128
--parallel 1
--kv-unified
--jinja
--reasoning auto
--reasoning-budget 8192
--reasoning-preserve
--host 127.0.0.1
--port 5004
--metrics
@@ -77,6 +81,7 @@ args=(
--spec-draft-n-max 3
--spec-draft-type-k f16
--spec-draft-type-v f16
--spec-draft-p-min 0.05
)
if [[ $MODE == vision ]]; then
@@ -84,7 +89,7 @@ if [[ $MODE == vision ]]; then
fi
label="ctx${CONTEXT}-split${SPLIT//,/-}-${MODE}"
docker run -d --rm \
docker run -d \
--name "$NAME" \
--gpus all \
--network host \
@@ -102,4 +107,3 @@ docker run -d --rm \
"$IMAGE" "${args[@]}"
printf 'Started isolated case %s on http://127.0.0.1:5004\n' "$label"
+5 -1
View File
@@ -3,6 +3,11 @@ set -Eeuo pipefail
deadline=$((SECONDS + ${READY_TIMEOUT:-900}))
until curl -fsS --max-time 3 http://127.0.0.1:5004/health >/dev/null; do
state=$(docker inspect -f '{{.State.Running}}' mike-ai-llama-dirk-test 2>/dev/null || true)
if [[ $state != true ]]; then
docker logs --tail 100 mike-ai-llama-dirk-test >&2 || true
exit 1
fi
if ((SECONDS >= deadline)); then
docker logs --tail 100 mike-ai-llama-dirk-test >&2 || true
exit 1
@@ -11,4 +16,3 @@ until curl -fsS --max-time 3 http://127.0.0.1:5004/health >/dev/null; do
done
curl -fsS http://127.0.0.1:5004/props
printf '\nDirk test server is ready.\n'