Prepare isolated Dirk Qwen3.8 benchmark

This commit is contained in:
Mikei386
2026-09-01 06:18:39 +02:00
parent b2ea53c383
commit ee28272999
5 changed files with 201 additions and 0 deletions
+49
View File
@@ -0,0 +1,49 @@
# Dirk Qwen3.8-27B experiment
Isolated A/B test environment for
`peculiar-ragdoll/Dirk-Qwen3.8-27B-GGUF`. It deliberately does not add a
production router profile and never stops or restarts production services.
## Candidate
- Main model: `Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf` (about 17.6 GB)
- Vision projector: `mmproj-F16.gguf`
- Runtime: existing `mike-ai/llama.cpp:local`
- Test endpoint: `127.0.0.1:5004`
- Results: `/data/benchmarks/dirk-qwen38/`
The candidate is intended to reduce unnecessary reasoning and total token use;
it is not expected to improve raw decode speed. The production Qwen model is
therefore the mandatory A/B reference.
## Safety boundary
`run-case.sh` refuses to start while any production `mike-ai-llama-*` model
container is running. It does not stop production itself. The model server is
bound to loopback only and cannot be reached from the LAN.
## Prepared matrix
Run the following only after the GPUs have explicitly been declared free:
```sh
./run-case.sh 80000 90,10 text
./run-case.sh 160000 90,10 text
./run-case.sh 160000 85,15 text
./run-case.sh 160000 80,20 text
./run-case.sh 192000 85,15 text
./run-case.sh 262144 80,20 text
```
The largest stable context is determined first. Vision is checked only after a
text winner exists:
```sh
./run-case.sh 160000 85,15 vision
```
For every case, record uncached prefill, cached prefill, decode throughput,
GPU memory, context recall, tool calling, code quality and total tokens needed
to finish the task. Do not promote Dirk unless it matches the base model on
technical correctness and improves real Hermes task completion.
+28
View File
@@ -0,0 +1,28 @@
#!/usr/bin/env bash
set -Eeuo pipefail
MODEL_DIR=${MODEL_DIR:-/data/models/qwen3.8-27b-dirk}
BASE_URL=https://huggingface.co/peculiar-ragdoll/Dirk-Qwen3.8-27B-GGUF/resolve/main
MODEL=Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf
PROJECTOR=mmproj-F16.gguf
install -d -m 0755 "$MODEL_DIR"
download() {
local name=$1
local target="$MODEL_DIR/$name"
curl --fail --location --retry 8 --retry-delay 5 --continue-at - \
--output "$target.part" "$BASE_URL/$name"
mv -f "$target.part" "$target"
}
[[ -s "$MODEL_DIR/$MODEL" ]] || download "$MODEL"
[[ -s "$MODEL_DIR/$PROJECTOR" ]] || download "$PROJECTOR"
(
cd "$MODEL_DIR"
sha256sum "$MODEL" "$PROJECTOR" > SHA256SUMS
)
printf 'Prepared model files in %s\n' "$MODEL_DIR"
+105
View File
@@ -0,0 +1,105 @@
#!/usr/bin/env bash
set -Eeuo pipefail
CONTEXT=${1:-}
SPLIT=${2:-}
MODE=${3:-text}
if [[ -z $CONTEXT || -z $SPLIT || ! $CONTEXT =~ ^[0-9]+$ || ! $SPLIT =~ ^[0-9]+,[0-9]+$ ]]; then
echo "Usage: $0 CONTEXT TENSOR_SPLIT {text|vision}" >&2
exit 2
fi
if [[ $MODE != text && $MODE != vision ]]; then
echo "Mode must be text or vision" >&2
exit 2
fi
MODEL_DIR=${MODEL_DIR:-/data/models/qwen3.8-27b-dirk}
MODEL=/models/qwen3.8-27b-dirk/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf
PROJECTOR=/models/qwen3.8-27b-dirk/mmproj-F16.gguf
IMAGE=${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
NAME=mike-ai-llama-dirk-test
RESULT_DIR=${RESULT_DIR:-/data/benchmarks/dirk-qwen38}
[[ -s "$MODEL_DIR/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf" ]] || {
echo "Dirk model is missing; run download-model.sh first" >&2
exit 1
}
if [[ $MODE == vision && ! -s "$MODEL_DIR/mmproj-F16.gguf" ]]; then
echo "Vision projector is missing" >&2
exit 1
fi
mapfile -t blockers < <(
docker ps --format '{{.Names}}' |
grep -E '^mike-ai-llama-' |
grep -vE '^mike-ai-llama-dashboard$|^mike-ai-llama-dirk-test$' || true
)
if ((${#blockers[@]})); then
printf 'Refusing to start: production model container(s) still running: %s\n' "${blockers[*]}" >&2
exit 1
fi
docker rm -f "$NAME" >/dev/null 2>&1 || true
install -d -m 0755 "$RESULT_DIR"
args=(
--model "$MODEL"
--alias qwen-dirk-test
--ctx-size "$CONTEXT"
--flash-attn on
--cache-type-k q4_0
--cache-type-v q4_0
--cache-prompt
--cache-ram 8192
--threads 6
--threads-batch 6
--batch-size 64
--ubatch-size 32
--parallel 1
--jinja
--reasoning auto
--reasoning-budget 8192
--host 127.0.0.1
--port 5004
--metrics
--fit off
--n-gpu-layers all
--no-mmap
--temperature 1.0
--top-p 0.95
--top-k 20
--device CUDA0,CUDA1
--main-gpu 0
--split-mode layer
--tensor-split "$SPLIT"
--spec-type draft-mtp
--spec-draft-n-max 3
--spec-draft-type-k f16
--spec-draft-type-v f16
)
if [[ $MODE == vision ]]; then
args+=(--mmproj "$PROJECTOR" --no-mmproj-offload)
fi
label="ctx${CONTEXT}-split${SPLIT//,/-}-${MODE}"
docker run -d --rm \
--name "$NAME" \
--gpus all \
--network host \
--read-only \
--tmpfs /tmp:rw,noexec,nosuid,nodev,size=256m \
--security-opt no-new-privileges:true \
--cap-drop ALL \
--pids-limit 1024 \
--log-opt max-size=20m \
--log-opt max-file=2 \
-v /data/models:/models:ro \
-v "$RESULT_DIR":/results \
--label mike-ai.experiment=dirk-qwen38 \
--label mike-ai.case="$label" \
"$IMAGE" "${args[@]}"
printf 'Started isolated case %s on http://127.0.0.1:5004\n' "$label"
+5
View File
@@ -0,0 +1,5 @@
#!/usr/bin/env bash
set -Eeuo pipefail
docker rm -f mike-ai-llama-dirk-test >/dev/null 2>&1 || true
echo "Dirk test container stopped. Production was not changed."
+14
View File
@@ -0,0 +1,14 @@
#!/usr/bin/env bash
set -Eeuo pipefail
deadline=$((SECONDS + ${READY_TIMEOUT:-900}))
until curl -fsS --max-time 3 http://127.0.0.1:5004/health >/dev/null; do
if ((SECONDS >= deadline)); then
docker logs --tail 100 mike-ai-llama-dirk-test >&2 || true
exit 1
fi
sleep 2
done
curl -fsS http://127.0.0.1:5004/props
printf '\nDirk test server is ready.\n'