Prepare isolated Dirk Qwen3.8 benchmark
This commit is contained in:
@@ -0,0 +1,49 @@
|
|||||||
|
# Dirk Qwen3.8-27B experiment
|
||||||
|
|
||||||
|
Isolated A/B test environment for
|
||||||
|
`peculiar-ragdoll/Dirk-Qwen3.8-27B-GGUF`. It deliberately does not add a
|
||||||
|
production router profile and never stops or restarts production services.
|
||||||
|
|
||||||
|
## Candidate
|
||||||
|
|
||||||
|
- Main model: `Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf` (about 17.6 GB)
|
||||||
|
- Vision projector: `mmproj-F16.gguf`
|
||||||
|
- Runtime: existing `mike-ai/llama.cpp:local`
|
||||||
|
- Test endpoint: `127.0.0.1:5004`
|
||||||
|
- Results: `/data/benchmarks/dirk-qwen38/`
|
||||||
|
|
||||||
|
The candidate is intended to reduce unnecessary reasoning and total token use;
|
||||||
|
it is not expected to improve raw decode speed. The production Qwen model is
|
||||||
|
therefore the mandatory A/B reference.
|
||||||
|
|
||||||
|
## Safety boundary
|
||||||
|
|
||||||
|
`run-case.sh` refuses to start while any production `mike-ai-llama-*` model
|
||||||
|
container is running. It does not stop production itself. The model server is
|
||||||
|
bound to loopback only and cannot be reached from the LAN.
|
||||||
|
|
||||||
|
## Prepared matrix
|
||||||
|
|
||||||
|
Run the following only after the GPUs have explicitly been declared free:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
./run-case.sh 80000 90,10 text
|
||||||
|
./run-case.sh 160000 90,10 text
|
||||||
|
./run-case.sh 160000 85,15 text
|
||||||
|
./run-case.sh 160000 80,20 text
|
||||||
|
./run-case.sh 192000 85,15 text
|
||||||
|
./run-case.sh 262144 80,20 text
|
||||||
|
```
|
||||||
|
|
||||||
|
The largest stable context is determined first. Vision is checked only after a
|
||||||
|
text winner exists:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
./run-case.sh 160000 85,15 vision
|
||||||
|
```
|
||||||
|
|
||||||
|
For every case, record uncached prefill, cached prefill, decode throughput,
|
||||||
|
GPU memory, context recall, tool calling, code quality and total tokens needed
|
||||||
|
to finish the task. Do not promote Dirk unless it matches the base model on
|
||||||
|
technical correctness and improves real Hermes task completion.
|
||||||
|
|
||||||
Executable
+28
@@ -0,0 +1,28 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
MODEL_DIR=${MODEL_DIR:-/data/models/qwen3.8-27b-dirk}
|
||||||
|
BASE_URL=https://huggingface.co/peculiar-ragdoll/Dirk-Qwen3.8-27B-GGUF/resolve/main
|
||||||
|
MODEL=Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf
|
||||||
|
PROJECTOR=mmproj-F16.gguf
|
||||||
|
|
||||||
|
install -d -m 0755 "$MODEL_DIR"
|
||||||
|
|
||||||
|
download() {
|
||||||
|
local name=$1
|
||||||
|
local target="$MODEL_DIR/$name"
|
||||||
|
curl --fail --location --retry 8 --retry-delay 5 --continue-at - \
|
||||||
|
--output "$target.part" "$BASE_URL/$name"
|
||||||
|
mv -f "$target.part" "$target"
|
||||||
|
}
|
||||||
|
|
||||||
|
[[ -s "$MODEL_DIR/$MODEL" ]] || download "$MODEL"
|
||||||
|
[[ -s "$MODEL_DIR/$PROJECTOR" ]] || download "$PROJECTOR"
|
||||||
|
|
||||||
|
(
|
||||||
|
cd "$MODEL_DIR"
|
||||||
|
sha256sum "$MODEL" "$PROJECTOR" > SHA256SUMS
|
||||||
|
)
|
||||||
|
|
||||||
|
printf 'Prepared model files in %s\n' "$MODEL_DIR"
|
||||||
|
|
||||||
Executable
+105
@@ -0,0 +1,105 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
CONTEXT=${1:-}
|
||||||
|
SPLIT=${2:-}
|
||||||
|
MODE=${3:-text}
|
||||||
|
|
||||||
|
if [[ -z $CONTEXT || -z $SPLIT || ! $CONTEXT =~ ^[0-9]+$ || ! $SPLIT =~ ^[0-9]+,[0-9]+$ ]]; then
|
||||||
|
echo "Usage: $0 CONTEXT TENSOR_SPLIT {text|vision}" >&2
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
if [[ $MODE != text && $MODE != vision ]]; then
|
||||||
|
echo "Mode must be text or vision" >&2
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
|
||||||
|
MODEL_DIR=${MODEL_DIR:-/data/models/qwen3.8-27b-dirk}
|
||||||
|
MODEL=/models/qwen3.8-27b-dirk/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf
|
||||||
|
PROJECTOR=/models/qwen3.8-27b-dirk/mmproj-F16.gguf
|
||||||
|
IMAGE=${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
|
||||||
|
NAME=mike-ai-llama-dirk-test
|
||||||
|
RESULT_DIR=${RESULT_DIR:-/data/benchmarks/dirk-qwen38}
|
||||||
|
|
||||||
|
[[ -s "$MODEL_DIR/Dirk-Qwen3.8-27B-UD-Q4_K_XL.gguf" ]] || {
|
||||||
|
echo "Dirk model is missing; run download-model.sh first" >&2
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
if [[ $MODE == vision && ! -s "$MODEL_DIR/mmproj-F16.gguf" ]]; then
|
||||||
|
echo "Vision projector is missing" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
mapfile -t blockers < <(
|
||||||
|
docker ps --format '{{.Names}}' |
|
||||||
|
grep -E '^mike-ai-llama-' |
|
||||||
|
grep -vE '^mike-ai-llama-dashboard$|^mike-ai-llama-dirk-test$' || true
|
||||||
|
)
|
||||||
|
if ((${#blockers[@]})); then
|
||||||
|
printf 'Refusing to start: production model container(s) still running: %s\n' "${blockers[*]}" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
docker rm -f "$NAME" >/dev/null 2>&1 || true
|
||||||
|
install -d -m 0755 "$RESULT_DIR"
|
||||||
|
|
||||||
|
args=(
|
||||||
|
--model "$MODEL"
|
||||||
|
--alias qwen-dirk-test
|
||||||
|
--ctx-size "$CONTEXT"
|
||||||
|
--flash-attn on
|
||||||
|
--cache-type-k q4_0
|
||||||
|
--cache-type-v q4_0
|
||||||
|
--cache-prompt
|
||||||
|
--cache-ram 8192
|
||||||
|
--threads 6
|
||||||
|
--threads-batch 6
|
||||||
|
--batch-size 64
|
||||||
|
--ubatch-size 32
|
||||||
|
--parallel 1
|
||||||
|
--jinja
|
||||||
|
--reasoning auto
|
||||||
|
--reasoning-budget 8192
|
||||||
|
--host 127.0.0.1
|
||||||
|
--port 5004
|
||||||
|
--metrics
|
||||||
|
--fit off
|
||||||
|
--n-gpu-layers all
|
||||||
|
--no-mmap
|
||||||
|
--temperature 1.0
|
||||||
|
--top-p 0.95
|
||||||
|
--top-k 20
|
||||||
|
--device CUDA0,CUDA1
|
||||||
|
--main-gpu 0
|
||||||
|
--split-mode layer
|
||||||
|
--tensor-split "$SPLIT"
|
||||||
|
--spec-type draft-mtp
|
||||||
|
--spec-draft-n-max 3
|
||||||
|
--spec-draft-type-k f16
|
||||||
|
--spec-draft-type-v f16
|
||||||
|
)
|
||||||
|
|
||||||
|
if [[ $MODE == vision ]]; then
|
||||||
|
args+=(--mmproj "$PROJECTOR" --no-mmproj-offload)
|
||||||
|
fi
|
||||||
|
|
||||||
|
label="ctx${CONTEXT}-split${SPLIT//,/-}-${MODE}"
|
||||||
|
docker run -d --rm \
|
||||||
|
--name "$NAME" \
|
||||||
|
--gpus all \
|
||||||
|
--network host \
|
||||||
|
--read-only \
|
||||||
|
--tmpfs /tmp:rw,noexec,nosuid,nodev,size=256m \
|
||||||
|
--security-opt no-new-privileges:true \
|
||||||
|
--cap-drop ALL \
|
||||||
|
--pids-limit 1024 \
|
||||||
|
--log-opt max-size=20m \
|
||||||
|
--log-opt max-file=2 \
|
||||||
|
-v /data/models:/models:ro \
|
||||||
|
-v "$RESULT_DIR":/results \
|
||||||
|
--label mike-ai.experiment=dirk-qwen38 \
|
||||||
|
--label mike-ai.case="$label" \
|
||||||
|
"$IMAGE" "${args[@]}"
|
||||||
|
|
||||||
|
printf 'Started isolated case %s on http://127.0.0.1:5004\n' "$label"
|
||||||
|
|
||||||
Executable
+5
@@ -0,0 +1,5 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
docker rm -f mike-ai-llama-dirk-test >/dev/null 2>&1 || true
|
||||||
|
echo "Dirk test container stopped. Production was not changed."
|
||||||
|
|
||||||
Executable
+14
@@ -0,0 +1,14 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
deadline=$((SECONDS + ${READY_TIMEOUT:-900}))
|
||||||
|
until curl -fsS --max-time 3 http://127.0.0.1:5004/health >/dev/null; do
|
||||||
|
if ((SECONDS >= deadline)); then
|
||||||
|
docker logs --tail 100 mike-ai-llama-dirk-test >&2 || true
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
curl -fsS http://127.0.0.1:5004/props
|
||||||
|
printf '\nDirk test server is ready.\n'
|
||||||
|
|
||||||
Reference in New Issue
Block a user