Document profile-wide GSQ-RCO A-B test

This commit is contained in:
Mikei386
2026-09-08 13:48:47 +02:00
parent edb845eb19
commit 3e3fbbe9bd
3 changed files with 193 additions and 0 deletions
+130
View File
@@ -0,0 +1,130 @@
#!/usr/bin/env bash
set -Eeuo pipefail
MODEL=${1:-}
CONTEXT=${2:-}
SPLIT=${3:-}
VISION=${4:-no}
MTP=${5:-3}
BATCH=${6:-2048}
UBATCH=${7:-128}
case "$MODEL" in
q4-pure)
MODEL_FILE=/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf
;;
q4-mix)
MODEL_FILE=/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
;;
iq3s)
MODEL_FILE=/models/qwen3.8-27b-gsq-rco-iq3s/Qwen3.8-27B-GSQ-RCO-IQ3_S-mtp.gguf
;;
*)
echo "Usage: $0 {q4-pure|q4-mix|iq3s} CONTEXT {none|PERCENT,PERCENT} {yes|no}" >&2
exit 2
;;
esac
[[ $CONTEXT =~ ^[0-9]+$ ]] || { echo "Invalid context" >&2; exit 2; }
[[ $SPLIT == none || $SPLIT =~ ^[0-9]+,[0-9]+$ ]] || { echo "Invalid split" >&2; exit 2; }
[[ $VISION == yes || $VISION == no ]] || { echo "Invalid vision setting" >&2; exit 2; }
[[ $MTP =~ ^[0-9]+$ ]] || { echo "Invalid MTP setting" >&2; exit 2; }
[[ $BATCH =~ ^[0-9]+$ ]] || { echo "Invalid batch setting" >&2; exit 2; }
[[ $UBATCH =~ ^[0-9]+$ ]] || { echo "Invalid ubatch setting" >&2; exit 2; }
NAME=mike-ai-llama-gsq-v2
IMAGE=${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
if docker ps --format '{{.Names}}' | grep -qx "$NAME"; then
echo "A/B container already running" >&2
exit 1
fi
mapfile -t blockers < <(
docker ps --format '{{.Names}}' |
grep -E '^mike-ai-llama-' |
grep -vE '^mike-ai-llama-dashboard$' || true
)
if ((${#blockers[@]})); then
printf 'Production model still running: %s\n' "${blockers[*]}" >&2
exit 1
fi
args=(
--model "$MODEL_FILE"
--alias benchmark
--ctx-size "$CONTEXT"
--flash-attn on
--cache-type-k q4_0
--cache-type-v q4_0
--cache-prompt
--cache-reuse 256
--cache-ram 8192
--threads 6
--threads-batch 6
--batch-size "$BATCH"
--ubatch-size "$UBATCH"
--parallel 1
--kv-unified
--jinja
--reasoning auto
--reasoning-preserve
--host 127.0.0.1
--port 5005
--metrics
--fit off
--n-gpu-layers all
--no-mmap
--no-ui
--temperature 0.2
--top-p 0.8
--top-k 20
--spec-type draft-mtp
--spec-draft-n-max "$MTP"
--spec-draft-type-k f16
--spec-draft-type-v f16
)
env_args=()
if [[ $VISION == yes ]]; then
env_args=(-e MTMD_BACKEND_DEVICE=CUDA1)
args+=(--mmproj /models/qwen/mmproj-BF16.gguf --mmproj-device CUDA1)
fi
if [[ $SPLIT == none ]]; then
args+=(--device CUDA0 --main-gpu 0 --split-mode none)
else
args+=(--device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split "$SPLIT")
fi
docker run -d \
--name "$NAME" \
--gpus all \
--network host \
--read-only \
--tmpfs /tmp:rw,noexec,nosuid,nodev,size=256m \
--security-opt no-new-privileges:true \
--cap-drop ALL \
--pids-limit 1024 \
--log-opt max-size=20m \
--log-opt max-file=2 \
-v /data/models:/models:ro \
"${env_args[@]}" \
--label mike-ai.experiment=gsq-rco-iq3s-ab-v2 \
"$IMAGE" "${args[@]}"
deadline=$((SECONDS + 900))
until curl -fsS --max-time 3 http://127.0.0.1:5005/health >/dev/null; do
if [[ $(docker inspect -f '{{.State.Running}}' "$NAME" 2>/dev/null || true) != true ]]; then
docker logs --tail 100 "$NAME" >&2 || true
exit 1
fi
if ((SECONDS >= deadline)); then
docker logs --tail 100 "$NAME" >&2 || true
exit 1
fi
sleep 2
done
curl -fsS http://127.0.0.1:5005/props
printf '\nReady: %s, context %s, split %s, vision %s\n' "$MODEL" "$CONTEXT" "$SPLIT" "$VISION"