#!/usr/bin/env bash set -Eeuo pipefail MODEL=${1:-} CONTEXT=${2:-} SPLIT=${3:-} VISION=${4:-no} MTP=${5:-3} BATCH=${6:-2048} UBATCH=${7:-128} case "$MODEL" in q4-pure) MODEL_FILE=/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf ;; q4-mix) MODEL_FILE=/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf ;; iq3s) MODEL_FILE=/models/qwen3.8-27b-gsq-rco-iq3s/Qwen3.8-27B-GSQ-RCO-IQ3_S-mtp.gguf ;; *) echo "Usage: $0 {q4-pure|q4-mix|iq3s} CONTEXT {none|PERCENT,PERCENT} {yes|no}" >&2 exit 2 ;; esac [[ $CONTEXT =~ ^[0-9]+$ ]] || { echo "Invalid context" >&2; exit 2; } [[ $SPLIT == none || $SPLIT =~ ^[0-9]+,[0-9]+$ ]] || { echo "Invalid split" >&2; exit 2; } [[ $VISION == yes || $VISION == no ]] || { echo "Invalid vision setting" >&2; exit 2; } [[ $MTP =~ ^[0-9]+$ ]] || { echo "Invalid MTP setting" >&2; exit 2; } [[ $BATCH =~ ^[0-9]+$ ]] || { echo "Invalid batch setting" >&2; exit 2; } [[ $UBATCH =~ ^[0-9]+$ ]] || { echo "Invalid ubatch setting" >&2; exit 2; } NAME=mike-ai-llama-gsq-v2 IMAGE=${LLAMA_IMAGE:-mike-ai/llama.cpp:local} if docker ps --format '{{.Names}}' | grep -qx "$NAME"; then echo "A/B container already running" >&2 exit 1 fi mapfile -t blockers < <( docker ps --format '{{.Names}}' | grep -E '^mike-ai-llama-' | grep -vE '^mike-ai-llama-dashboard$' || true ) if ((${#blockers[@]})); then printf 'Production model still running: %s\n' "${blockers[*]}" >&2 exit 1 fi args=( --model "$MODEL_FILE" --alias benchmark --ctx-size "$CONTEXT" --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-reuse 256 --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size "$BATCH" --ubatch-size "$UBATCH" --parallel 1 --kv-unified --jinja --reasoning auto --reasoning-preserve --host 127.0.0.1 --port 5005 --metrics --fit off --n-gpu-layers all --no-mmap --no-ui --temperature 0.2 --top-p 0.8 --top-k 20 --spec-type draft-mtp --spec-draft-n-max "$MTP" --spec-draft-type-k f16 --spec-draft-type-v f16 ) env_args=() if [[ $VISION == yes ]]; then env_args=(-e MTMD_BACKEND_DEVICE=CUDA1) args+=(--mmproj /models/qwen/mmproj-BF16.gguf --mmproj-device CUDA1) fi if [[ $SPLIT == none ]]; then args+=(--device CUDA0 --main-gpu 0 --split-mode none) else args+=(--device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split "$SPLIT") fi docker run -d \ --name "$NAME" \ --gpus all \ --network host \ --read-only \ --tmpfs /tmp:rw,noexec,nosuid,nodev,size=256m \ --security-opt no-new-privileges:true \ --cap-drop ALL \ --pids-limit 1024 \ --log-opt max-size=20m \ --log-opt max-file=2 \ -v /data/models:/models:ro \ "${env_args[@]}" \ --label mike-ai.experiment=gsq-rco-iq3s-ab-v2 \ "$IMAGE" "${args[@]}" deadline=$((SECONDS + 900)) until curl -fsS --max-time 3 http://127.0.0.1:5005/health >/dev/null; do if [[ $(docker inspect -f '{{.State.Running}}' "$NAME" 2>/dev/null || true) != true ]]; then docker logs --tail 100 "$NAME" >&2 || true exit 1 fi if ((SECONDS >= deadline)); then docker logs --tail 100 "$NAME" >&2 || true exit 1 fi sleep 2 done curl -fsS http://127.0.0.1:5005/props printf '\nReady: %s, context %s, split %s, vision %s\n' "$MODEL" "$CONTEXT" "$SPLIT" "$VISION"