Define four-profile production matrix with Medium default

This commit is contained in:
Mikei386
2026-08-22 11:30:38 +02:00
parent 977f8f5c76
commit 41b9c17f0f
33 changed files with 237 additions and 160 deletions
+1 -1
View File
@@ -40,7 +40,7 @@ for container in mike-ai-profile-controller mike-ai-router mike-ai-piper mike-ai
fi
done
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|long|ultra|experimental)$' || true)"
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|large|ultra|experimental)$' || true)"
if [[ $ACTIVE_LLAMA -eq 1 ]]; then
pass "exakt ein llama.cpp-Profil aktiv"
else
@@ -21,7 +21,7 @@ PORT = int(os.environ.get("CONTROLLER_PORT", "8090"))
SOCKET_PATH = os.environ.get("DOCKER_SOCKET", "/var/run/docker.sock")
TOKEN_FILE = os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token")
ALLOWED = tuple(x.strip() for x in os.environ.get(
"ALLOWED_PROFILES", "fast,medium,long,ultra,experimental").split(",") if x.strip())
"ALLOWED_PROFILES", "fast,medium,large,ultra,experimental").split(",") if x.strip())
LABEL_KEY = "com.mike-ai.llama-profile"
LOCK = threading.Lock()
log = logging.getLogger("profile-controller")
+3 -2
View File
@@ -21,7 +21,8 @@ install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf"
install -m 0644 "$PLATFORM/profiles/profile-large.conf" "$PROFILE_TARGET/profile-large.conf"
install -m 0644 "$PLATFORM/profiles/profile-ultra.conf" "$PROFILE_TARGET/profile-ultra.conf"
rsync -a --delete "$PLATFORM/mcp/" "$MCP_TARGET/"
rsync -a --delete --exclude searxng-settings.yml \
"$PLATFORM/web-search/" "$MCP_SEARCH_TARGET/"
@@ -39,7 +40,7 @@ Kernkonfiguration installiert, aber noch nicht gestartet.
Vor dem Start:
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
2. llama.cpp bauen und mit `llama-profile fast` starten.
2. llama.cpp bauen und mit `llama-profile medium` starten.
3. Fach-Secrets unter /etc/mike-ai ablegen.
4. Danach: /opt/mike-ai/mcp-containers/platform/mcp/install-tools.sh
EOF
+1 -1
View File
@@ -1,7 +1,7 @@
schema: 1
models:
qwen_fast_long:
role: primary-text-fast-and-long
role: primary-text-fast
source: "local migration from the reference host; public origin still to document"
file: Qwen3.8-27B-IQ4-MIX.gguf
target: /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
@@ -16,9 +16,10 @@ class Filter:
class Valves(BaseModel):
priority: int = 30
fast_context_tokens: int = 76800
medium_context_tokens: int = 94208
long_context_tokens: int = 131072
default_context_tokens: int = 76800
medium_context_tokens: int = 160000
large_context_tokens: int = 192000
ultra_context_tokens: int = 262144
default_context_tokens: int = 160000
soft_context_ratio: float = 0.70
hard_context_ratio: float = 0.84
reserved_output_tokens: int = 8192
@@ -107,8 +108,10 @@ class Filter:
def _context_limit(self, model: str) -> int:
model = (model or "").lower()
if "long" in model or "large" in model:
return self.valves.long_context_tokens
if "ultra" in model:
return self.valves.ultra_context_tokens
if "large" in model:
return self.valves.large_context_tokens
if "medium" in model:
return self.valves.medium_context_tokens
if "fast" in model:
+6
View File
@@ -0,0 +1,6 @@
[Unit]
Description=Legacy native Qwen Large 192K profile (Docker is the production path)
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
-6
View File
@@ -1,6 +0,0 @@
[Unit]
Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU with CPU Vision
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
+2 -2
View File
@@ -1,6 +1,6 @@
[Unit]
Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure with CPU Vision
Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
+6
View File
@@ -0,0 +1,6 @@
[Unit]
Description=Legacy native Qwen Ultra 256K profile (Docker is the production path)
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen-ultra --ctx-size 262144 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 80,20 --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
+2 -3
View File
@@ -7,10 +7,9 @@ SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
PROFILE="${1:-}"
case "$PROFILE" in
fast|medium|long) ;;
large) PROFILE=long ;;
fast|medium|large|ultra) ;;
*)
echo "Usage: llama-profile {fast|medium|long|large}" >&2
echo "Usage: llama-profile {fast|medium|large|ultra}" >&2
exit 2
;;
esac