Define four-profile production matrix with Medium default
This commit is contained in:
+9
-7
@@ -9,21 +9,23 @@ PIPER_TTS_VERSION=1.6.0
|
|||||||
PIPER_VOICE=de_DE-thorsten-high
|
PIPER_VOICE=de_DE-thorsten-high
|
||||||
AI_DNS=192.168.1.1
|
AI_DNS=192.168.1.1
|
||||||
|
|
||||||
FAST_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
MEDIUM_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
LONG_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
EXPERIMENTAL_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
EXPERIMENTAL_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||||
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
||||||
|
|
||||||
FAST_CONTEXT=76800
|
FAST_CONTEXT=76800
|
||||||
MEDIUM_CONTEXT=94208
|
MEDIUM_CONTEXT=160000
|
||||||
LONG_CONTEXT=131072
|
LARGE_CONTEXT=192000
|
||||||
ULTRA_CONTEXT=262144
|
ULTRA_CONTEXT=262144
|
||||||
EXPERIMENTAL_CONTEXT=76800
|
EXPERIMENTAL_CONTEXT=76800
|
||||||
FAST_GPU_DEVICES=0
|
FAST_GPU_DEVICES=0
|
||||||
MEDIUM_GPU_DEVICES=0
|
MEDIUM_GPU_DEVICES=0,1
|
||||||
LONG_GPU_DEVICES=0
|
MEDIUM_TENSOR_SPLIT=90,10
|
||||||
|
LARGE_GPU_DEVICES=0,1
|
||||||
|
LARGE_TENSOR_SPLIT=86,14
|
||||||
ULTRA_GPU_DEVICES=0,1
|
ULTRA_GPU_DEVICES=0,1
|
||||||
ULTRA_TENSOR_SPLIT=80,20
|
ULTRA_TENSOR_SPLIT=80,20
|
||||||
EXPERIMENTAL_GPU_DEVICES=0
|
EXPERIMENTAL_GPU_DEVICES=0
|
||||||
|
|||||||
@@ -10,7 +10,9 @@ WireGuard-Isolation.
|
|||||||
- llama.cpp selbst gebaut und auf einen geprüften Commit festgelegt
|
- llama.cpp selbst gebaut und auf einen geprüften Commit festgelegt
|
||||||
- vier schaltbare Profilcontainer plus ein isolierter Experimentalcontainer;
|
- vier schaltbare Profilcontainer plus ein isolierter Experimentalcontainer;
|
||||||
davon ist immer exakt ein Inferenzcontainer aktiv
|
davon ist immer exakt ein Inferenzcontainer aktiv
|
||||||
- `/fast`, `/medium`, `/long` und `/ultra` über den Profile Router
|
- `/fast`, `/medium`, `/large` und `/ultra` über den Profile Router
|
||||||
|
- verbindliche Standardmatrix: Fast MIX 76,8K, Medium Pure 160K (Default),
|
||||||
|
Large Pure 192K und Ultra Pure 256K
|
||||||
- `/ultra`: getestetes text-only 256K-Profil (IQ4_XS Pure, beide GPUs,
|
- `/ultra`: getestetes text-only 256K-Profil (IQ4_XS Pure, beide GPUs,
|
||||||
80:20); etwa 68 Token/s und erfolgreicher 220K-Prompt-Fülltest
|
80:20); etwa 68 Token/s und erfolgreicher 220K-Prompt-Fülltest
|
||||||
- Open WebUI als einzige normale Oberfläche
|
- Open WebUI als einzige normale Oberfläche
|
||||||
@@ -21,6 +23,9 @@ WireGuard-Isolation.
|
|||||||
- KI-Ausgangsverkehr über das Heimnetz, bei Tunnelausfall fail-closed
|
- KI-Ausgangsverkehr über das Heimnetz, bei Tunnelausfall fail-closed
|
||||||
- keine Secrets, Chats, Logs oder Modelldateien im Repository
|
- keine Secrets, Chats, Logs oder Modelldateien im Repository
|
||||||
|
|
||||||
|
Die gemessenen Startparameter und Zuständigkeiten stehen in
|
||||||
|
[`docs/STANDARD_PROFILE_MATRIX.md`](docs/STANDARD_PROFILE_MATRIX.md).
|
||||||
|
|
||||||
## Schnellstart
|
## Schnellstart
|
||||||
|
|
||||||
Auf einem frisch installierten Debian 12/13 amd64:
|
Auf einem frisch installierten Debian 12/13 amd64:
|
||||||
@@ -94,6 +99,6 @@ router/ OpenAI-kompatibler Profile Router
|
|||||||
- Ein Blackhole-Fallback verhindert Traffic-Leaks bei WireGuard-Ausfall.
|
- Ein Blackhole-Fallback verhindert Traffic-Leaks bei WireGuard-Ausfall.
|
||||||
- Das Uni-Netz und das Heimnetz dürfen diesen Host nicht als Transit benutzen.
|
- Das Uni-Netz und das Heimnetz dürfen diesen Host nicht als Transit benutzen.
|
||||||
|
|
||||||
Die Profilwerte sind Ausgangswerte. Nach dem Neuaufbau werden RTX 5080 und
|
Die Profilwerte wurden auf RTX 5080 und RTX 3060 vermessen und bilden die
|
||||||
RTX 3060 mit der bestehenden Standard-Testserie neu vermessen, bevor die zweite
|
verbindliche Standardmatrix. Neue Varianten ersetzen sie erst nach demselben
|
||||||
GPU in ein Produktionsprofil einfließt.
|
Vergleichstest und einer dokumentierten Entscheidung.
|
||||||
|
|||||||
+35
-20
@@ -100,7 +100,7 @@ services:
|
|||||||
labels:
|
labels:
|
||||||
com.mike-ai.llama-profile: medium
|
com.mike-ai.llama-profile: medium
|
||||||
environment:
|
environment:
|
||||||
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0}
|
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
|
||||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
command:
|
command:
|
||||||
- --model
|
- --model
|
||||||
@@ -111,7 +111,7 @@ services:
|
|||||||
- --alias
|
- --alias
|
||||||
- qwen-medium
|
- qwen-medium
|
||||||
- --ctx-size
|
- --ctx-size
|
||||||
- "${MEDIUM_CONTEXT:-94208}"
|
- "${MEDIUM_CONTEXT:-160000}"
|
||||||
- --flash-attn
|
- --flash-attn
|
||||||
- "on"
|
- "on"
|
||||||
- --cache-type-k
|
- --cache-type-k
|
||||||
@@ -149,28 +149,40 @@ services:
|
|||||||
- --top-k
|
- --top-k
|
||||||
- "20"
|
- "20"
|
||||||
- --device
|
- --device
|
||||||
- CUDA0
|
- CUDA0,CUDA1
|
||||||
|
- --main-gpu
|
||||||
|
- "0"
|
||||||
- --split-mode
|
- --split-mode
|
||||||
- none
|
- layer
|
||||||
|
- --tensor-split
|
||||||
|
- "${MEDIUM_TENSOR_SPLIT:-90,10}"
|
||||||
|
- --spec-type
|
||||||
|
- draft-mtp
|
||||||
|
- --spec-draft-n-max
|
||||||
|
- "3"
|
||||||
|
- --spec-draft-type-k
|
||||||
|
- f16
|
||||||
|
- --spec-draft-type-v
|
||||||
|
- f16
|
||||||
|
|
||||||
llama-long:
|
llama-large:
|
||||||
<<: *llama-common
|
<<: *llama-common
|
||||||
container_name: mike-ai-llama-long
|
container_name: mike-ai-llama-large
|
||||||
labels:
|
labels:
|
||||||
com.mike-ai.llama-profile: long
|
com.mike-ai.llama-profile: large
|
||||||
environment:
|
environment:
|
||||||
NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0}
|
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
|
||||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
command:
|
command:
|
||||||
- --model
|
- --model
|
||||||
- "/models/${LONG_MODEL_FILE:?LONG_MODEL_FILE is required}"
|
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
|
||||||
- --mmproj
|
- --mmproj
|
||||||
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
||||||
- --no-mmproj-offload
|
- --no-mmproj-offload
|
||||||
- --alias
|
- --alias
|
||||||
- qwen-long
|
- qwen-large
|
||||||
- --ctx-size
|
- --ctx-size
|
||||||
- "${LONG_CONTEXT:-131072}"
|
- "${LARGE_CONTEXT:-192000}"
|
||||||
- --flash-attn
|
- --flash-attn
|
||||||
- "on"
|
- "on"
|
||||||
- --cache-type-k
|
- --cache-type-k
|
||||||
@@ -199,8 +211,6 @@ services:
|
|||||||
- "off"
|
- "off"
|
||||||
- --n-gpu-layers
|
- --n-gpu-layers
|
||||||
- all
|
- all
|
||||||
- --override-tensor
|
|
||||||
- blk.([0-9]|1[0-1]).ffn_.*=CPU
|
|
||||||
- --no-mmap
|
- --no-mmap
|
||||||
- --no-ui
|
- --no-ui
|
||||||
- --temperature
|
- --temperature
|
||||||
@@ -210,19 +220,23 @@ services:
|
|||||||
- --top-k
|
- --top-k
|
||||||
- "20"
|
- "20"
|
||||||
- --device
|
- --device
|
||||||
- CUDA0
|
- CUDA0,CUDA1
|
||||||
|
- --main-gpu
|
||||||
|
- "0"
|
||||||
- --split-mode
|
- --split-mode
|
||||||
- none
|
- layer
|
||||||
|
- --tensor-split
|
||||||
|
- "${LARGE_TENSOR_SPLIT:-86,14}"
|
||||||
- --spec-type
|
- --spec-type
|
||||||
- draft-mtp
|
- draft-mtp
|
||||||
- --spec-draft-n-max
|
- --spec-draft-n-max
|
||||||
- "2"
|
- "3"
|
||||||
- --spec-draft-type-k
|
- --spec-draft-type-k
|
||||||
- q4_0
|
- f16
|
||||||
- --spec-draft-type-v
|
- --spec-draft-type-v
|
||||||
- q4_0
|
- f16
|
||||||
|
|
||||||
# Text-only long-context profile. This exact IQ4_XS-pure / 256K / 80:20
|
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
|
||||||
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
|
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
|
||||||
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
|
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
|
||||||
llama-ultra:
|
llama-ultra:
|
||||||
@@ -347,7 +361,7 @@ services:
|
|||||||
- /var/run/docker.sock:/var/run/docker.sock
|
- /var/run/docker.sock:/var/run/docker.sock
|
||||||
environment:
|
environment:
|
||||||
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||||
ALLOWED_PROFILES: fast,medium,long,ultra,experimental
|
ALLOWED_PROFILES: fast,medium,large,ultra,experimental
|
||||||
networks: [control]
|
networks: [control]
|
||||||
security_opt: ["no-new-privileges:true"]
|
security_opt: ["no-new-privileges:true"]
|
||||||
healthcheck:
|
healthcheck:
|
||||||
@@ -456,6 +470,7 @@ services:
|
|||||||
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
|
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
|
||||||
environment:
|
environment:
|
||||||
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
|
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
|
||||||
|
DEFAULT_MODELS: qwen-medium
|
||||||
OLLAMA_BASE_URL: ""
|
OLLAMA_BASE_URL: ""
|
||||||
OPENAI_API_BASE_URLS: http://router:8081/v1
|
OPENAI_API_BASE_URLS: http://router:8081/v1
|
||||||
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
||||||
|
|||||||
+14
-12
@@ -28,15 +28,15 @@ WG_DNS=192.168.1.1
|
|||||||
WG_ROUTE_AI_INTERNET=true
|
WG_ROUTE_AI_INTERNET=true
|
||||||
|
|
||||||
# Exact model artifacts. Never put access tokens in these URLs.
|
# Exact model artifacts. Never put access tokens in these URLs.
|
||||||
FAST_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
FAST_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf
|
FAST_MODEL_URL=https://huggingface.co/vmarcelo/Qwen3.8-27B-MIX_GGUF/resolve/main/Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
FAST_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199
|
FAST_MODEL_SHA256=54879ae8738d5938f46cb3b8cbf16bf42b8c85b7d68d7c73f062b612ec183e36
|
||||||
MEDIUM_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
MEDIUM_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf
|
MEDIUM_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
MEDIUM_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199
|
MEDIUM_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
|
||||||
LONG_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
LONG_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf
|
LARGE_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
LONG_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199
|
LARGE_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
|
||||||
ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
ULTRA_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
|
ULTRA_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
ULTRA_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
|
ULTRA_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
|
||||||
@@ -46,12 +46,14 @@ EXPERIMENTAL_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd74
|
|||||||
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
||||||
VISION_PROJECTOR_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/mmproj-BF16.gguf
|
VISION_PROJECTOR_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/mmproj-BF16.gguf
|
||||||
VISION_PROJECTOR_SHA256=83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
VISION_PROJECTOR_SHA256=83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||||
# FAST/LONG use the MTP tensor embedded in the IQ4-MIX GGUF. A separate
|
# All standard profiles use the MTP tensor embedded in their GGUF. A separate
|
||||||
# draft-model artifact is neither downloaded nor passed to llama-server.
|
# draft-model artifact is neither downloaded nor passed to llama-server.
|
||||||
|
|
||||||
FAST_CONTEXT=76800
|
FAST_CONTEXT=76800
|
||||||
MEDIUM_CONTEXT=94208
|
MEDIUM_CONTEXT=160000
|
||||||
LONG_CONTEXT=131072
|
MEDIUM_TENSOR_SPLIT=90,10
|
||||||
|
LARGE_CONTEXT=192000
|
||||||
|
LARGE_TENSOR_SPLIT=86,14
|
||||||
ULTRA_CONTEXT=262144
|
ULTRA_CONTEXT=262144
|
||||||
ULTRA_TENSOR_SPLIT=80,20
|
ULTRA_TENSOR_SPLIT=80,20
|
||||||
EXPERIMENTAL_CONTEXT=76800
|
EXPERIMENTAL_CONTEXT=76800
|
||||||
|
|||||||
@@ -5,8 +5,8 @@ set -e
|
|||||||
D="$(cd "$(dirname "$0")" && pwd)/fake-profile-dir"
|
D="$(cd "$(dirname "$0")" && pwd)/fake-profile-dir"
|
||||||
|
|
||||||
case "${1:-}" in
|
case "${1:-}" in
|
||||||
fast|medium|long) ;;
|
fast|medium|large|ultra) ;;
|
||||||
*) echo "Usage: fake-llama-profile {fast|medium|long}"; exit 1 ;;
|
*) echo "Usage: fake-llama-profile {fast|medium|large|ultra}"; exit 1 ;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
if [ -f /tmp/fake-profile-fail ] \
|
if [ -f /tmp/fake-profile-fail ] \
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
ExecStart=/usr/bin/mock-llama --ctx-size 192000
|
||||||
@@ -1 +0,0 @@
|
|||||||
ExecStart=/usr/bin/mock-llama --ctx-size 131072
|
|
||||||
@@ -1 +1 @@
|
|||||||
ExecStart=/usr/bin/mock-llama --ctx-size 94208
|
ExecStart=/usr/bin/mock-llama --ctx-size 160000
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
ExecStart=/usr/bin/mock-llama --ctx-size 262144
|
||||||
@@ -19,7 +19,7 @@ ROUTER_STATUS=http://127.0.0.1:8081/status
|
|||||||
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
||||||
# vergleichen (robuster als String-Matching).
|
# vergleichen (robuster als String-Matching).
|
||||||
PROFILE=""
|
PROFILE=""
|
||||||
for p in fast medium long; do
|
for p in fast medium large ultra; do
|
||||||
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
||||||
PROFILE=$p
|
PROFILE=$p
|
||||||
break
|
break
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ ROUTER_STATUS=http://127.0.0.1:8081/status
|
|||||||
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
||||||
# vergleichen (robuster als String-Matching).
|
# vergleichen (robuster als String-Matching).
|
||||||
PROFILE=""
|
PROFILE=""
|
||||||
for p in fast medium long; do
|
for p in fast medium large ultra; do
|
||||||
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
||||||
PROFILE=$p
|
PROFILE=$p
|
||||||
break
|
break
|
||||||
|
|||||||
+25
-24
@@ -123,11 +123,12 @@ echo "$RESP" | python3 -c '
|
|||||||
import json,sys
|
import json,sys
|
||||||
d=json.load(sys.stdin)
|
d=json.load(sys.stdin)
|
||||||
ids={m["id"]:m for m in d["data"]}
|
ids={m["id"]:m for m in d["data"]}
|
||||||
assert set(ids)=={"qwen-fast","qwen-medium","qwen-long"}, ids
|
assert set(ids)=={"qwen-fast","qwen-medium","qwen-large","qwen-ultra"}, ids
|
||||||
assert ids["qwen-fast"]["context_length"]==76800
|
assert ids["qwen-fast"]["context_length"]==76800
|
||||||
assert ids["qwen-medium"]["context_length"]==94208
|
assert ids["qwen-medium"]["context_length"]==160000
|
||||||
assert ids["qwen-long"]["context_length"]==131072
|
assert ids["qwen-large"]["context_length"]==192000
|
||||||
' && ok "drei virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
|
assert ids["qwen-ultra"]["context_length"]==262144
|
||||||
|
' && ok "vier virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
|
||||||
|
|
||||||
# --- 2. /status -----------------------------------------------------------------
|
# --- 2. /status -----------------------------------------------------------------
|
||||||
echo "== Test 2: /status"
|
echo "== Test 2: /status"
|
||||||
@@ -183,14 +184,14 @@ echo "$RESP" | python3 -m json.tool
|
|||||||
echo "$RESP" | python3 -c '
|
echo "$RESP" | python3 -c '
|
||||||
import json,sys
|
import json,sys
|
||||||
d=json.load(sys.stdin)
|
d=json.load(sys.stdin)
|
||||||
assert d["profile"]=="medium" and d["context_length"]==94208, d
|
assert d["profile"]=="medium" and d["context_length"]==160000, d
|
||||||
' && ok "Profil medium aktiv" || bad "Profilwechsel medium"
|
' && ok "Profil medium aktiv" || bad "Profilwechsel medium"
|
||||||
curl -sf "$BASE/status" | python3 -c '
|
curl -sf "$BASE/status" | python3 -c '
|
||||||
import json,sys
|
import json,sys
|
||||||
d=json.load(sys.stdin)
|
d=json.load(sys.stdin)
|
||||||
assert d["current_profile"]=="medium", d
|
assert d["current_profile"]=="medium", d
|
||||||
assert d["upstream"]["ctx"]==94208, d
|
assert d["upstream"]["ctx"]==160000, d
|
||||||
' && ok "Status bestätigt medium (ctx 94208)" || bad "Status nach Wechsel"
|
' && ok "Status bestätigt medium (ctx 160000)" || bad "Status nach Wechsel"
|
||||||
|
|
||||||
# --- 7. Profilwechsel medium -> fast --------------------------------------------------
|
# --- 7. Profilwechsel medium -> fast --------------------------------------------------
|
||||||
echo "== Test 7: Profilwechsel medium -> fast"
|
echo "== Test 7: Profilwechsel medium -> fast"
|
||||||
@@ -203,21 +204,21 @@ assert d["profile"]=="fast" and d["context_length"]==76800, d
|
|||||||
' && ok "Profil fast wieder aktiv" || bad "Profilwechsel fast"
|
' && ok "Profil fast wieder aktiv" || bad "Profilwechsel fast"
|
||||||
|
|
||||||
# --- 8. Virtuelles Modell triggert Profilwechsel ----------------------------------------
|
# --- 8. Virtuelles Modell triggert Profilwechsel ----------------------------------------
|
||||||
echo "== Test 8: Chat mit qwen-long triggert Wechsel auf long"
|
echo "== Test 8: Chat mit qwen-large triggert Wechsel auf large"
|
||||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||||
-d '{"model":"qwen-long","messages":[{"role":"user","content":"Hallo"}]}')
|
-d '{"model":"qwen-large","messages":[{"role":"user","content":"Hallo"}]}')
|
||||||
echo "$RESP" | python3 -m json.tool
|
echo "$RESP" | python3 -m json.tool
|
||||||
echo "$RESP" | python3 -c '
|
echo "$RESP" | python3 -c '
|
||||||
import json,sys
|
import json,sys
|
||||||
d=json.load(sys.stdin)
|
d=json.load(sys.stdin)
|
||||||
assert d["model"]=="mock-model-131072", d
|
assert d["model"]=="mock-model-192000", d
|
||||||
' && ok "qwen-long hat Profil long aktiviert und weitergeleitet" || bad "virtuelles Modell"
|
' && ok "qwen-large hat Profil large aktiviert und weitergeleitet" || bad "virtuelles Modell"
|
||||||
|
|
||||||
# --- 9. Methoden und ungültiges virtuelles Modell -----------------------------------------
|
# --- 9. Methoden und ungültiges virtuelles Modell -----------------------------------------
|
||||||
echo "== Test 9: sichere Profilmethoden + ungültiges virtuelles Modell"
|
echo "== Test 9: sichere Profilmethoden + ungültiges virtuelles Modell"
|
||||||
CODE=$(curl -s -o /tmp/err9.json -w "%{http_code}" "$BASE/long")
|
CODE=$(curl -s -o /tmp/err9.json -w "%{http_code}" "$BASE/large")
|
||||||
cat /tmp/err9.json; echo
|
cat /tmp/err9.json; echo
|
||||||
[ "$CODE" = "405" ] && ok "GET /long verändert kein Profil" || bad "erwartet 405, bekam $CODE"
|
[ "$CODE" = "405" ] && ok "GET /large verändert kein Profil" || bad "erwartet 405, bekam $CODE"
|
||||||
|
|
||||||
CODE=$(curl -s -o /tmp/err9b.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
|
CODE=$(curl -s -o /tmp/err9b.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
|
||||||
-H "Content-Type: application/json" -d '{"model":"qwen-huge","messages":[]}')
|
-H "Content-Type: application/json" -d '{"model":"qwen-huge","messages":[]}')
|
||||||
@@ -243,7 +244,7 @@ CODE=$(curl -s -o /tmp/chat9e.json -w "%{http_code}" "$BASE/v1/chat/completions"
|
|||||||
|
|
||||||
# --- 10. llama.cpp down -> 502, danach Recovery ---------------------------------------------------
|
# --- 10. llama.cpp down -> 502, danach Recovery ---------------------------------------------------
|
||||||
echo "== Test 10: Upstream down -> 502, danach Recovery"
|
echo "== Test 10: Upstream down -> 502, danach Recovery"
|
||||||
# Profil auf fast setzen (aus Test 8 ist long aktiv)
|
# Profil auf fast setzen (aus Test 8 ist large aktiv)
|
||||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||||
# Mock stoppen (simuliert Crash) – über Fake-systemctl
|
# Mock stoppen (simuliert Crash) – über Fake-systemctl
|
||||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
|
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||||
@@ -423,14 +424,14 @@ sleep 0.5
|
|||||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||||
[ "$PROFILE" = "medium" ] && ok "Medium → Image → Medium" || bad "Profil nach Image: $PROFILE (erwartet medium)"
|
[ "$PROFILE" = "medium" ] && ok "Medium → Image → Medium" || bad "Profil nach Image: $PROFILE (erwartet medium)"
|
||||||
|
|
||||||
# --- 24. Long → Image → Long ------------------------------------------------------------------------
|
# --- 24. Large → Image → Large ----------------------------------------------------------------------
|
||||||
echo "== Test 24: Long → Image → Long"
|
echo "== Test 24: Large → Image → Large"
|
||||||
curl -sf -X POST "$BASE/long" >/dev/null
|
curl -sf -X POST "$BASE/large" >/dev/null
|
||||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||||
-d '{"prompt":"long test","size":"1024x1024"}')
|
-d '{"prompt":"large test","size":"1024x1024"}')
|
||||||
sleep 0.5
|
sleep 0.5
|
||||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||||
[ "$PROFILE" = "long" ] && ok "Long → Image → Long" || bad "Profil nach Image: $PROFILE (erwartet long)"
|
[ "$PROFILE" = "large" ] && ok "Large → Image → Large" || bad "Profil nach Image: $PROFILE (erwartet large)"
|
||||||
|
|
||||||
# --- 25. /status während Image-Job -------------------------------------------------------------------
|
# --- 25. /status während Image-Job -------------------------------------------------------------------
|
||||||
echo "== Test 25: /status während Image-Job"
|
echo "== Test 25: /status während Image-Job"
|
||||||
@@ -702,8 +703,8 @@ wait $STT_PID43
|
|||||||
# --- 44. Zwei konkurrierende Profilanfragen ------------------------------------------------
|
# --- 44. Zwei konkurrierende Profilanfragen ------------------------------------------------
|
||||||
echo "== Test 44: Profil-Lease verhindert Wechsel während eines Chats"
|
echo "== Test 44: Profil-Lease verhindert Wechsel während eines Chats"
|
||||||
curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||||
-d '{"model":"qwen-long","mock_delay":1.0,"messages":[{"role":"user","content":"Lang"}]}' \
|
-d '{"model":"qwen-large","mock_delay":1.0,"messages":[{"role":"user","content":"Groß"}]}' \
|
||||||
>/tmp/chat44-long.json &
|
>/tmp/chat44-large.json &
|
||||||
CHAT44_PID=$!
|
CHAT44_PID=$!
|
||||||
sleep 0.2
|
sleep 0.2
|
||||||
curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||||
@@ -712,10 +713,10 @@ curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
|||||||
wait "$CHAT44_PID"
|
wait "$CHAT44_PID"
|
||||||
python3 -c '
|
python3 -c '
|
||||||
import json
|
import json
|
||||||
long=json.load(open("/tmp/chat44-long.json"))
|
large=json.load(open("/tmp/chat44-large.json"))
|
||||||
medium=json.load(open("/tmp/chat44-medium.json"))
|
medium=json.load(open("/tmp/chat44-medium.json"))
|
||||||
assert long["mock_ctx"] == 131072, long
|
assert large["mock_ctx"] == 192000, large
|
||||||
assert medium["mock_ctx"] == 94208, medium
|
assert medium["mock_ctx"] == 160000, medium
|
||||||
' && ok "konkurrierende Chats behielten jeweils ihr Profil" \
|
' && ok "konkurrierende Chats behielten jeweils ihr Profil" \
|
||||||
|| bad "Profil-Lease bei konkurrierenden Chats"
|
|| bad "Profil-Lease bei konkurrierenden Chats"
|
||||||
|
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ Heimnetz / VPN-Clients
|
|||||||
+-- Profile Controller -- Docker Socket (feste Allowlist)
|
+-- Profile Controller -- Docker Socket (feste Allowlist)
|
||||||
+-- llama-fast --\
|
+-- llama-fast --\
|
||||||
+-- llama-medium > exakt einer aktiv
|
+-- llama-medium > exakt einer aktiv
|
||||||
+-- llama-long --/
|
+-- llama-large --/
|
||||||
+-- llama-experimental
|
+-- llama-experimental
|
||||||
+-- llama-ultra (256K, text-only, dual GPU)
|
+-- llama-ultra (256K, text-only, dual GPU)
|
||||||
+-- Piper-TTS (CPU, nur intern)
|
+-- Piper-TTS (CPU, nur intern)
|
||||||
@@ -44,8 +44,8 @@ Heimnetz / VPN-Clients
|
|||||||
| SearXNG/TinySearch | nein | Suchbackend des Web-MCP |
|
| SearXNG/TinySearch | nein | Suchbackend des Web-MCP |
|
||||||
|
|
||||||
Nur der Profile Controller erhält den Docker-Socket. Der Router erhält weder
|
Nur der Profile Controller erhält den Docker-Socket. Der Router erhält weder
|
||||||
Socket noch Shell-Zugriff und kann dem Controller nur `fast`, `medium`, `long`
|
Socket noch Shell-Zugriff und kann dem Controller nur `fast`, `medium`, `large`,
|
||||||
oder `experimental` übergeben. Die llama-Container laufen ohne UI,
|
`ultra` oder `experimental` übergeben. Die llama-Container laufen ohne UI,
|
||||||
Capabilities und Schreibzugriff auf die Modelldateien.
|
Capabilities und Schreibzugriff auf die Modelldateien.
|
||||||
|
|
||||||
## Profilprinzip
|
## Profilprinzip
|
||||||
@@ -60,13 +60,13 @@ halten.
|
|||||||
| Profil | Ausgangswert | Zweck |
|
| Profil | Ausgangswert | Zweck |
|
||||||
|---|---:|---|
|
|---|---:|---|
|
||||||
| fast | 76.800 Kontext, MTP | mindestens ungefähr 80 Token/s anstreben |
|
| fast | 76.800 Kontext, MTP | mindestens ungefähr 80 Token/s anstreben |
|
||||||
| medium | 94.208 Kontext | mehr Kontext ohne CPU-FFN-Offload |
|
| medium | 160.000 Kontext, Pure, 90:10, MTP3 | Standardprofil |
|
||||||
| long | 131.072 Kontext | maximale Nutzbarkeit, CPU-Offload erlaubt |
|
| large | 192.000 Kontext, Pure, 86:14, MTP3 | große Agenten-/MCP-Sitzungen |
|
||||||
|
| ultra | 262.144 Kontext, Pure, 80:20, MTP2 | maximaler Textkontext |
|
||||||
| experimental | 76.800 Kontext | isolierte Tests ohne Produktion zu ändern |
|
| experimental | 76.800 Kontext | isolierte Tests ohne Produktion zu ändern |
|
||||||
|
|
||||||
Diese Werte sind reproduzierbare Startwerte, keine Garantie. Nach Einbau der
|
Diese Matrix wurde auf RTX 5080 und RTX 3060 gemessen und ist bis zu einer
|
||||||
RTX 3060 werden sie auf dem Zielhost erneut gemessen. Die zweite Karte wird
|
bewussten Neubewertung der verbindliche Produktionsstandard.
|
||||||
nicht automatisch in die Produktionsprofile aufgenommen.
|
|
||||||
|
|
||||||
## Netzwerk
|
## Netzwerk
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,9 @@
|
|||||||
# Athena-Leerhostaufbau – Praxisprotokoll
|
# Athena-Leerhostaufbau – Praxisprotokoll
|
||||||
|
|
||||||
|
> Dieses Dokument ist ein chronologisches Aufbauprotokoll. Darin genannte alte
|
||||||
|
> Profilnamen und Kontextwerte sind keine aktuelle Konfiguration. Seit dem
|
||||||
|
> 22. August 2026 gilt `STANDARD_PROFILE_MATRIX.md`.
|
||||||
|
|
||||||
Dieses Protokoll hält die Abweichungen fest, die beim realen Neuaufbau auf
|
Dieses Protokoll hält die Abweichungen fest, die beim realen Neuaufbau auf
|
||||||
einem frischen Debian-13-Host sichtbar wurden. Jede dauerhaft notwendige
|
einem frischen Debian-13-Host sichtbar wurden. Jede dauerhaft notwendige
|
||||||
Korrektur muss zusätzlich im Installer, Restore-Skript oder in der regulären
|
Korrektur muss zusätzlich im Installer, Restore-Skript oder in der regulären
|
||||||
|
|||||||
+23
-24
@@ -1,8 +1,8 @@
|
|||||||
# Aktueller produktiver Referenzstand
|
# Aktueller produktiver Referenzstand
|
||||||
|
|
||||||
Stand: 20. August 2026. Dieses Dokument beschreibt die funktionierende
|
Stand: 22. August 2026. Dieses Dokument beschreibt die auf Athena installierte
|
||||||
Referenz vor dem geplanten Neuaufbau. Es ist keine Empfehlung, jede Altlast des
|
und geprüfte Docker-Referenz. Die verbindlichen Profilparameter stehen in
|
||||||
Hosts zu übernehmen.
|
`STANDARD_PROFILE_MATRIX.md`.
|
||||||
|
|
||||||
## Hardware und Betriebssystem
|
## Hardware und Betriebssystem
|
||||||
|
|
||||||
@@ -12,10 +12,10 @@ Hosts zu übernehmen.
|
|||||||
| Kernel | 6.12.101+deb13-amd64 |
|
| Kernel | 6.12.101+deb13-amd64 |
|
||||||
| CPU | AMD Ryzen 5 5600, 6 Kerne/12 Threads |
|
| CPU | AMD Ryzen 5 5600, 6 Kerne/12 Threads |
|
||||||
| RAM | 48 GiB DDR4-2666 |
|
| RAM | 48 GiB DDR4-2666 |
|
||||||
| GPU | NVIDIA GeForce RTX 5080, 16 GiB VRAM |
|
| GPUs | NVIDIA GeForce RTX 5080, 16 GiB + RTX 3060, 12 GiB VRAM |
|
||||||
| NVIDIA-Treiber | 610.57.04 |
|
| NVIDIA-Treiber | 610.57.04 |
|
||||||
| System-SSD | Samsung 980 PRO 1 TB |
|
| System-SSD | Samsung 980 PRO 1 TB |
|
||||||
| Daten-SSD | Samsung 980 PRO 2 TB |
|
| Daten-SSD | WD Blue SN580 1 TB, unter `/data` |
|
||||||
|
|
||||||
Die früher verwendete Radeon RX 470 ist ausgebaut und gehört nicht zur
|
Die früher verwendete Radeon RX 470 ist ausgebaut und gehört nicht zur
|
||||||
Zielplattform.
|
Zielplattform.
|
||||||
@@ -28,23 +28,22 @@ Zielplattform.
|
|||||||
| Repository | `https://github.com/ggml-org/llama.cpp.git` |
|
| Repository | `https://github.com/ggml-org/llama.cpp.git` |
|
||||||
| Commit | `4df29be4f4c3673f428170fda944a5b19f743bb8` |
|
| Commit | `4df29be4f4c3673f428170fda944a5b19f743bb8` |
|
||||||
| Compiler | GCC 14.2 |
|
| Compiler | GCC 14.2 |
|
||||||
| Hauptdienst | `mike-ai-llama-ui.service` |
|
| Hauptdienst | jeweils ein Container `mike-ai-llama-<profil>` |
|
||||||
| llama.cpp-Port | 8080, auf dem alten Host noch im LAN gebunden |
|
| llama.cpp-Port | 8080, ausschließlich im internen Docker-Netz |
|
||||||
| Client-Port | 8081 über den Router |
|
| Client-Port | 8081 über den Router |
|
||||||
| MCP-Konfiguration | getrennte Container unter `/opt/mike-ai/mcp-containers` |
|
| MCP-Konfiguration | getrennte Container unter `/opt/mike-ai/mcp-containers` |
|
||||||
|
|
||||||
### Aktives Fast-Profil
|
### Aktives Standardprofil
|
||||||
|
|
||||||
- Qwen3.8-27B IQ4-MIX
|
- Qwen3.8-27B IQ4_XS Pure
|
||||||
- Dateigröße: 14.111.614.400 Bytes
|
- Dateigröße: 14.534.384.640 Bytes
|
||||||
- SHA256: `54879ae8738d5938f46cb3b8cbf16bf42b8c85b7d68d7c73f062b612ec183e36`
|
- SHA256: `ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675`
|
||||||
- Öffentliche Herkunft ist noch nicht ausreichend dokumentiert; für eine
|
- Quelle: `jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF`
|
||||||
bitgenaue Migration muss die geprüfte Datei vom Referenzhost gesichert werden.
|
- Kontext 160.000
|
||||||
- Kontext 76.800
|
- RTX 5080 + RTX 3060 im Verhältnis 90:10
|
||||||
- vollständig auf CUDA0
|
|
||||||
- Flash Attention
|
- Flash Attention
|
||||||
- KV-Cache Q4_0 für K und V
|
- KV-Cache Q4_0 für K und V
|
||||||
- MTP Draft, maximal zwei Tokens
|
- MTP Draft, maximal drei Tokens
|
||||||
- sechs Threads und sechs Batch-Threads
|
- sechs Threads und sechs Batch-Threads
|
||||||
- Batch 64, Micro-Batch 32
|
- Batch 64, Micro-Batch 32
|
||||||
- ein paralleler Slot
|
- ein paralleler Slot
|
||||||
@@ -56,17 +55,17 @@ Zielplattform.
|
|||||||
| Profil | Virtuelles Modell | Kontext | Besonderheit |
|
| Profil | Virtuelles Modell | Kontext | Besonderheit |
|
||||||
|---|---|---:|---|
|
|---|---|---:|---|
|
||||||
| Fast | `qwen-fast` | 76.800 | IQ4-MIX, MTP2, vollständig GPU, CPU-mmproj |
|
| Fast | `qwen-fast` | 76.800 | IQ4-MIX, MTP2, vollständig GPU, CPU-mmproj |
|
||||||
| Medium | `qwen-medium` | 94.208 | IQ4_XS Pure, ohne MTP, CPU-mmproj |
|
| Medium **(Standard)** | `qwen-medium` | 160.000 | IQ4_XS Pure, MTP3, beide GPUs 90:10, CPU-mmproj |
|
||||||
| Long | `qwen-long` | 131.072 | IQ4-MIX, MTP2, FFN 0–11 auf CPU, CPU-mmproj |
|
| Large | `qwen-large` | 192.000 | IQ4_XS Pure, MTP3, beide GPUs 86:14, CPU-mmproj |
|
||||||
| Ultra | `qwen-ultra` | 262.144 | IQ4_XS Pure, MTP2, beide GPUs 80:20, text-only; 68,2 Tok/s und 220K-Fülltest bestanden |
|
| Ultra | `qwen-ultra` | 262.144 | IQ4_XS Pure, MTP2, beide GPUs 80:20, text-only; 68,2 Tok/s und 220K-Fülltest bestanden |
|
||||||
|
|
||||||
## Router
|
## Router
|
||||||
|
|
||||||
- Dienst: `mike-ai-profile-router.service`
|
- Container: `mike-ai-router`
|
||||||
- Port: 8081
|
- Port: 8081
|
||||||
- Upstream: `127.0.0.1:8080`
|
- Upstream: `llama-upstream:8080` im internen Inferenznetz
|
||||||
- Commit des Plattform-Repositories: siehe jeweils aktuelles `main`
|
- Commit des Plattform-Repositories: siehe jeweils aktuelles `main`
|
||||||
- Umschaltskript: `/usr/local/bin/llama-profile`
|
- Umschaltung: `mike-ai-profile-controller` mit fester Container-Allowlist
|
||||||
- Timeout für Profilwechsel und Requests: 600 Sekunden
|
- Timeout für Profilwechsel und Requests: 600 Sekunden
|
||||||
|
|
||||||
Der Router übernimmt:
|
Der Router übernimmt:
|
||||||
@@ -74,7 +73,7 @@ Der Router übernimmt:
|
|||||||
- OpenAI-kompatibles Chat-Proxying und Streaming
|
- OpenAI-kompatibles Chat-Proxying und Streaming
|
||||||
- virtuelle Modelle und automatische Profilumschaltung
|
- virtuelle Modelle und automatische Profilumschaltung
|
||||||
- Tool Calls
|
- Tool Calls
|
||||||
- direkte integrierte Vision in allen Qwen-Profilen
|
- direkte integrierte Vision in Fast, Medium und Large
|
||||||
- FLUX-Hotswap zur Bildgenerierung
|
- FLUX-Hotswap zur Bildgenerierung
|
||||||
- Whisper Speech-to-Text
|
- Whisper Speech-to-Text
|
||||||
- XTTS Text-to-Speech (historische Referenz; Zielsystem verwendet Piper)
|
- XTTS Text-to-Speech (historische Referenz; Zielsystem verwendet Piper)
|
||||||
@@ -87,9 +86,9 @@ Der Router übernimmt:
|
|||||||
| Text-/Visionmodell | jeweils aktives Qwen3.8-27B-Profil |
|
| Text-/Visionmodell | jeweils aktives Qwen3.8-27B-Profil |
|
||||||
| Projektor | BF16-mmproj |
|
| Projektor | BF16-mmproj |
|
||||||
| Speicherort des Projektors | System-RAM (`--no-mmproj-offload`) |
|
| Speicherort des Projektors | System-RAM (`--no-mmproj-offload`) |
|
||||||
| Kontext | entspricht Fast/Medium/Long |
|
| Kontext | entspricht Fast/Medium/Large; Ultra ist bewusst text-only |
|
||||||
|
|
||||||
Vision ist Bestandteil jedes Profils. Der Router prüft Bildgröße und URL,
|
Vision ist Bestandteil von Fast, Medium und Large. Der Router prüft Bildgröße und URL,
|
||||||
leitet das Bild dann direkt weiter und führt keinen Modellwechsel mehr aus.
|
leitet das Bild dann direkt weiter und führt keinen Modellwechsel mehr aus.
|
||||||
|
|
||||||
## Bildgenerierung
|
## Bildgenerierung
|
||||||
|
|||||||
@@ -37,7 +37,7 @@ laufen, sondern alle fachlichen Funktionen geprüft wurden.
|
|||||||
- [ ] Router-Key erscheint weder im Upstream noch im Journal
|
- [ ] Router-Key erscheint weder im Upstream noch im Journal
|
||||||
- [ ] `/status` meldet den richtigen Upstream
|
- [ ] `/status` meldet den richtigen Upstream
|
||||||
- [ ] `/v1/models` liefert drei virtuelle Modelle
|
- [ ] `/v1/models` liefert drei virtuelle Modelle
|
||||||
- [ ] `/fast`, `/medium` und `/long` wechseln zuverlässig
|
- [ ] `/fast`, `/medium`, `/large` und `/ultra` wechseln zuverlässig
|
||||||
- [ ] automatischer Wechsel über virtuellen Modellnamen funktioniert
|
- [ ] automatischer Wechsel über virtuellen Modellnamen funktioniert
|
||||||
- [ ] paralleler Wechsel wird sauber gesperrt
|
- [ ] paralleler Wechsel wird sauber gesperrt
|
||||||
- [ ] Streaming funktioniert
|
- [ ] Streaming funktioniert
|
||||||
|
|||||||
+8
-8
@@ -36,8 +36,9 @@ Folgefragen (`task.follow_up.enable=false`). Dieselbe Vorgabe steht zusätzlich
|
|||||||
als Container-Umgebungswert im Compose-Stack, damit bereits eine frische
|
als Container-Umgebungswert im Compose-Stack, damit bereits eine frische
|
||||||
OpenWebUI-Datenbank ohne Folgefragen startet.
|
OpenWebUI-Datenbank ohne Folgefragen startet.
|
||||||
|
|
||||||
Der Stabilitätsschutz kennt die drei Profilgrenzen 76.800, 94.208 und 131.072
|
Der Stabilitätsschutz kennt die vier Profilgrenzen 76.800, 160.000, 192.000
|
||||||
Token. Er reserviert Ausgabetoken und greift vor der harten llama.cpp-Grenze
|
und 262.144 Token. Für unbekannte Modelle gilt Medium (160.000) als sichere
|
||||||
|
Vorgabe. Er reserviert Ausgabetoken und greift vor der harten llama.cpp-Grenze
|
||||||
ein. Bilder bleiben unangetastet; JSON-Werkzeugschemas werden niemals
|
ein. Bilder bleiben unangetastet; JSON-Werkzeugschemas werden niemals
|
||||||
abgeschnitten. Sind allein die ausgewählten Schemas zu groß, wird der
|
abgeschnitten. Sind allein die ausgewählten Schemas zu groß, wird der
|
||||||
Werkzeugzugriff nur für diesen Schritt deaktiviert und das Modell erhält eine
|
Werkzeugzugriff nur für diesen Schritt deaktiviert und das Modell erhält eine
|
||||||
@@ -72,15 +73,14 @@ Community Store nachgeladen.
|
|||||||
| Profil | Virtuelles Modell | Kontext | Zweck |
|
| Profil | Virtuelles Modell | Kontext | Zweck |
|
||||||
|---|---|---:|---|
|
|---|---|---:|---|
|
||||||
| Fast | `qwen-fast` | 76.800 | Alltag, Agenten, hohe Geschwindigkeit, integrierte Vision |
|
| Fast | `qwen-fast` | 76.800 | Alltag, Agenten, hohe Geschwindigkeit, integrierte Vision |
|
||||||
| Medium | `qwen-medium` | 94.208 | mehr Kontext, reine IQ4_XS-Variante |
|
| Medium **(Standard)** | `qwen-medium` | 160.000 | IQ4_XS Pure, beide GPUs 90:10, MTP3, integrierte Vision |
|
||||||
| Long | `qwen-long` | 131.072 | lange Hermes-/MCP-Sitzungen |
|
| Large | `qwen-large` | 192.000 | IQ4_XS Pure, beide GPUs 86:14, MTP3, integrierte Vision |
|
||||||
| Ultra | `qwen-ultra` | 262.144 | maximaler Textkontext, IQ4_XS Pure auf RTX 5080 + RTX 3060 (80:20), ohne Vision-Projektor |
|
| Ultra | `qwen-ultra` | 262.144 | maximaler Textkontext, IQ4_XS Pure auf RTX 5080 + RTX 3060 (80:20), ohne Vision-Projektor |
|
||||||
|
|
||||||
Manuell wird mit `llama-profile fast|medium|long|ultra` gewechselt. Über HTTP stehen
|
Manuell wird mit `llama-profile fast|medium|large|ultra` gewechselt. Über HTTP stehen
|
||||||
`POST /fast`, `/medium`, `/long` und `/ultra` zur Verfügung. Ultra erreichte im
|
`POST /fast`, `/medium`, `/large` und `/ultra` zur Verfügung. Ultra erreichte im
|
||||||
Referenzlauf etwa 68 Token/s; ein Prompt-Fülltest mit rund 220.000 Tokens war
|
Referenzlauf etwa 68 Token/s; ein Prompt-Fülltest mit rund 220.000 Tokens war
|
||||||
erfolgreich. Für eine spätere Version ist
|
erfolgreich. Medium ist das Start- und Standardprofil.
|
||||||
`large` als Alias für `long` vorgesehen; bestehende Namen bleiben kompatibel.
|
|
||||||
|
|
||||||
## Clients
|
## Clients
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,9 @@
|
|||||||
# Router V2 – Migration und Kompatibilität
|
# Router V2 – Migration und Kompatibilität
|
||||||
|
|
||||||
|
> Historischer Stand: Die damalige Bezeichnung `long` wurde am 22. August
|
||||||
|
> 2026 durch `large` ersetzt und um `ultra` ergänzt. Für den aktuellen Betrieb
|
||||||
|
> gilt ausschließlich `STANDARD_PROFILE_MATRIX.md`.
|
||||||
|
|
||||||
## Ergebnis
|
## Ergebnis
|
||||||
|
|
||||||
V2 behält die OpenAI-kompatible Basis-URL und die virtuellen Modelle
|
V2 behält die OpenAI-kompatible Basis-URL und die virtuellen Modelle
|
||||||
|
|||||||
@@ -0,0 +1,33 @@
|
|||||||
|
# Verbindliche Standard-Profilmatrix
|
||||||
|
|
||||||
|
Stand: 22. August 2026. Diese vier Profile sind die Produktionsmatrix für
|
||||||
|
Athena. Änderungen gelten erst nach einem vergleichbaren synthetischen Test
|
||||||
|
und einer bewussten Aktualisierung dieser Datei.
|
||||||
|
|
||||||
|
| Profil | Virtuelles Modell | GGUF | Kontext | GPUs / Split | MTP | Vision | gemessene kurze Ausgabe |
|
||||||
|
|---|---|---|---:|---|---:|---|---:|
|
||||||
|
| Fast | `qwen-fast` | IQ4-MIX | 76.800 | RTX 5080 | 2 | ja, Projektor auf CPU | 85,5 Tok/s |
|
||||||
|
| **Medium (Default)** | `qwen-medium` | IQ4_XS Pure | 160.000 | RTX 5080 + RTX 3060, 90:10 | 3 | ja, Projektor auf CPU | 77,2 Tok/s |
|
||||||
|
| Large | `qwen-large` | IQ4_XS Pure | 192.000 | RTX 5080 + RTX 3060, 86:14 | 3 | ja, Projektor auf CPU | 75,3 Tok/s |
|
||||||
|
| Ultra | `qwen-ultra` | IQ4_XS Pure | 262.144 | RTX 5080 + RTX 3060, 80:20 | 2 | nein, text-only | 68,2 Tok/s |
|
||||||
|
|
||||||
|
## Standardverhalten
|
||||||
|
|
||||||
|
- Medium ist nach Neuinstallation und bewusstem Plattformstart das aktive
|
||||||
|
Standardprofil.
|
||||||
|
- Open WebUI erhält `qwen-medium` als Standardmodell.
|
||||||
|
- Ein explizit gewähltes anderes Modell löst den zugehörigen Containerwechsel
|
||||||
|
aus; es ist immer nur ein Inferenzcontainer aktiv.
|
||||||
|
- Ultra reserviert den verfügbaren Speicher für nativen 256K-Textkontext und
|
||||||
|
lädt deshalb keinen Vision-Projektor.
|
||||||
|
- Das gesonderte Experimentalprofil gehört nicht zur Benutzer-Matrix und wird
|
||||||
|
in Open WebUI nicht als reguläres Modell angeboten.
|
||||||
|
|
||||||
|
## Nachweise
|
||||||
|
|
||||||
|
- Fast 76,8K: synthetischer Referenzlauf, 85,50 Tok/s.
|
||||||
|
- Medium 160K: Pure 90:10 mit MTP3, 77,22 Tok/s.
|
||||||
|
- Large 192K: Pure 86:14 mit MTP3, 75,28 Tok/s.
|
||||||
|
- Ultra 256K: Pure 80:20 mit MTP2, 68,19 Tok/s; 220.190 Tokens
|
||||||
|
erfolgreich verarbeitet und Sentinel korrekt wiedergefunden.
|
||||||
|
|
||||||
+15
-13
@@ -40,7 +40,7 @@ source "$CONFIG"
|
|||||||
|
|
||||||
required=(AI_HOSTNAME ADMIN_USER AI_BIND_ADDRESS MODEL_DIR FAST_MODEL_FILE
|
required=(AI_HOSTNAME ADMIN_USER AI_BIND_ADDRESS MODEL_DIR FAST_MODEL_FILE
|
||||||
FAST_MODEL_URL FAST_MODEL_SHA256 MEDIUM_MODEL_FILE MEDIUM_MODEL_URL
|
FAST_MODEL_URL FAST_MODEL_SHA256 MEDIUM_MODEL_FILE MEDIUM_MODEL_URL
|
||||||
MEDIUM_MODEL_SHA256 LONG_MODEL_FILE LONG_MODEL_URL LONG_MODEL_SHA256
|
MEDIUM_MODEL_SHA256 LARGE_MODEL_FILE LARGE_MODEL_URL LARGE_MODEL_SHA256
|
||||||
ULTRA_MODEL_FILE ULTRA_MODEL_URL ULTRA_MODEL_SHA256
|
ULTRA_MODEL_FILE ULTRA_MODEL_URL ULTRA_MODEL_SHA256
|
||||||
EXPERIMENTAL_MODEL_FILE EXPERIMENTAL_MODEL_URL EXPERIMENTAL_MODEL_SHA256
|
EXPERIMENTAL_MODEL_FILE EXPERIMENTAL_MODEL_URL EXPERIMENTAL_MODEL_SHA256
|
||||||
VISION_PROJECTOR_FILE VISION_PROJECTOR_URL VISION_PROJECTOR_SHA256)
|
VISION_PROJECTOR_FILE VISION_PROJECTOR_URL VISION_PROJECTOR_SHA256)
|
||||||
@@ -209,18 +209,20 @@ PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
|
|||||||
AI_DNS=${WG_DNS:-1.1.1.1}
|
AI_DNS=${WG_DNS:-1.1.1.1}
|
||||||
FAST_MODEL_FILE=$FAST_MODEL_FILE
|
FAST_MODEL_FILE=$FAST_MODEL_FILE
|
||||||
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
|
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
|
||||||
LONG_MODEL_FILE=$LONG_MODEL_FILE
|
LARGE_MODEL_FILE=$LARGE_MODEL_FILE
|
||||||
ULTRA_MODEL_FILE=$ULTRA_MODEL_FILE
|
ULTRA_MODEL_FILE=$ULTRA_MODEL_FILE
|
||||||
EXPERIMENTAL_MODEL_FILE=$EXPERIMENTAL_MODEL_FILE
|
EXPERIMENTAL_MODEL_FILE=$EXPERIMENTAL_MODEL_FILE
|
||||||
VISION_PROJECTOR_FILE=$VISION_PROJECTOR_FILE
|
VISION_PROJECTOR_FILE=$VISION_PROJECTOR_FILE
|
||||||
FAST_CONTEXT=${FAST_CONTEXT:-76800}
|
FAST_CONTEXT=${FAST_CONTEXT:-76800}
|
||||||
MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-94208}
|
MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-160000}
|
||||||
LONG_CONTEXT=${LONG_CONTEXT:-131072}
|
LARGE_CONTEXT=${LARGE_CONTEXT:-192000}
|
||||||
ULTRA_CONTEXT=${ULTRA_CONTEXT:-262144}
|
ULTRA_CONTEXT=${ULTRA_CONTEXT:-262144}
|
||||||
EXPERIMENTAL_CONTEXT=${EXPERIMENTAL_CONTEXT:-76800}
|
EXPERIMENTAL_CONTEXT=${EXPERIMENTAL_CONTEXT:-76800}
|
||||||
FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
||||||
MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||||
LONG_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
MEDIUM_TENSOR_SPLIT=${MEDIUM_TENSOR_SPLIT:-90,10}
|
||||||
|
LARGE_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||||
|
LARGE_TENSOR_SPLIT=${LARGE_TENSOR_SPLIT:-86,14}
|
||||||
ULTRA_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
ULTRA_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||||
ULTRA_TENSOR_SPLIT=${ULTRA_TENSOR_SPLIT:-80,20}
|
ULTRA_TENSOR_SPLIT=${ULTRA_TENSOR_SPLIT:-80,20}
|
||||||
EXPERIMENTAL_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
EXPERIMENTAL_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
||||||
@@ -260,7 +262,7 @@ download_models() {
|
|||||||
done <<EOF
|
done <<EOF
|
||||||
$FAST_MODEL_FILE|$FAST_MODEL_URL|$FAST_MODEL_SHA256
|
$FAST_MODEL_FILE|$FAST_MODEL_URL|$FAST_MODEL_SHA256
|
||||||
$MEDIUM_MODEL_FILE|$MEDIUM_MODEL_URL|$MEDIUM_MODEL_SHA256
|
$MEDIUM_MODEL_FILE|$MEDIUM_MODEL_URL|$MEDIUM_MODEL_SHA256
|
||||||
$LONG_MODEL_FILE|$LONG_MODEL_URL|$LONG_MODEL_SHA256
|
$LARGE_MODEL_FILE|$LARGE_MODEL_URL|$LARGE_MODEL_SHA256
|
||||||
$ULTRA_MODEL_FILE|$ULTRA_MODEL_URL|$ULTRA_MODEL_SHA256
|
$ULTRA_MODEL_FILE|$ULTRA_MODEL_URL|$ULTRA_MODEL_SHA256
|
||||||
$EXPERIMENTAL_MODEL_FILE|$EXPERIMENTAL_MODEL_URL|$EXPERIMENTAL_MODEL_SHA256
|
$EXPERIMENTAL_MODEL_FILE|$EXPERIMENTAL_MODEL_URL|$EXPERIMENTAL_MODEL_SHA256
|
||||||
$VISION_PROJECTOR_FILE|$VISION_PROJECTOR_URL|$VISION_PROJECTOR_SHA256
|
$VISION_PROJECTOR_FILE|$VISION_PROJECTOR_URL|$VISION_PROJECTOR_SHA256
|
||||||
@@ -329,7 +331,7 @@ build_and_start() {
|
|||||||
# secret files and required local artifacts are present.
|
# secret files and required local artifacts are present.
|
||||||
"$STACK_DIR/platform/mcp/install-tools.sh"
|
"$STACK_DIR/platform/mcp/install-tools.sh"
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \
|
docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \
|
||||||
llama-fast llama-medium llama-long llama-ultra llama-experimental
|
llama-fast llama-medium llama-large llama-ultra llama-experimental
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
||||||
profile-controller router open-webui
|
profile-controller router open-webui
|
||||||
|
|
||||||
@@ -343,7 +345,7 @@ build_and_start() {
|
|||||||
sleep 2
|
sleep 2
|
||||||
done
|
done
|
||||||
|
|
||||||
log "Fast-Profil aktivieren und Readiness prüfen"
|
log "Medium-Profil als Standard aktivieren und Readiness prüfen"
|
||||||
# Use docker exec directly here. Some Compose/Docker combinations return a
|
# Use docker exec directly here. Some Compose/Docker combinations return a
|
||||||
# transient HTTP 409 while upgrading the exec stream immediately after a
|
# transient HTTP 409 while upgrading the exec stream immediately after a
|
||||||
# freshly built service has been recreated.
|
# freshly built service has been recreated.
|
||||||
@@ -352,7 +354,7 @@ build_and_start() {
|
|||||||
import json, os, time, urllib.request
|
import json, os, time, urllib.request
|
||||||
key = os.environ["ROUTER_API_KEY"]
|
key = os.environ["ROUTER_API_KEY"]
|
||||||
request = urllib.request.Request(
|
request = urllib.request.Request(
|
||||||
"http://127.0.0.1:8081/fast", method="POST",
|
"http://127.0.0.1:8081/medium", method="POST",
|
||||||
headers={"Authorization": f"Bearer {key}"})
|
headers={"Authorization": f"Bearer {key}"})
|
||||||
with urllib.request.urlopen(request, timeout=700) as response:
|
with urllib.request.urlopen(request, timeout=700) as response:
|
||||||
print(json.dumps(json.load(response), indent=2))
|
print(json.dumps(json.load(response), indent=2))
|
||||||
@@ -361,7 +363,7 @@ while time.monotonic() < deadline:
|
|||||||
try:
|
try:
|
||||||
with urllib.request.urlopen("http://127.0.0.1:8081/ready", timeout=5) as response:
|
with urllib.request.urlopen("http://127.0.0.1:8081/ready", timeout=5) as response:
|
||||||
if response.status == 200:
|
if response.status == 200:
|
||||||
print("Router und Fast-Profil sind bereit.")
|
print("Router und Medium-Standardprofil sind bereit.")
|
||||||
print("INSTALL_READINESS_OK")
|
print("INSTALL_READINESS_OK")
|
||||||
break
|
break
|
||||||
except Exception:
|
except Exception:
|
||||||
@@ -370,10 +372,10 @@ while time.monotonic() < deadline:
|
|||||||
else:
|
else:
|
||||||
raise SystemExit("Readiness-Check fehlgeschlagen")
|
raise SystemExit("Readiness-Check fehlgeschlagen")
|
||||||
PY
|
PY
|
||||||
) || die "Fast-Profil konnte nicht aktiviert werden"
|
) || die "Medium-Standardprofil konnte nicht aktiviert werden"
|
||||||
printf '%s\n' "$activation_output"
|
printf '%s\n' "$activation_output"
|
||||||
grep -Fxq 'INSTALL_READINESS_OK' <<<"$activation_output" || \
|
grep -Fxq 'INSTALL_READINESS_OK' <<<"$activation_output" || \
|
||||||
die "Fast-Profil lieferte keinen bestätigten Readiness-Marker"
|
die "Medium-Standardprofil lieferte keinen bestätigten Readiness-Marker"
|
||||||
}
|
}
|
||||||
|
|
||||||
hostnamectl set-hostname "$AI_HOSTNAME"
|
hostnamectl set-hostname "$AI_HOSTNAME"
|
||||||
|
|||||||
@@ -40,7 +40,7 @@ for container in mike-ai-profile-controller mike-ai-router mike-ai-piper mike-ai
|
|||||||
fi
|
fi
|
||||||
done
|
done
|
||||||
|
|
||||||
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|long|ultra|experimental)$' || true)"
|
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|large|ultra|experimental)$' || true)"
|
||||||
if [[ $ACTIVE_LLAMA -eq 1 ]]; then
|
if [[ $ACTIVE_LLAMA -eq 1 ]]; then
|
||||||
pass "exakt ein llama.cpp-Profil aktiv"
|
pass "exakt ein llama.cpp-Profil aktiv"
|
||||||
else
|
else
|
||||||
|
|||||||
@@ -21,7 +21,7 @@ PORT = int(os.environ.get("CONTROLLER_PORT", "8090"))
|
|||||||
SOCKET_PATH = os.environ.get("DOCKER_SOCKET", "/var/run/docker.sock")
|
SOCKET_PATH = os.environ.get("DOCKER_SOCKET", "/var/run/docker.sock")
|
||||||
TOKEN_FILE = os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token")
|
TOKEN_FILE = os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token")
|
||||||
ALLOWED = tuple(x.strip() for x in os.environ.get(
|
ALLOWED = tuple(x.strip() for x in os.environ.get(
|
||||||
"ALLOWED_PROFILES", "fast,medium,long,ultra,experimental").split(",") if x.strip())
|
"ALLOWED_PROFILES", "fast,medium,large,ultra,experimental").split(",") if x.strip())
|
||||||
LABEL_KEY = "com.mike-ai.llama-profile"
|
LABEL_KEY = "com.mike-ai.llama-profile"
|
||||||
LOCK = threading.Lock()
|
LOCK = threading.Lock()
|
||||||
log = logging.getLogger("profile-controller")
|
log = logging.getLogger("profile-controller")
|
||||||
|
|||||||
@@ -21,7 +21,8 @@ install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \
|
|||||||
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
|
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
|
||||||
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
|
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
|
||||||
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
|
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
|
||||||
install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf"
|
install -m 0644 "$PLATFORM/profiles/profile-large.conf" "$PROFILE_TARGET/profile-large.conf"
|
||||||
|
install -m 0644 "$PLATFORM/profiles/profile-ultra.conf" "$PROFILE_TARGET/profile-ultra.conf"
|
||||||
rsync -a --delete "$PLATFORM/mcp/" "$MCP_TARGET/"
|
rsync -a --delete "$PLATFORM/mcp/" "$MCP_TARGET/"
|
||||||
rsync -a --delete --exclude searxng-settings.yml \
|
rsync -a --delete --exclude searxng-settings.yml \
|
||||||
"$PLATFORM/web-search/" "$MCP_SEARCH_TARGET/"
|
"$PLATFORM/web-search/" "$MCP_SEARCH_TARGET/"
|
||||||
@@ -39,7 +40,7 @@ Kernkonfiguration installiert, aber noch nicht gestartet.
|
|||||||
|
|
||||||
Vor dem Start:
|
Vor dem Start:
|
||||||
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
|
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
|
||||||
2. llama.cpp bauen und mit `llama-profile fast` starten.
|
2. llama.cpp bauen und mit `llama-profile medium` starten.
|
||||||
3. Fach-Secrets unter /etc/mike-ai ablegen.
|
3. Fach-Secrets unter /etc/mike-ai ablegen.
|
||||||
4. Danach: /opt/mike-ai/mcp-containers/platform/mcp/install-tools.sh
|
4. Danach: /opt/mike-ai/mcp-containers/platform/mcp/install-tools.sh
|
||||||
EOF
|
EOF
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
schema: 1
|
schema: 1
|
||||||
models:
|
models:
|
||||||
qwen_fast_long:
|
qwen_fast_long:
|
||||||
role: primary-text-fast-and-long
|
role: primary-text-fast
|
||||||
source: "local migration from the reference host; public origin still to document"
|
source: "local migration from the reference host; public origin still to document"
|
||||||
file: Qwen3.8-27B-IQ4-MIX.gguf
|
file: Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
target: /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
target: /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
|
|||||||
@@ -16,9 +16,10 @@ class Filter:
|
|||||||
class Valves(BaseModel):
|
class Valves(BaseModel):
|
||||||
priority: int = 30
|
priority: int = 30
|
||||||
fast_context_tokens: int = 76800
|
fast_context_tokens: int = 76800
|
||||||
medium_context_tokens: int = 94208
|
medium_context_tokens: int = 160000
|
||||||
long_context_tokens: int = 131072
|
large_context_tokens: int = 192000
|
||||||
default_context_tokens: int = 76800
|
ultra_context_tokens: int = 262144
|
||||||
|
default_context_tokens: int = 160000
|
||||||
soft_context_ratio: float = 0.70
|
soft_context_ratio: float = 0.70
|
||||||
hard_context_ratio: float = 0.84
|
hard_context_ratio: float = 0.84
|
||||||
reserved_output_tokens: int = 8192
|
reserved_output_tokens: int = 8192
|
||||||
@@ -107,8 +108,10 @@ class Filter:
|
|||||||
|
|
||||||
def _context_limit(self, model: str) -> int:
|
def _context_limit(self, model: str) -> int:
|
||||||
model = (model or "").lower()
|
model = (model or "").lower()
|
||||||
if "long" in model or "large" in model:
|
if "ultra" in model:
|
||||||
return self.valves.long_context_tokens
|
return self.valves.ultra_context_tokens
|
||||||
|
if "large" in model:
|
||||||
|
return self.valves.large_context_tokens
|
||||||
if "medium" in model:
|
if "medium" in model:
|
||||||
return self.valves.medium_context_tokens
|
return self.valves.medium_context_tokens
|
||||||
if "fast" in model:
|
if "fast" in model:
|
||||||
|
|||||||
@@ -0,0 +1,6 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Legacy native Qwen Large 192K profile (Docker is the production path)
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
ExecStart=
|
||||||
|
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||||
@@ -1,6 +0,0 @@
|
|||||||
[Unit]
|
|
||||||
Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU with CPU Vision
|
|
||||||
|
|
||||||
[Service]
|
|
||||||
ExecStart=
|
|
||||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
|
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
[Unit]
|
[Unit]
|
||||||
Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure with CPU Vision
|
Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
|
||||||
|
|
||||||
[Service]
|
[Service]
|
||||||
ExecStart=
|
ExecStart=
|
||||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none
|
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||||
|
|||||||
@@ -0,0 +1,6 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Legacy native Qwen Ultra 256K profile (Docker is the production path)
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
ExecStart=
|
||||||
|
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen-ultra --ctx-size 262144 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 80,20 --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||||
@@ -7,10 +7,9 @@ SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
|
|||||||
PROFILE="${1:-}"
|
PROFILE="${1:-}"
|
||||||
|
|
||||||
case "$PROFILE" in
|
case "$PROFILE" in
|
||||||
fast|medium|long) ;;
|
fast|medium|large|ultra) ;;
|
||||||
large) PROFILE=long ;;
|
|
||||||
*)
|
*)
|
||||||
echo "Usage: llama-profile {fast|medium|long|large}" >&2
|
echo "Usage: llama-profile {fast|medium|large|ultra}" >&2
|
||||||
exit 2
|
exit 2
|
||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|||||||
@@ -8,12 +8,12 @@ Profilen um:
|
|||||||
Profil Kontext
|
Profil Kontext
|
||||||
------ --------
|
------ --------
|
||||||
fast 76800
|
fast 76800
|
||||||
medium 94208
|
medium 160000
|
||||||
long 131072
|
large 192000
|
||||||
ultra 262144
|
ultra 262144
|
||||||
|
|
||||||
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-long, qwen-ultra
|
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large, qwen-ultra
|
||||||
Kommandos: POST /fast, /medium, /long, /ultra (Profilwechsel)
|
Kommandos: POST /fast, /medium, /large, /ultra (Profilwechsel)
|
||||||
GET /status (Zustand)
|
GET /status (Zustand)
|
||||||
|
|
||||||
Bildgenerierung (FLUX.2 [klein] 4B Base):
|
Bildgenerierung (FLUX.2 [klein] 4B Base):
|
||||||
@@ -1110,11 +1110,11 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
self._images_list()
|
self._images_list()
|
||||||
elif path.startswith("/images/") and self.command == "GET":
|
elif path.startswith("/images/") and self.command == "GET":
|
||||||
self._image_serve(path[len("/images/"):])
|
self._image_serve(path[len("/images/"):])
|
||||||
elif (path in ("/fast", "/medium", "/long", "/ultra")
|
elif (path in ("/fast", "/medium", "/large", "/ultra")
|
||||||
and (self.command == "POST"
|
and (self.command == "POST"
|
||||||
or (self.command == "GET" and ALLOW_LEGACY_GET_SWITCH))):
|
or (self.command == "GET" and ALLOW_LEGACY_GET_SWITCH))):
|
||||||
self._switch(path[1:])
|
self._switch(path[1:])
|
||||||
elif (path in ("/fast", "/medium", "/long", "/ultra")
|
elif (path in ("/fast", "/medium", "/large", "/ultra")
|
||||||
and self.command == "GET"):
|
and self.command == "GET"):
|
||||||
self._send_error(405, "Profilwechsel erfordert POST",
|
self._send_error(405, "Profilwechsel erfordert POST",
|
||||||
"invalid_request_error", "method_not_allowed")
|
"invalid_request_error", "method_not_allowed")
|
||||||
|
|||||||
@@ -5,12 +5,12 @@
|
|||||||
"model_alias": "qwen-fast"
|
"model_alias": "qwen-fast"
|
||||||
},
|
},
|
||||||
"medium": {
|
"medium": {
|
||||||
"context": 94208,
|
"context": 160000,
|
||||||
"model_alias": "qwen-medium"
|
"model_alias": "qwen-medium"
|
||||||
},
|
},
|
||||||
"long": {
|
"large": {
|
||||||
"context": 131072,
|
"context": 192000,
|
||||||
"model_alias": "qwen-long"
|
"model_alias": "qwen-large"
|
||||||
},
|
},
|
||||||
"ultra": {
|
"ultra": {
|
||||||
"context": 262144,
|
"context": 262144,
|
||||||
|
|||||||
@@ -34,8 +34,8 @@ def load_profile_registry(path: str | None) -> dict[str, dict]:
|
|||||||
|
|
||||||
fallback = {
|
fallback = {
|
||||||
"fast": {"context": 76800, "model_alias": None},
|
"fast": {"context": 76800, "model_alias": None},
|
||||||
"medium": {"context": 94208, "model_alias": None},
|
"medium": {"context": 160000, "model_alias": None},
|
||||||
"long": {"context": 131072, "model_alias": None},
|
"large": {"context": 192000, "model_alias": None},
|
||||||
"ultra": {"context": 262144, "model_alias": None},
|
"ultra": {"context": 262144, "model_alias": None},
|
||||||
}
|
}
|
||||||
if not path:
|
if not path:
|
||||||
|
|||||||
Reference in New Issue
Block a user