Define four-profile production matrix with Medium default
This commit is contained in:
+9
-7
@@ -9,21 +9,23 @@ PIPER_TTS_VERSION=1.6.0
|
||||
PIPER_VOICE=de_DE-thorsten-high
|
||||
AI_DNS=192.168.1.1
|
||||
|
||||
FAST_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
MEDIUM_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
LONG_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||
MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
EXPERIMENTAL_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
||||
|
||||
FAST_CONTEXT=76800
|
||||
MEDIUM_CONTEXT=94208
|
||||
LONG_CONTEXT=131072
|
||||
MEDIUM_CONTEXT=160000
|
||||
LARGE_CONTEXT=192000
|
||||
ULTRA_CONTEXT=262144
|
||||
EXPERIMENTAL_CONTEXT=76800
|
||||
FAST_GPU_DEVICES=0
|
||||
MEDIUM_GPU_DEVICES=0
|
||||
LONG_GPU_DEVICES=0
|
||||
MEDIUM_GPU_DEVICES=0,1
|
||||
MEDIUM_TENSOR_SPLIT=90,10
|
||||
LARGE_GPU_DEVICES=0,1
|
||||
LARGE_TENSOR_SPLIT=86,14
|
||||
ULTRA_GPU_DEVICES=0,1
|
||||
ULTRA_TENSOR_SPLIT=80,20
|
||||
EXPERIMENTAL_GPU_DEVICES=0
|
||||
|
||||
@@ -10,7 +10,9 @@ WireGuard-Isolation.
|
||||
- llama.cpp selbst gebaut und auf einen geprüften Commit festgelegt
|
||||
- vier schaltbare Profilcontainer plus ein isolierter Experimentalcontainer;
|
||||
davon ist immer exakt ein Inferenzcontainer aktiv
|
||||
- `/fast`, `/medium`, `/long` und `/ultra` über den Profile Router
|
||||
- `/fast`, `/medium`, `/large` und `/ultra` über den Profile Router
|
||||
- verbindliche Standardmatrix: Fast MIX 76,8K, Medium Pure 160K (Default),
|
||||
Large Pure 192K und Ultra Pure 256K
|
||||
- `/ultra`: getestetes text-only 256K-Profil (IQ4_XS Pure, beide GPUs,
|
||||
80:20); etwa 68 Token/s und erfolgreicher 220K-Prompt-Fülltest
|
||||
- Open WebUI als einzige normale Oberfläche
|
||||
@@ -21,6 +23,9 @@ WireGuard-Isolation.
|
||||
- KI-Ausgangsverkehr über das Heimnetz, bei Tunnelausfall fail-closed
|
||||
- keine Secrets, Chats, Logs oder Modelldateien im Repository
|
||||
|
||||
Die gemessenen Startparameter und Zuständigkeiten stehen in
|
||||
[`docs/STANDARD_PROFILE_MATRIX.md`](docs/STANDARD_PROFILE_MATRIX.md).
|
||||
|
||||
## Schnellstart
|
||||
|
||||
Auf einem frisch installierten Debian 12/13 amd64:
|
||||
@@ -94,6 +99,6 @@ router/ OpenAI-kompatibler Profile Router
|
||||
- Ein Blackhole-Fallback verhindert Traffic-Leaks bei WireGuard-Ausfall.
|
||||
- Das Uni-Netz und das Heimnetz dürfen diesen Host nicht als Transit benutzen.
|
||||
|
||||
Die Profilwerte sind Ausgangswerte. Nach dem Neuaufbau werden RTX 5080 und
|
||||
RTX 3060 mit der bestehenden Standard-Testserie neu vermessen, bevor die zweite
|
||||
GPU in ein Produktionsprofil einfließt.
|
||||
Die Profilwerte wurden auf RTX 5080 und RTX 3060 vermessen und bilden die
|
||||
verbindliche Standardmatrix. Neue Varianten ersetzen sie erst nach demselben
|
||||
Vergleichstest und einer dokumentierten Entscheidung.
|
||||
|
||||
+35
-20
@@ -100,7 +100,7 @@ services:
|
||||
labels:
|
||||
com.mike-ai.llama-profile: medium
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0}
|
||||
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0,1}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
command:
|
||||
- --model
|
||||
@@ -111,7 +111,7 @@ services:
|
||||
- --alias
|
||||
- qwen-medium
|
||||
- --ctx-size
|
||||
- "${MEDIUM_CONTEXT:-94208}"
|
||||
- "${MEDIUM_CONTEXT:-160000}"
|
||||
- --flash-attn
|
||||
- "on"
|
||||
- --cache-type-k
|
||||
@@ -149,28 +149,40 @@ services:
|
||||
- --top-k
|
||||
- "20"
|
||||
- --device
|
||||
- CUDA0
|
||||
- CUDA0,CUDA1
|
||||
- --main-gpu
|
||||
- "0"
|
||||
- --split-mode
|
||||
- none
|
||||
- layer
|
||||
- --tensor-split
|
||||
- "${MEDIUM_TENSOR_SPLIT:-90,10}"
|
||||
- --spec-type
|
||||
- draft-mtp
|
||||
- --spec-draft-n-max
|
||||
- "3"
|
||||
- --spec-draft-type-k
|
||||
- f16
|
||||
- --spec-draft-type-v
|
||||
- f16
|
||||
|
||||
llama-long:
|
||||
llama-large:
|
||||
<<: *llama-common
|
||||
container_name: mike-ai-llama-long
|
||||
container_name: mike-ai-llama-large
|
||||
labels:
|
||||
com.mike-ai.llama-profile: long
|
||||
com.mike-ai.llama-profile: large
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0}
|
||||
NVIDIA_VISIBLE_DEVICES: ${LARGE_GPU_DEVICES:-0,1}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
command:
|
||||
- --model
|
||||
- "/models/${LONG_MODEL_FILE:?LONG_MODEL_FILE is required}"
|
||||
- "/models/${LARGE_MODEL_FILE:?LARGE_MODEL_FILE is required}"
|
||||
- --mmproj
|
||||
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
|
||||
- --no-mmproj-offload
|
||||
- --alias
|
||||
- qwen-long
|
||||
- qwen-large
|
||||
- --ctx-size
|
||||
- "${LONG_CONTEXT:-131072}"
|
||||
- "${LARGE_CONTEXT:-192000}"
|
||||
- --flash-attn
|
||||
- "on"
|
||||
- --cache-type-k
|
||||
@@ -199,8 +211,6 @@ services:
|
||||
- "off"
|
||||
- --n-gpu-layers
|
||||
- all
|
||||
- --override-tensor
|
||||
- blk.([0-9]|1[0-1]).ffn_.*=CPU
|
||||
- --no-mmap
|
||||
- --no-ui
|
||||
- --temperature
|
||||
@@ -210,19 +220,23 @@ services:
|
||||
- --top-k
|
||||
- "20"
|
||||
- --device
|
||||
- CUDA0
|
||||
- CUDA0,CUDA1
|
||||
- --main-gpu
|
||||
- "0"
|
||||
- --split-mode
|
||||
- none
|
||||
- layer
|
||||
- --tensor-split
|
||||
- "${LARGE_TENSOR_SPLIT:-86,14}"
|
||||
- --spec-type
|
||||
- draft-mtp
|
||||
- --spec-draft-n-max
|
||||
- "2"
|
||||
- "3"
|
||||
- --spec-draft-type-k
|
||||
- q4_0
|
||||
- f16
|
||||
- --spec-draft-type-v
|
||||
- q4_0
|
||||
- f16
|
||||
|
||||
# Text-only long-context profile. This exact IQ4_XS-pure / 256K / 80:20
|
||||
# Text-only maximum-context profile. This exact IQ4_XS-pure / 256K / 80:20
|
||||
# combination completed the 220K fill test on RTX 5080 + RTX 3060.
|
||||
# Deliberately no vision projector: Ultra prioritizes maximum usable context.
|
||||
llama-ultra:
|
||||
@@ -347,7 +361,7 @@ services:
|
||||
- /var/run/docker.sock:/var/run/docker.sock
|
||||
environment:
|
||||
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||
ALLOWED_PROFILES: fast,medium,long,ultra,experimental
|
||||
ALLOWED_PROFILES: fast,medium,large,ultra,experimental
|
||||
networks: [control]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
healthcheck:
|
||||
@@ -456,6 +470,7 @@ services:
|
||||
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
|
||||
environment:
|
||||
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
|
||||
DEFAULT_MODELS: qwen-medium
|
||||
OLLAMA_BASE_URL: ""
|
||||
OPENAI_API_BASE_URLS: http://router:8081/v1
|
||||
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
||||
|
||||
+14
-12
@@ -28,15 +28,15 @@ WG_DNS=192.168.1.1
|
||||
WG_ROUTE_AI_INTERNET=true
|
||||
|
||||
# Exact model artifacts. Never put access tokens in these URLs.
|
||||
FAST_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
FAST_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
FAST_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199
|
||||
MEDIUM_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
MEDIUM_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
MEDIUM_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199
|
||||
LONG_MODEL_FILE=qwen/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
LONG_MODEL_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/Qwen3.8-27B-UD-IQ4_XS.gguf
|
||||
LONG_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd7483470e6199
|
||||
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||
FAST_MODEL_URL=https://huggingface.co/vmarcelo/Qwen3.8-27B-MIX_GGUF/resolve/main/Qwen3.8-27B-IQ4-MIX.gguf
|
||||
FAST_MODEL_SHA256=54879ae8738d5938f46cb3b8cbf16bf42b8c85b7d68d7c73f062b612ec183e36
|
||||
MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
MEDIUM_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
MEDIUM_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
|
||||
LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
LARGE_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
LARGE_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
|
||||
ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
ULTRA_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
ULTRA_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
|
||||
@@ -46,12 +46,14 @@ EXPERIMENTAL_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd74
|
||||
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
||||
VISION_PROJECTOR_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/mmproj-BF16.gguf
|
||||
VISION_PROJECTOR_SHA256=83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||
# FAST/LONG use the MTP tensor embedded in the IQ4-MIX GGUF. A separate
|
||||
# All standard profiles use the MTP tensor embedded in their GGUF. A separate
|
||||
# draft-model artifact is neither downloaded nor passed to llama-server.
|
||||
|
||||
FAST_CONTEXT=76800
|
||||
MEDIUM_CONTEXT=94208
|
||||
LONG_CONTEXT=131072
|
||||
MEDIUM_CONTEXT=160000
|
||||
MEDIUM_TENSOR_SPLIT=90,10
|
||||
LARGE_CONTEXT=192000
|
||||
LARGE_TENSOR_SPLIT=86,14
|
||||
ULTRA_CONTEXT=262144
|
||||
ULTRA_TENSOR_SPLIT=80,20
|
||||
EXPERIMENTAL_CONTEXT=76800
|
||||
|
||||
@@ -5,8 +5,8 @@ set -e
|
||||
D="$(cd "$(dirname "$0")" && pwd)/fake-profile-dir"
|
||||
|
||||
case "${1:-}" in
|
||||
fast|medium|long) ;;
|
||||
*) echo "Usage: fake-llama-profile {fast|medium|long}"; exit 1 ;;
|
||||
fast|medium|large|ultra) ;;
|
||||
*) echo "Usage: fake-llama-profile {fast|medium|large|ultra}"; exit 1 ;;
|
||||
esac
|
||||
|
||||
if [ -f /tmp/fake-profile-fail ] \
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
ExecStart=/usr/bin/mock-llama --ctx-size 192000
|
||||
@@ -1 +0,0 @@
|
||||
ExecStart=/usr/bin/mock-llama --ctx-size 131072
|
||||
@@ -1 +1 @@
|
||||
ExecStart=/usr/bin/mock-llama --ctx-size 94208
|
||||
ExecStart=/usr/bin/mock-llama --ctx-size 160000
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
ExecStart=/usr/bin/mock-llama --ctx-size 262144
|
||||
@@ -19,7 +19,7 @@ ROUTER_STATUS=http://127.0.0.1:8081/status
|
||||
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
||||
# vergleichen (robuster als String-Matching).
|
||||
PROFILE=""
|
||||
for p in fast medium long; do
|
||||
for p in fast medium large ultra; do
|
||||
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
||||
PROFILE=$p
|
||||
break
|
||||
|
||||
@@ -19,7 +19,7 @@ ROUTER_STATUS=http://127.0.0.1:8081/status
|
||||
# Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien
|
||||
# vergleichen (robuster als String-Matching).
|
||||
PROFILE=""
|
||||
for p in fast medium long; do
|
||||
for p in fast medium large ultra; do
|
||||
if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then
|
||||
PROFILE=$p
|
||||
break
|
||||
|
||||
+25
-24
@@ -123,11 +123,12 @@ echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
ids={m["id"]:m for m in d["data"]}
|
||||
assert set(ids)=={"qwen-fast","qwen-medium","qwen-long"}, ids
|
||||
assert set(ids)=={"qwen-fast","qwen-medium","qwen-large","qwen-ultra"}, ids
|
||||
assert ids["qwen-fast"]["context_length"]==76800
|
||||
assert ids["qwen-medium"]["context_length"]==94208
|
||||
assert ids["qwen-long"]["context_length"]==131072
|
||||
' && ok "drei virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
|
||||
assert ids["qwen-medium"]["context_length"]==160000
|
||||
assert ids["qwen-large"]["context_length"]==192000
|
||||
assert ids["qwen-ultra"]["context_length"]==262144
|
||||
' && ok "vier virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
|
||||
|
||||
# --- 2. /status -----------------------------------------------------------------
|
||||
echo "== Test 2: /status"
|
||||
@@ -183,14 +184,14 @@ echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["profile"]=="medium" and d["context_length"]==94208, d
|
||||
assert d["profile"]=="medium" and d["context_length"]==160000, d
|
||||
' && ok "Profil medium aktiv" || bad "Profilwechsel medium"
|
||||
curl -sf "$BASE/status" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["current_profile"]=="medium", d
|
||||
assert d["upstream"]["ctx"]==94208, d
|
||||
' && ok "Status bestätigt medium (ctx 94208)" || bad "Status nach Wechsel"
|
||||
assert d["upstream"]["ctx"]==160000, d
|
||||
' && ok "Status bestätigt medium (ctx 160000)" || bad "Status nach Wechsel"
|
||||
|
||||
# --- 7. Profilwechsel medium -> fast --------------------------------------------------
|
||||
echo "== Test 7: Profilwechsel medium -> fast"
|
||||
@@ -203,21 +204,21 @@ assert d["profile"]=="fast" and d["context_length"]==76800, d
|
||||
' && ok "Profil fast wieder aktiv" || bad "Profilwechsel fast"
|
||||
|
||||
# --- 8. Virtuelles Modell triggert Profilwechsel ----------------------------------------
|
||||
echo "== Test 8: Chat mit qwen-long triggert Wechsel auf long"
|
||||
echo "== Test 8: Chat mit qwen-large triggert Wechsel auf large"
|
||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-long","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
-d '{"model":"qwen-large","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["model"]=="mock-model-131072", d
|
||||
' && ok "qwen-long hat Profil long aktiviert und weitergeleitet" || bad "virtuelles Modell"
|
||||
assert d["model"]=="mock-model-192000", d
|
||||
' && ok "qwen-large hat Profil large aktiviert und weitergeleitet" || bad "virtuelles Modell"
|
||||
|
||||
# --- 9. Methoden und ungültiges virtuelles Modell -----------------------------------------
|
||||
echo "== Test 9: sichere Profilmethoden + ungültiges virtuelles Modell"
|
||||
CODE=$(curl -s -o /tmp/err9.json -w "%{http_code}" "$BASE/long")
|
||||
CODE=$(curl -s -o /tmp/err9.json -w "%{http_code}" "$BASE/large")
|
||||
cat /tmp/err9.json; echo
|
||||
[ "$CODE" = "405" ] && ok "GET /long verändert kein Profil" || bad "erwartet 405, bekam $CODE"
|
||||
[ "$CODE" = "405" ] && ok "GET /large verändert kein Profil" || bad "erwartet 405, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err9b.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
|
||||
-H "Content-Type: application/json" -d '{"model":"qwen-huge","messages":[]}')
|
||||
@@ -243,7 +244,7 @@ CODE=$(curl -s -o /tmp/chat9e.json -w "%{http_code}" "$BASE/v1/chat/completions"
|
||||
|
||||
# --- 10. llama.cpp down -> 502, danach Recovery ---------------------------------------------------
|
||||
echo "== Test 10: Upstream down -> 502, danach Recovery"
|
||||
# Profil auf fast setzen (aus Test 8 ist long aktiv)
|
||||
# Profil auf fast setzen (aus Test 8 ist large aktiv)
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
# Mock stoppen (simuliert Crash) – über Fake-systemctl
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
@@ -423,14 +424,14 @@ sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "medium" ] && ok "Medium → Image → Medium" || bad "Profil nach Image: $PROFILE (erwartet medium)"
|
||||
|
||||
# --- 24. Long → Image → Long ------------------------------------------------------------------------
|
||||
echo "== Test 24: Long → Image → Long"
|
||||
curl -sf -X POST "$BASE/long" >/dev/null
|
||||
# --- 24. Large → Image → Large ----------------------------------------------------------------------
|
||||
echo "== Test 24: Large → Image → Large"
|
||||
curl -sf -X POST "$BASE/large" >/dev/null
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"long test","size":"1024x1024"}')
|
||||
-d '{"prompt":"large test","size":"1024x1024"}')
|
||||
sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "long" ] && ok "Long → Image → Long" || bad "Profil nach Image: $PROFILE (erwartet long)"
|
||||
[ "$PROFILE" = "large" ] && ok "Large → Image → Large" || bad "Profil nach Image: $PROFILE (erwartet large)"
|
||||
|
||||
# --- 25. /status während Image-Job -------------------------------------------------------------------
|
||||
echo "== Test 25: /status während Image-Job"
|
||||
@@ -702,8 +703,8 @@ wait $STT_PID43
|
||||
# --- 44. Zwei konkurrierende Profilanfragen ------------------------------------------------
|
||||
echo "== Test 44: Profil-Lease verhindert Wechsel während eines Chats"
|
||||
curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-long","mock_delay":1.0,"messages":[{"role":"user","content":"Lang"}]}' \
|
||||
>/tmp/chat44-long.json &
|
||||
-d '{"model":"qwen-large","mock_delay":1.0,"messages":[{"role":"user","content":"Groß"}]}' \
|
||||
>/tmp/chat44-large.json &
|
||||
CHAT44_PID=$!
|
||||
sleep 0.2
|
||||
curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
@@ -712,10 +713,10 @@ curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
wait "$CHAT44_PID"
|
||||
python3 -c '
|
||||
import json
|
||||
long=json.load(open("/tmp/chat44-long.json"))
|
||||
large=json.load(open("/tmp/chat44-large.json"))
|
||||
medium=json.load(open("/tmp/chat44-medium.json"))
|
||||
assert long["mock_ctx"] == 131072, long
|
||||
assert medium["mock_ctx"] == 94208, medium
|
||||
assert large["mock_ctx"] == 192000, large
|
||||
assert medium["mock_ctx"] == 160000, medium
|
||||
' && ok "konkurrierende Chats behielten jeweils ihr Profil" \
|
||||
|| bad "Profil-Lease bei konkurrierenden Chats"
|
||||
|
||||
|
||||
@@ -20,7 +20,7 @@ Heimnetz / VPN-Clients
|
||||
+-- Profile Controller -- Docker Socket (feste Allowlist)
|
||||
+-- llama-fast --\
|
||||
+-- llama-medium > exakt einer aktiv
|
||||
+-- llama-long --/
|
||||
+-- llama-large --/
|
||||
+-- llama-experimental
|
||||
+-- llama-ultra (256K, text-only, dual GPU)
|
||||
+-- Piper-TTS (CPU, nur intern)
|
||||
@@ -44,8 +44,8 @@ Heimnetz / VPN-Clients
|
||||
| SearXNG/TinySearch | nein | Suchbackend des Web-MCP |
|
||||
|
||||
Nur der Profile Controller erhält den Docker-Socket. Der Router erhält weder
|
||||
Socket noch Shell-Zugriff und kann dem Controller nur `fast`, `medium`, `long`
|
||||
oder `experimental` übergeben. Die llama-Container laufen ohne UI,
|
||||
Socket noch Shell-Zugriff und kann dem Controller nur `fast`, `medium`, `large`,
|
||||
`ultra` oder `experimental` übergeben. Die llama-Container laufen ohne UI,
|
||||
Capabilities und Schreibzugriff auf die Modelldateien.
|
||||
|
||||
## Profilprinzip
|
||||
@@ -60,13 +60,13 @@ halten.
|
||||
| Profil | Ausgangswert | Zweck |
|
||||
|---|---:|---|
|
||||
| fast | 76.800 Kontext, MTP | mindestens ungefähr 80 Token/s anstreben |
|
||||
| medium | 94.208 Kontext | mehr Kontext ohne CPU-FFN-Offload |
|
||||
| long | 131.072 Kontext | maximale Nutzbarkeit, CPU-Offload erlaubt |
|
||||
| medium | 160.000 Kontext, Pure, 90:10, MTP3 | Standardprofil |
|
||||
| large | 192.000 Kontext, Pure, 86:14, MTP3 | große Agenten-/MCP-Sitzungen |
|
||||
| ultra | 262.144 Kontext, Pure, 80:20, MTP2 | maximaler Textkontext |
|
||||
| experimental | 76.800 Kontext | isolierte Tests ohne Produktion zu ändern |
|
||||
|
||||
Diese Werte sind reproduzierbare Startwerte, keine Garantie. Nach Einbau der
|
||||
RTX 3060 werden sie auf dem Zielhost erneut gemessen. Die zweite Karte wird
|
||||
nicht automatisch in die Produktionsprofile aufgenommen.
|
||||
Diese Matrix wurde auf RTX 5080 und RTX 3060 gemessen und ist bis zu einer
|
||||
bewussten Neubewertung der verbindliche Produktionsstandard.
|
||||
|
||||
## Netzwerk
|
||||
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
# Athena-Leerhostaufbau – Praxisprotokoll
|
||||
|
||||
> Dieses Dokument ist ein chronologisches Aufbauprotokoll. Darin genannte alte
|
||||
> Profilnamen und Kontextwerte sind keine aktuelle Konfiguration. Seit dem
|
||||
> 22. August 2026 gilt `STANDARD_PROFILE_MATRIX.md`.
|
||||
|
||||
Dieses Protokoll hält die Abweichungen fest, die beim realen Neuaufbau auf
|
||||
einem frischen Debian-13-Host sichtbar wurden. Jede dauerhaft notwendige
|
||||
Korrektur muss zusätzlich im Installer, Restore-Skript oder in der regulären
|
||||
|
||||
+23
-24
@@ -1,8 +1,8 @@
|
||||
# Aktueller produktiver Referenzstand
|
||||
|
||||
Stand: 20. August 2026. Dieses Dokument beschreibt die funktionierende
|
||||
Referenz vor dem geplanten Neuaufbau. Es ist keine Empfehlung, jede Altlast des
|
||||
Hosts zu übernehmen.
|
||||
Stand: 22. August 2026. Dieses Dokument beschreibt die auf Athena installierte
|
||||
und geprüfte Docker-Referenz. Die verbindlichen Profilparameter stehen in
|
||||
`STANDARD_PROFILE_MATRIX.md`.
|
||||
|
||||
## Hardware und Betriebssystem
|
||||
|
||||
@@ -12,10 +12,10 @@ Hosts zu übernehmen.
|
||||
| Kernel | 6.12.101+deb13-amd64 |
|
||||
| CPU | AMD Ryzen 5 5600, 6 Kerne/12 Threads |
|
||||
| RAM | 48 GiB DDR4-2666 |
|
||||
| GPU | NVIDIA GeForce RTX 5080, 16 GiB VRAM |
|
||||
| GPUs | NVIDIA GeForce RTX 5080, 16 GiB + RTX 3060, 12 GiB VRAM |
|
||||
| NVIDIA-Treiber | 610.57.04 |
|
||||
| System-SSD | Samsung 980 PRO 1 TB |
|
||||
| Daten-SSD | Samsung 980 PRO 2 TB |
|
||||
| Daten-SSD | WD Blue SN580 1 TB, unter `/data` |
|
||||
|
||||
Die früher verwendete Radeon RX 470 ist ausgebaut und gehört nicht zur
|
||||
Zielplattform.
|
||||
@@ -28,23 +28,22 @@ Zielplattform.
|
||||
| Repository | `https://github.com/ggml-org/llama.cpp.git` |
|
||||
| Commit | `4df29be4f4c3673f428170fda944a5b19f743bb8` |
|
||||
| Compiler | GCC 14.2 |
|
||||
| Hauptdienst | `mike-ai-llama-ui.service` |
|
||||
| llama.cpp-Port | 8080, auf dem alten Host noch im LAN gebunden |
|
||||
| Hauptdienst | jeweils ein Container `mike-ai-llama-<profil>` |
|
||||
| llama.cpp-Port | 8080, ausschließlich im internen Docker-Netz |
|
||||
| Client-Port | 8081 über den Router |
|
||||
| MCP-Konfiguration | getrennte Container unter `/opt/mike-ai/mcp-containers` |
|
||||
|
||||
### Aktives Fast-Profil
|
||||
### Aktives Standardprofil
|
||||
|
||||
- Qwen3.8-27B IQ4-MIX
|
||||
- Dateigröße: 14.111.614.400 Bytes
|
||||
- SHA256: `54879ae8738d5938f46cb3b8cbf16bf42b8c85b7d68d7c73f062b612ec183e36`
|
||||
- Öffentliche Herkunft ist noch nicht ausreichend dokumentiert; für eine
|
||||
bitgenaue Migration muss die geprüfte Datei vom Referenzhost gesichert werden.
|
||||
- Kontext 76.800
|
||||
- vollständig auf CUDA0
|
||||
- Qwen3.8-27B IQ4_XS Pure
|
||||
- Dateigröße: 14.534.384.640 Bytes
|
||||
- SHA256: `ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675`
|
||||
- Quelle: `jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF`
|
||||
- Kontext 160.000
|
||||
- RTX 5080 + RTX 3060 im Verhältnis 90:10
|
||||
- Flash Attention
|
||||
- KV-Cache Q4_0 für K und V
|
||||
- MTP Draft, maximal zwei Tokens
|
||||
- MTP Draft, maximal drei Tokens
|
||||
- sechs Threads und sechs Batch-Threads
|
||||
- Batch 64, Micro-Batch 32
|
||||
- ein paralleler Slot
|
||||
@@ -56,17 +55,17 @@ Zielplattform.
|
||||
| Profil | Virtuelles Modell | Kontext | Besonderheit |
|
||||
|---|---|---:|---|
|
||||
| Fast | `qwen-fast` | 76.800 | IQ4-MIX, MTP2, vollständig GPU, CPU-mmproj |
|
||||
| Medium | `qwen-medium` | 94.208 | IQ4_XS Pure, ohne MTP, CPU-mmproj |
|
||||
| Long | `qwen-long` | 131.072 | IQ4-MIX, MTP2, FFN 0–11 auf CPU, CPU-mmproj |
|
||||
| Medium **(Standard)** | `qwen-medium` | 160.000 | IQ4_XS Pure, MTP3, beide GPUs 90:10, CPU-mmproj |
|
||||
| Large | `qwen-large` | 192.000 | IQ4_XS Pure, MTP3, beide GPUs 86:14, CPU-mmproj |
|
||||
| Ultra | `qwen-ultra` | 262.144 | IQ4_XS Pure, MTP2, beide GPUs 80:20, text-only; 68,2 Tok/s und 220K-Fülltest bestanden |
|
||||
|
||||
## Router
|
||||
|
||||
- Dienst: `mike-ai-profile-router.service`
|
||||
- Container: `mike-ai-router`
|
||||
- Port: 8081
|
||||
- Upstream: `127.0.0.1:8080`
|
||||
- Upstream: `llama-upstream:8080` im internen Inferenznetz
|
||||
- Commit des Plattform-Repositories: siehe jeweils aktuelles `main`
|
||||
- Umschaltskript: `/usr/local/bin/llama-profile`
|
||||
- Umschaltung: `mike-ai-profile-controller` mit fester Container-Allowlist
|
||||
- Timeout für Profilwechsel und Requests: 600 Sekunden
|
||||
|
||||
Der Router übernimmt:
|
||||
@@ -74,7 +73,7 @@ Der Router übernimmt:
|
||||
- OpenAI-kompatibles Chat-Proxying und Streaming
|
||||
- virtuelle Modelle und automatische Profilumschaltung
|
||||
- Tool Calls
|
||||
- direkte integrierte Vision in allen Qwen-Profilen
|
||||
- direkte integrierte Vision in Fast, Medium und Large
|
||||
- FLUX-Hotswap zur Bildgenerierung
|
||||
- Whisper Speech-to-Text
|
||||
- XTTS Text-to-Speech (historische Referenz; Zielsystem verwendet Piper)
|
||||
@@ -87,9 +86,9 @@ Der Router übernimmt:
|
||||
| Text-/Visionmodell | jeweils aktives Qwen3.8-27B-Profil |
|
||||
| Projektor | BF16-mmproj |
|
||||
| Speicherort des Projektors | System-RAM (`--no-mmproj-offload`) |
|
||||
| Kontext | entspricht Fast/Medium/Long |
|
||||
| Kontext | entspricht Fast/Medium/Large; Ultra ist bewusst text-only |
|
||||
|
||||
Vision ist Bestandteil jedes Profils. Der Router prüft Bildgröße und URL,
|
||||
Vision ist Bestandteil von Fast, Medium und Large. Der Router prüft Bildgröße und URL,
|
||||
leitet das Bild dann direkt weiter und führt keinen Modellwechsel mehr aus.
|
||||
|
||||
## Bildgenerierung
|
||||
|
||||
@@ -37,7 +37,7 @@ laufen, sondern alle fachlichen Funktionen geprüft wurden.
|
||||
- [ ] Router-Key erscheint weder im Upstream noch im Journal
|
||||
- [ ] `/status` meldet den richtigen Upstream
|
||||
- [ ] `/v1/models` liefert drei virtuelle Modelle
|
||||
- [ ] `/fast`, `/medium` und `/long` wechseln zuverlässig
|
||||
- [ ] `/fast`, `/medium`, `/large` und `/ultra` wechseln zuverlässig
|
||||
- [ ] automatischer Wechsel über virtuellen Modellnamen funktioniert
|
||||
- [ ] paralleler Wechsel wird sauber gesperrt
|
||||
- [ ] Streaming funktioniert
|
||||
|
||||
+8
-8
@@ -36,8 +36,9 @@ Folgefragen (`task.follow_up.enable=false`). Dieselbe Vorgabe steht zusätzlich
|
||||
als Container-Umgebungswert im Compose-Stack, damit bereits eine frische
|
||||
OpenWebUI-Datenbank ohne Folgefragen startet.
|
||||
|
||||
Der Stabilitätsschutz kennt die drei Profilgrenzen 76.800, 94.208 und 131.072
|
||||
Token. Er reserviert Ausgabetoken und greift vor der harten llama.cpp-Grenze
|
||||
Der Stabilitätsschutz kennt die vier Profilgrenzen 76.800, 160.000, 192.000
|
||||
und 262.144 Token. Für unbekannte Modelle gilt Medium (160.000) als sichere
|
||||
Vorgabe. Er reserviert Ausgabetoken und greift vor der harten llama.cpp-Grenze
|
||||
ein. Bilder bleiben unangetastet; JSON-Werkzeugschemas werden niemals
|
||||
abgeschnitten. Sind allein die ausgewählten Schemas zu groß, wird der
|
||||
Werkzeugzugriff nur für diesen Schritt deaktiviert und das Modell erhält eine
|
||||
@@ -72,15 +73,14 @@ Community Store nachgeladen.
|
||||
| Profil | Virtuelles Modell | Kontext | Zweck |
|
||||
|---|---|---:|---|
|
||||
| Fast | `qwen-fast` | 76.800 | Alltag, Agenten, hohe Geschwindigkeit, integrierte Vision |
|
||||
| Medium | `qwen-medium` | 94.208 | mehr Kontext, reine IQ4_XS-Variante |
|
||||
| Long | `qwen-long` | 131.072 | lange Hermes-/MCP-Sitzungen |
|
||||
| Medium **(Standard)** | `qwen-medium` | 160.000 | IQ4_XS Pure, beide GPUs 90:10, MTP3, integrierte Vision |
|
||||
| Large | `qwen-large` | 192.000 | IQ4_XS Pure, beide GPUs 86:14, MTP3, integrierte Vision |
|
||||
| Ultra | `qwen-ultra` | 262.144 | maximaler Textkontext, IQ4_XS Pure auf RTX 5080 + RTX 3060 (80:20), ohne Vision-Projektor |
|
||||
|
||||
Manuell wird mit `llama-profile fast|medium|long|ultra` gewechselt. Über HTTP stehen
|
||||
`POST /fast`, `/medium`, `/long` und `/ultra` zur Verfügung. Ultra erreichte im
|
||||
Manuell wird mit `llama-profile fast|medium|large|ultra` gewechselt. Über HTTP stehen
|
||||
`POST /fast`, `/medium`, `/large` und `/ultra` zur Verfügung. Ultra erreichte im
|
||||
Referenzlauf etwa 68 Token/s; ein Prompt-Fülltest mit rund 220.000 Tokens war
|
||||
erfolgreich. Für eine spätere Version ist
|
||||
`large` als Alias für `long` vorgesehen; bestehende Namen bleiben kompatibel.
|
||||
erfolgreich. Medium ist das Start- und Standardprofil.
|
||||
|
||||
## Clients
|
||||
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
# Router V2 – Migration und Kompatibilität
|
||||
|
||||
> Historischer Stand: Die damalige Bezeichnung `long` wurde am 22. August
|
||||
> 2026 durch `large` ersetzt und um `ultra` ergänzt. Für den aktuellen Betrieb
|
||||
> gilt ausschließlich `STANDARD_PROFILE_MATRIX.md`.
|
||||
|
||||
## Ergebnis
|
||||
|
||||
V2 behält die OpenAI-kompatible Basis-URL und die virtuellen Modelle
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
# Verbindliche Standard-Profilmatrix
|
||||
|
||||
Stand: 22. August 2026. Diese vier Profile sind die Produktionsmatrix für
|
||||
Athena. Änderungen gelten erst nach einem vergleichbaren synthetischen Test
|
||||
und einer bewussten Aktualisierung dieser Datei.
|
||||
|
||||
| Profil | Virtuelles Modell | GGUF | Kontext | GPUs / Split | MTP | Vision | gemessene kurze Ausgabe |
|
||||
|---|---|---|---:|---|---:|---|---:|
|
||||
| Fast | `qwen-fast` | IQ4-MIX | 76.800 | RTX 5080 | 2 | ja, Projektor auf CPU | 85,5 Tok/s |
|
||||
| **Medium (Default)** | `qwen-medium` | IQ4_XS Pure | 160.000 | RTX 5080 + RTX 3060, 90:10 | 3 | ja, Projektor auf CPU | 77,2 Tok/s |
|
||||
| Large | `qwen-large` | IQ4_XS Pure | 192.000 | RTX 5080 + RTX 3060, 86:14 | 3 | ja, Projektor auf CPU | 75,3 Tok/s |
|
||||
| Ultra | `qwen-ultra` | IQ4_XS Pure | 262.144 | RTX 5080 + RTX 3060, 80:20 | 2 | nein, text-only | 68,2 Tok/s |
|
||||
|
||||
## Standardverhalten
|
||||
|
||||
- Medium ist nach Neuinstallation und bewusstem Plattformstart das aktive
|
||||
Standardprofil.
|
||||
- Open WebUI erhält `qwen-medium` als Standardmodell.
|
||||
- Ein explizit gewähltes anderes Modell löst den zugehörigen Containerwechsel
|
||||
aus; es ist immer nur ein Inferenzcontainer aktiv.
|
||||
- Ultra reserviert den verfügbaren Speicher für nativen 256K-Textkontext und
|
||||
lädt deshalb keinen Vision-Projektor.
|
||||
- Das gesonderte Experimentalprofil gehört nicht zur Benutzer-Matrix und wird
|
||||
in Open WebUI nicht als reguläres Modell angeboten.
|
||||
|
||||
## Nachweise
|
||||
|
||||
- Fast 76,8K: synthetischer Referenzlauf, 85,50 Tok/s.
|
||||
- Medium 160K: Pure 90:10 mit MTP3, 77,22 Tok/s.
|
||||
- Large 192K: Pure 86:14 mit MTP3, 75,28 Tok/s.
|
||||
- Ultra 256K: Pure 80:20 mit MTP2, 68,19 Tok/s; 220.190 Tokens
|
||||
erfolgreich verarbeitet und Sentinel korrekt wiedergefunden.
|
||||
|
||||
+15
-13
@@ -40,7 +40,7 @@ source "$CONFIG"
|
||||
|
||||
required=(AI_HOSTNAME ADMIN_USER AI_BIND_ADDRESS MODEL_DIR FAST_MODEL_FILE
|
||||
FAST_MODEL_URL FAST_MODEL_SHA256 MEDIUM_MODEL_FILE MEDIUM_MODEL_URL
|
||||
MEDIUM_MODEL_SHA256 LONG_MODEL_FILE LONG_MODEL_URL LONG_MODEL_SHA256
|
||||
MEDIUM_MODEL_SHA256 LARGE_MODEL_FILE LARGE_MODEL_URL LARGE_MODEL_SHA256
|
||||
ULTRA_MODEL_FILE ULTRA_MODEL_URL ULTRA_MODEL_SHA256
|
||||
EXPERIMENTAL_MODEL_FILE EXPERIMENTAL_MODEL_URL EXPERIMENTAL_MODEL_SHA256
|
||||
VISION_PROJECTOR_FILE VISION_PROJECTOR_URL VISION_PROJECTOR_SHA256)
|
||||
@@ -209,18 +209,20 @@ PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
|
||||
AI_DNS=${WG_DNS:-1.1.1.1}
|
||||
FAST_MODEL_FILE=$FAST_MODEL_FILE
|
||||
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
|
||||
LONG_MODEL_FILE=$LONG_MODEL_FILE
|
||||
LARGE_MODEL_FILE=$LARGE_MODEL_FILE
|
||||
ULTRA_MODEL_FILE=$ULTRA_MODEL_FILE
|
||||
EXPERIMENTAL_MODEL_FILE=$EXPERIMENTAL_MODEL_FILE
|
||||
VISION_PROJECTOR_FILE=$VISION_PROJECTOR_FILE
|
||||
FAST_CONTEXT=${FAST_CONTEXT:-76800}
|
||||
MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-94208}
|
||||
LONG_CONTEXT=${LONG_CONTEXT:-131072}
|
||||
MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-160000}
|
||||
LARGE_CONTEXT=${LARGE_CONTEXT:-192000}
|
||||
ULTRA_CONTEXT=${ULTRA_CONTEXT:-262144}
|
||||
EXPERIMENTAL_CONTEXT=${EXPERIMENTAL_CONTEXT:-76800}
|
||||
FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
||||
MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
||||
LONG_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
||||
MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||
MEDIUM_TENSOR_SPLIT=${MEDIUM_TENSOR_SPLIT:-90,10}
|
||||
LARGE_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||
LARGE_TENSOR_SPLIT=${LARGE_TENSOR_SPLIT:-86,14}
|
||||
ULTRA_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
|
||||
ULTRA_TENSOR_SPLIT=${ULTRA_TENSOR_SPLIT:-80,20}
|
||||
EXPERIMENTAL_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}
|
||||
@@ -260,7 +262,7 @@ download_models() {
|
||||
done <<EOF
|
||||
$FAST_MODEL_FILE|$FAST_MODEL_URL|$FAST_MODEL_SHA256
|
||||
$MEDIUM_MODEL_FILE|$MEDIUM_MODEL_URL|$MEDIUM_MODEL_SHA256
|
||||
$LONG_MODEL_FILE|$LONG_MODEL_URL|$LONG_MODEL_SHA256
|
||||
$LARGE_MODEL_FILE|$LARGE_MODEL_URL|$LARGE_MODEL_SHA256
|
||||
$ULTRA_MODEL_FILE|$ULTRA_MODEL_URL|$ULTRA_MODEL_SHA256
|
||||
$EXPERIMENTAL_MODEL_FILE|$EXPERIMENTAL_MODEL_URL|$EXPERIMENTAL_MODEL_SHA256
|
||||
$VISION_PROJECTOR_FILE|$VISION_PROJECTOR_URL|$VISION_PROJECTOR_SHA256
|
||||
@@ -329,7 +331,7 @@ build_and_start() {
|
||||
# secret files and required local artifacts are present.
|
||||
"$STACK_DIR/platform/mcp/install-tools.sh"
|
||||
docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \
|
||||
llama-fast llama-medium llama-long llama-ultra llama-experimental
|
||||
llama-fast llama-medium llama-large llama-ultra llama-experimental
|
||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
||||
profile-controller router open-webui
|
||||
|
||||
@@ -343,7 +345,7 @@ build_and_start() {
|
||||
sleep 2
|
||||
done
|
||||
|
||||
log "Fast-Profil aktivieren und Readiness prüfen"
|
||||
log "Medium-Profil als Standard aktivieren und Readiness prüfen"
|
||||
# Use docker exec directly here. Some Compose/Docker combinations return a
|
||||
# transient HTTP 409 while upgrading the exec stream immediately after a
|
||||
# freshly built service has been recreated.
|
||||
@@ -352,7 +354,7 @@ build_and_start() {
|
||||
import json, os, time, urllib.request
|
||||
key = os.environ["ROUTER_API_KEY"]
|
||||
request = urllib.request.Request(
|
||||
"http://127.0.0.1:8081/fast", method="POST",
|
||||
"http://127.0.0.1:8081/medium", method="POST",
|
||||
headers={"Authorization": f"Bearer {key}"})
|
||||
with urllib.request.urlopen(request, timeout=700) as response:
|
||||
print(json.dumps(json.load(response), indent=2))
|
||||
@@ -361,7 +363,7 @@ while time.monotonic() < deadline:
|
||||
try:
|
||||
with urllib.request.urlopen("http://127.0.0.1:8081/ready", timeout=5) as response:
|
||||
if response.status == 200:
|
||||
print("Router und Fast-Profil sind bereit.")
|
||||
print("Router und Medium-Standardprofil sind bereit.")
|
||||
print("INSTALL_READINESS_OK")
|
||||
break
|
||||
except Exception:
|
||||
@@ -370,10 +372,10 @@ while time.monotonic() < deadline:
|
||||
else:
|
||||
raise SystemExit("Readiness-Check fehlgeschlagen")
|
||||
PY
|
||||
) || die "Fast-Profil konnte nicht aktiviert werden"
|
||||
) || die "Medium-Standardprofil konnte nicht aktiviert werden"
|
||||
printf '%s\n' "$activation_output"
|
||||
grep -Fxq 'INSTALL_READINESS_OK' <<<"$activation_output" || \
|
||||
die "Fast-Profil lieferte keinen bestätigten Readiness-Marker"
|
||||
die "Medium-Standardprofil lieferte keinen bestätigten Readiness-Marker"
|
||||
}
|
||||
|
||||
hostnamectl set-hostname "$AI_HOSTNAME"
|
||||
|
||||
@@ -40,7 +40,7 @@ for container in mike-ai-profile-controller mike-ai-router mike-ai-piper mike-ai
|
||||
fi
|
||||
done
|
||||
|
||||
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|long|ultra|experimental)$' || true)"
|
||||
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|large|ultra|experimental)$' || true)"
|
||||
if [[ $ACTIVE_LLAMA -eq 1 ]]; then
|
||||
pass "exakt ein llama.cpp-Profil aktiv"
|
||||
else
|
||||
|
||||
@@ -21,7 +21,7 @@ PORT = int(os.environ.get("CONTROLLER_PORT", "8090"))
|
||||
SOCKET_PATH = os.environ.get("DOCKER_SOCKET", "/var/run/docker.sock")
|
||||
TOKEN_FILE = os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token")
|
||||
ALLOWED = tuple(x.strip() for x in os.environ.get(
|
||||
"ALLOWED_PROFILES", "fast,medium,long,ultra,experimental").split(",") if x.strip())
|
||||
"ALLOWED_PROFILES", "fast,medium,large,ultra,experimental").split(",") if x.strip())
|
||||
LABEL_KEY = "com.mike-ai.llama-profile"
|
||||
LOCK = threading.Lock()
|
||||
log = logging.getLogger("profile-controller")
|
||||
|
||||
@@ -21,7 +21,8 @@ install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \
|
||||
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
|
||||
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-large.conf" "$PROFILE_TARGET/profile-large.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-ultra.conf" "$PROFILE_TARGET/profile-ultra.conf"
|
||||
rsync -a --delete "$PLATFORM/mcp/" "$MCP_TARGET/"
|
||||
rsync -a --delete --exclude searxng-settings.yml \
|
||||
"$PLATFORM/web-search/" "$MCP_SEARCH_TARGET/"
|
||||
@@ -39,7 +40,7 @@ Kernkonfiguration installiert, aber noch nicht gestartet.
|
||||
|
||||
Vor dem Start:
|
||||
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
|
||||
2. llama.cpp bauen und mit `llama-profile fast` starten.
|
||||
2. llama.cpp bauen und mit `llama-profile medium` starten.
|
||||
3. Fach-Secrets unter /etc/mike-ai ablegen.
|
||||
4. Danach: /opt/mike-ai/mcp-containers/platform/mcp/install-tools.sh
|
||||
EOF
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
schema: 1
|
||||
models:
|
||||
qwen_fast_long:
|
||||
role: primary-text-fast-and-long
|
||||
role: primary-text-fast
|
||||
source: "local migration from the reference host; public origin still to document"
|
||||
file: Qwen3.8-27B-IQ4-MIX.gguf
|
||||
target: /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||
|
||||
@@ -16,9 +16,10 @@ class Filter:
|
||||
class Valves(BaseModel):
|
||||
priority: int = 30
|
||||
fast_context_tokens: int = 76800
|
||||
medium_context_tokens: int = 94208
|
||||
long_context_tokens: int = 131072
|
||||
default_context_tokens: int = 76800
|
||||
medium_context_tokens: int = 160000
|
||||
large_context_tokens: int = 192000
|
||||
ultra_context_tokens: int = 262144
|
||||
default_context_tokens: int = 160000
|
||||
soft_context_ratio: float = 0.70
|
||||
hard_context_ratio: float = 0.84
|
||||
reserved_output_tokens: int = 8192
|
||||
@@ -107,8 +108,10 @@ class Filter:
|
||||
|
||||
def _context_limit(self, model: str) -> int:
|
||||
model = (model or "").lower()
|
||||
if "long" in model or "large" in model:
|
||||
return self.valves.long_context_tokens
|
||||
if "ultra" in model:
|
||||
return self.valves.ultra_context_tokens
|
||||
if "large" in model:
|
||||
return self.valves.large_context_tokens
|
||||
if "medium" in model:
|
||||
return self.valves.medium_context_tokens
|
||||
if "fast" in model:
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
[Unit]
|
||||
Description=Legacy native Qwen Large 192K profile (Docker is the production path)
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
@@ -1,6 +0,0 @@
|
||||
[Unit]
|
||||
Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU with CPU Vision
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
|
||||
@@ -1,6 +1,6 @@
|
||||
[Unit]
|
||||
Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure with CPU Vision
|
||||
Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
[Unit]
|
||||
Description=Legacy native Qwen Ultra 256K profile (Docker is the production path)
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen-ultra --ctx-size 262144 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 80,20 --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
@@ -7,10 +7,9 @@ SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
|
||||
PROFILE="${1:-}"
|
||||
|
||||
case "$PROFILE" in
|
||||
fast|medium|long) ;;
|
||||
large) PROFILE=long ;;
|
||||
fast|medium|large|ultra) ;;
|
||||
*)
|
||||
echo "Usage: llama-profile {fast|medium|long|large}" >&2
|
||||
echo "Usage: llama-profile {fast|medium|large|ultra}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
@@ -8,12 +8,12 @@ Profilen um:
|
||||
Profil Kontext
|
||||
------ --------
|
||||
fast 76800
|
||||
medium 94208
|
||||
long 131072
|
||||
medium 160000
|
||||
large 192000
|
||||
ultra 262144
|
||||
|
||||
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-long, qwen-ultra
|
||||
Kommandos: POST /fast, /medium, /long, /ultra (Profilwechsel)
|
||||
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large, qwen-ultra
|
||||
Kommandos: POST /fast, /medium, /large, /ultra (Profilwechsel)
|
||||
GET /status (Zustand)
|
||||
|
||||
Bildgenerierung (FLUX.2 [klein] 4B Base):
|
||||
@@ -1110,11 +1110,11 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self._images_list()
|
||||
elif path.startswith("/images/") and self.command == "GET":
|
||||
self._image_serve(path[len("/images/"):])
|
||||
elif (path in ("/fast", "/medium", "/long", "/ultra")
|
||||
elif (path in ("/fast", "/medium", "/large", "/ultra")
|
||||
and (self.command == "POST"
|
||||
or (self.command == "GET" and ALLOW_LEGACY_GET_SWITCH))):
|
||||
self._switch(path[1:])
|
||||
elif (path in ("/fast", "/medium", "/long", "/ultra")
|
||||
elif (path in ("/fast", "/medium", "/large", "/ultra")
|
||||
and self.command == "GET"):
|
||||
self._send_error(405, "Profilwechsel erfordert POST",
|
||||
"invalid_request_error", "method_not_allowed")
|
||||
|
||||
@@ -5,12 +5,12 @@
|
||||
"model_alias": "qwen-fast"
|
||||
},
|
||||
"medium": {
|
||||
"context": 94208,
|
||||
"context": 160000,
|
||||
"model_alias": "qwen-medium"
|
||||
},
|
||||
"long": {
|
||||
"context": 131072,
|
||||
"model_alias": "qwen-long"
|
||||
"large": {
|
||||
"context": 192000,
|
||||
"model_alias": "qwen-large"
|
||||
},
|
||||
"ultra": {
|
||||
"context": 262144,
|
||||
|
||||
@@ -34,8 +34,8 @@ def load_profile_registry(path: str | None) -> dict[str, dict]:
|
||||
|
||||
fallback = {
|
||||
"fast": {"context": 76800, "model_alias": None},
|
||||
"medium": {"context": 94208, "model_alias": None},
|
||||
"long": {"context": 131072, "model_alias": None},
|
||||
"medium": {"context": 160000, "model_alias": None},
|
||||
"large": {"context": 192000, "model_alias": None},
|
||||
"ultra": {"context": 262144, "model_alias": None},
|
||||
}
|
||||
if not path:
|
||||
|
||||
Reference in New Issue
Block a user