Define four-profile production matrix with Medium default
This commit is contained in:
@@ -40,7 +40,7 @@ for container in mike-ai-profile-controller mike-ai-router mike-ai-piper mike-ai
|
||||
fi
|
||||
done
|
||||
|
||||
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|long|ultra|experimental)$' || true)"
|
||||
ACTIVE_LLAMA="$(docker ps --format '{{.Names}}' | grep -Ec '^mike-ai-llama-(fast|medium|large|ultra|experimental)$' || true)"
|
||||
if [[ $ACTIVE_LLAMA -eq 1 ]]; then
|
||||
pass "exakt ein llama.cpp-Profil aktiv"
|
||||
else
|
||||
|
||||
@@ -21,7 +21,7 @@ PORT = int(os.environ.get("CONTROLLER_PORT", "8090"))
|
||||
SOCKET_PATH = os.environ.get("DOCKER_SOCKET", "/var/run/docker.sock")
|
||||
TOKEN_FILE = os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token")
|
||||
ALLOWED = tuple(x.strip() for x in os.environ.get(
|
||||
"ALLOWED_PROFILES", "fast,medium,long,ultra,experimental").split(",") if x.strip())
|
||||
"ALLOWED_PROFILES", "fast,medium,large,ultra,experimental").split(",") if x.strip())
|
||||
LABEL_KEY = "com.mike-ai.llama-profile"
|
||||
LOCK = threading.Lock()
|
||||
log = logging.getLogger("profile-controller")
|
||||
|
||||
@@ -21,7 +21,8 @@ install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \
|
||||
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
|
||||
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-large.conf" "$PROFILE_TARGET/profile-large.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-ultra.conf" "$PROFILE_TARGET/profile-ultra.conf"
|
||||
rsync -a --delete "$PLATFORM/mcp/" "$MCP_TARGET/"
|
||||
rsync -a --delete --exclude searxng-settings.yml \
|
||||
"$PLATFORM/web-search/" "$MCP_SEARCH_TARGET/"
|
||||
@@ -39,7 +40,7 @@ Kernkonfiguration installiert, aber noch nicht gestartet.
|
||||
|
||||
Vor dem Start:
|
||||
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
|
||||
2. llama.cpp bauen und mit `llama-profile fast` starten.
|
||||
2. llama.cpp bauen und mit `llama-profile medium` starten.
|
||||
3. Fach-Secrets unter /etc/mike-ai ablegen.
|
||||
4. Danach: /opt/mike-ai/mcp-containers/platform/mcp/install-tools.sh
|
||||
EOF
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
schema: 1
|
||||
models:
|
||||
qwen_fast_long:
|
||||
role: primary-text-fast-and-long
|
||||
role: primary-text-fast
|
||||
source: "local migration from the reference host; public origin still to document"
|
||||
file: Qwen3.8-27B-IQ4-MIX.gguf
|
||||
target: /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||
|
||||
@@ -16,9 +16,10 @@ class Filter:
|
||||
class Valves(BaseModel):
|
||||
priority: int = 30
|
||||
fast_context_tokens: int = 76800
|
||||
medium_context_tokens: int = 94208
|
||||
long_context_tokens: int = 131072
|
||||
default_context_tokens: int = 76800
|
||||
medium_context_tokens: int = 160000
|
||||
large_context_tokens: int = 192000
|
||||
ultra_context_tokens: int = 262144
|
||||
default_context_tokens: int = 160000
|
||||
soft_context_ratio: float = 0.70
|
||||
hard_context_ratio: float = 0.84
|
||||
reserved_output_tokens: int = 8192
|
||||
@@ -107,8 +108,10 @@ class Filter:
|
||||
|
||||
def _context_limit(self, model: str) -> int:
|
||||
model = (model or "").lower()
|
||||
if "long" in model or "large" in model:
|
||||
return self.valves.long_context_tokens
|
||||
if "ultra" in model:
|
||||
return self.valves.ultra_context_tokens
|
||||
if "large" in model:
|
||||
return self.valves.large_context_tokens
|
||||
if "medium" in model:
|
||||
return self.valves.medium_context_tokens
|
||||
if "fast" in model:
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
[Unit]
|
||||
Description=Legacy native Qwen Large 192K profile (Docker is the production path)
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-large --ctx-size 192000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 86,14 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
@@ -1,6 +0,0 @@
|
||||
[Unit]
|
||||
Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU with CPU Vision
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
|
||||
@@ -1,6 +1,6 @@
|
||||
[Unit]
|
||||
Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure with CPU Vision
|
||||
Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
[Unit]
|
||||
Description=Legacy native Qwen Ultra 256K profile (Docker is the production path)
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen-ultra --ctx-size 262144 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 80,20 --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
@@ -7,10 +7,9 @@ SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
|
||||
PROFILE="${1:-}"
|
||||
|
||||
case "$PROFILE" in
|
||||
fast|medium|long) ;;
|
||||
large) PROFILE=long ;;
|
||||
fast|medium|large|ultra) ;;
|
||||
*)
|
||||
echo "Usage: llama-profile {fast|medium|long|large}" >&2
|
||||
echo "Usage: llama-profile {fast|medium|large|ultra}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
Reference in New Issue
Block a user