Files
AI-Profile-Router/platform/profiles/profile-fast.conf
T

7 lines
715 B
Plaintext

[Unit]
Description=Local AI llama.cpp - Qwen Fast 72K MTP2
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --alias qwen38-27b-iq4mix-72k-mtp2 --ctx-size 73728 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0