Expand router into reproducible local AI platform

This commit is contained in:
Mikei386
2026-08-20 12:56:53 +02:00
parent a84220725a
commit 0e4a9de5ba
34 changed files with 2584 additions and 31 deletions
+6
View File
@@ -0,0 +1,6 @@
[Unit]
Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none