Expand router into reproducible local AI platform
This commit is contained in:
1 parent
a84220725a
commit
0e4a9de5ba
34 files changed
+2584
-31
No files matched your search
+27
@@ -1,3 +1,30 @@
|
|||||||
__pycache__/
|
__pycache__/
|
||||||
*.pyc
|
*.pyc
|
||||||
.DS_Store
|
.DS_Store
|
||||||
|
.env
|
||||||
|
.env.*
|
||||||
|
!.env.example
|
||||||
|
*.pem
|
||||||
|
*.key
|
||||||
|
*.crt
|
||||||
|
*.p12
|
||||||
|
*.pfx
|
||||||
|
manifest.local.yaml
|
||||||
|
models/
|
||||||
|
!platform/models/
|
||||||
|
!platform/models/manifest.example.yaml
|
||||||
|
images/
|
||||||
|
samples/
|
||||||
|
logs/
|
||||||
|
*.log
|
||||||
|
*.wav
|
||||||
|
*.mp3
|
||||||
|
*.png
|
||||||
|
*.jpg
|
||||||
|
*.jpeg
|
||||||
|
*.webp
|
||||||
|
*.gguf
|
||||||
|
*.safetensors
|
||||||
|
*.bin
|
||||||
|
.venv/
|
||||||
|
venv/
|
||||||
@@ -1,4 +1,59 @@
|
|||||||
# AI Profile Router
|
# Lokale KI-Plattform
|
||||||
|
|
||||||
|
Dieses private Repository dokumentiert und installiert die reproduzierbare
|
||||||
|
KI-Umgebung rund um einen lokalen `llama.cpp`-Host. Der **AI Profile Router**
|
||||||
|
bleibt der zentrale Bestandteil, ist aber nicht mehr die einzige Komponente.
|
||||||
|
|
||||||
|
## Plattform auf einen Blick
|
||||||
|
|
||||||
|
```text
|
||||||
|
Clients (Open WebUI, Hermes, Apps)
|
||||||
|
|
|
||||||
|
v
|
||||||
|
AI Profile Router :8081
|
||||||
|
|-- qwen-fast (72K, maximale Geschwindigkeit)
|
||||||
|
|-- qwen-medium (92K, reines IQ4_XS)
|
||||||
|
|-- qwen-long (128K, CPU-Offload)
|
||||||
|
|-- Vision-Hotswap
|
||||||
|
|-- lokale Bildgenerierung
|
||||||
|
|-- Whisper STT
|
||||||
|
`-- XTTS TTS
|
||||||
|
|
|
||||||
|
v
|
||||||
|
llama.cpp :8080 + profilabhängige MCP-Server
|
||||||
|
```
|
||||||
|
|
||||||
|
### Enthalten
|
||||||
|
|
||||||
|
- produktiver Router samt Tests und Deployment
|
||||||
|
- fest definierte Fast-/Medium-/Long-Profile
|
||||||
|
- reproduzierbarer `llama.cpp`-Build über einen festgelegten Commit
|
||||||
|
- systemd-Vorlagen und Profilumschaltung
|
||||||
|
- Modellmanifest ohne Modelldateien
|
||||||
|
- sichere MCP-Beispielkonfiguration ohne Zugangsdaten
|
||||||
|
- Installations-, Betriebs-, Sicherheits- und Migrationsanleitung
|
||||||
|
- Prüfskript für einen frisch installierten Host
|
||||||
|
|
||||||
|
### Bewusst nicht enthalten
|
||||||
|
|
||||||
|
- GGUF-, Whisper-, FLUX- oder XTTS-Modelldateien
|
||||||
|
- API-Schlüssel, Tokens, SSH-Schlüssel oder Zertifikate
|
||||||
|
- Chats, Prompts, Logs, Bilder oder Audiodateien
|
||||||
|
- alte Benchmarks, experimentelle Builds und ausgemusterte RX-Dienste
|
||||||
|
- hostgebundene Backups und Cache-Verzeichnisse
|
||||||
|
|
||||||
|
## Dokumentation
|
||||||
|
|
||||||
|
- [Architektur](docs/ARCHITECTURE.md)
|
||||||
|
- [Komponentenverzeichnis](docs/COMPONENTS.md)
|
||||||
|
- [Saubere Installation](docs/INSTALLATION.md)
|
||||||
|
- [Betrieb und Profilwechsel](docs/OPERATIONS.md)
|
||||||
|
- [Sicherheitsmodell](docs/SECURITY.md)
|
||||||
|
- [Migration vom bestehenden Host](docs/MIGRATION.md)
|
||||||
|
- [MCP-Aufteilung](platform/mcp/README.md)
|
||||||
|
- [llama.cpp-Build und Profile](platform/llama/README.md)
|
||||||
|
|
||||||
|
## AI Profile Router
|
||||||
|
|
||||||
Kleiner OpenAI-kompatibler Proxy (Python, nur Standardbibliothek) vor einem
|
Kleiner OpenAI-kompatibler Proxy (Python, nur Standardbibliothek) vor einem
|
||||||
lokalen llama.cpp-Server. Er leitet normale OpenAI-Requests transparent
|
lokalen llama.cpp-Server. Er leitet normale OpenAI-Requests transparent
|
||||||
@@ -12,8 +67,8 @@ Sprachausgabe bereit (XTTS-v2, CPU-only, OpenAI-kompatibel).
|
|||||||
|
|
||||||
| | |
|
| | |
|
||||||
|---|---|
|
|---|---|
|
||||||
| Host | 192.168.1.196 |
|
| Host | frei wählbarer Linux-KI-Host |
|
||||||
| SSH | `root` mit Key `lmstudio_unraid` |
|
| SSH | administrativer Zugang nur für Installation und Wartung |
|
||||||
| llama.cpp | `http://127.0.0.1:8080` (Service `mike-ai-llama-ui.service`) |
|
| llama.cpp | `http://127.0.0.1:8080` (Service `mike-ai-llama-ui.service`) |
|
||||||
| Profil-Skript | `/usr/local/bin/llama-profile {fast\|medium\|long}` |
|
| Profil-Skript | `/usr/local/bin/llama-profile {fast\|medium\|long}` |
|
||||||
| Router-Port | **8081** |
|
| Router-Port | **8081** |
|
||||||
@@ -100,7 +155,7 @@ OpenAI-kompatibel. Unterstützt `prompt`, `size`, `n`, `seed`, `quality`,
|
|||||||
Beispiel:
|
Beispiel:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
curl -s http://192.168.1.196:8081/v1/images/generations \
|
curl -s http://AI_HOST:8081/v1/images/generations \
|
||||||
-H 'Content-Type: application/json' \
|
-H 'Content-Type: application/json' \
|
||||||
-d '{"prompt":"ein roter Würfel auf weißem Grund","size":"1024x1024","quality":"standard"}'
|
-d '{"prompt":"ein roter Würfel auf weißem Grund","size":"1024x1024","quality":"standard"}'
|
||||||
```
|
```
|
||||||
@@ -247,12 +302,12 @@ Beispiele:
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
# MP3 (Default), Stimme claribel
|
# MP3 (Default), Stimme claribel
|
||||||
curl -s http://192.168.1.196:8081/v1/audio/speech \
|
curl -s http://AI_HOST:8081/v1/audio/speech \
|
||||||
-H 'Content-Type: application/json' \
|
-H 'Content-Type: application/json' \
|
||||||
-d '{"input":"Hallo, dies ist ein Test.","voice":"claribel"}' -o out.mp3
|
-d '{"input":"Hallo, dies ist ein Test.","voice":"claribel"}' -o out.mp3
|
||||||
|
|
||||||
# WAV, 1.5x Tempo
|
# WAV, 1.5x Tempo
|
||||||
curl -s http://192.168.1.196:8081/v1/audio/speech \
|
curl -s http://AI_HOST:8081/v1/audio/speech \
|
||||||
-H 'Content-Type: application/json' \
|
-H 'Content-Type: application/json' \
|
||||||
-d '{"input":"Guten Tag.","voice":"claribel","speed":1.5,"response_format":"wav"}' -o out.wav
|
-d '{"input":"Guten Tag.","voice":"claribel","speed":1.5,"response_format":"wav"}' -o out.wav
|
||||||
```
|
```
|
||||||
@@ -341,18 +396,18 @@ Beispiele:
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
# WebM/Opus (z.B. aus Open WebUI-Mikrofon)
|
# WebM/Opus (z.B. aus Open WebUI-Mikrofon)
|
||||||
curl -s http://192.168.1.196:8081/v1/audio/transcriptions \
|
curl -s http://AI_HOST:8081/v1/audio/transcriptions \
|
||||||
-F "file=@aufnahme.webm" \
|
-F "file=@aufnahme.webm" \
|
||||||
-F "model=whisper-1"
|
-F "model=whisper-1"
|
||||||
|
|
||||||
# WAV mit expliziter Sprache
|
# WAV mit expliziter Sprache
|
||||||
curl -s http://192.168.1.196:8081/v1/audio/transcriptions \
|
curl -s http://AI_HOST:8081/v1/audio/transcriptions \
|
||||||
-F "file=@aufnahme.wav" \
|
-F "file=@aufnahme.wav" \
|
||||||
-F "model=whisper-1" \
|
-F "model=whisper-1" \
|
||||||
-F "language=de"
|
-F "language=de"
|
||||||
|
|
||||||
# Verbose-Format
|
# Verbose-Format
|
||||||
curl -s http://192.168.1.196:8081/v1/audio/transcriptions \
|
curl -s http://AI_HOST:8081/v1/audio/transcriptions \
|
||||||
-F "file=@aufnahme.wav" \
|
-F "file=@aufnahme.wav" \
|
||||||
-F "model=whisper-1" \
|
-F "model=whisper-1" \
|
||||||
-F "response_format=verbose_json"
|
-F "response_format=verbose_json"
|
||||||
@@ -415,7 +470,8 @@ sind getrennt. Auf dem Zielsystem landet nur `router/` + `deploy/`.
|
|||||||
|
|
||||||
## Deployment
|
## Deployment
|
||||||
|
|
||||||
Voraussetzung: SSH-Key `~/.ssh/lmstudio_unraid` (bereits vorhanden).
|
Voraussetzung: administrativer SSH-Key für den Zielhost. Ziel und Key werden
|
||||||
|
explizit über `TARGET` und `SSH_KEY` übergeben.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
./deploy/deploy.sh
|
./deploy/deploy.sh
|
||||||
@@ -517,10 +573,10 @@ systemctl status mike-ai-profile-router
|
|||||||
systemctl status mike-ai-xtts
|
systemctl status mike-ai-xtts
|
||||||
journalctl -u mike-ai-profile-router -f
|
journalctl -u mike-ai-profile-router -f
|
||||||
journalctl -u mike-ai-xtts -f
|
journalctl -u mike-ai-xtts -f
|
||||||
curl -s http://192.168.1.196:8081/status | python3 -m json.tool
|
curl -s http://AI_HOST:8081/status | python3 -m json.tool
|
||||||
curl -s -X POST http://192.168.1.196:8081/medium
|
curl -s -X POST http://AI_HOST:8081/medium
|
||||||
# TTS-Test
|
# TTS-Test
|
||||||
curl -s http://192.168.1.196:8081/v1/audio/speech \
|
curl -s http://AI_HOST:8081/v1/audio/speech \
|
||||||
-H 'Content-Type: application/json' \
|
-H 'Content-Type: application/json' \
|
||||||
-d '{"input":"Hallo","voice":"claribel"}' -o test.mp3
|
-d '{"input":"Hallo","voice":"claribel"}' -o test.mp3
|
||||||
```
|
```
|
||||||
|
|||||||
+4
-3
@@ -4,15 +4,16 @@
|
|||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
cd "$(dirname "$0")/.."
|
cd "$(dirname "$0")/.."
|
||||||
|
|
||||||
TARGET="${TARGET:-root@192.168.1.196}"
|
: "${TARGET:?TARGET muss gesetzt sein, z. B. root@ai-host}"
|
||||||
SSH_KEY="${SSH_KEY:-$HOME/.ssh/lmstudio_unraid}"
|
SSH_KEY="${SSH_KEY:-$HOME/.ssh/id_ed25519}"
|
||||||
STAGE="/tmp/ai-profile-router-$$"
|
STAGE="/tmp/ai-profile-router-$$"
|
||||||
|
|
||||||
mkdir -p "$STAGE"
|
mkdir -p "$STAGE"
|
||||||
cp router/ai_profile_router.py router/image_worker.py router/xtts_worker.py \
|
cp router/ai_profile_router.py router/image_worker.py router/xtts_worker.py \
|
||||||
router/stt_worker.py \
|
router/stt_worker.py \
|
||||||
deploy/install.sh deploy/mike-ai-profile-router.service \
|
deploy/install.sh deploy/mike-ai-profile-router.service \
|
||||||
deploy/mike-ai-xtts.service deploy/mike-ai-whisper.service "$STAGE/"
|
deploy/mike-ai-xtts.service deploy/mike-ai-whisper.service \
|
||||||
|
deploy/requirements-image.lock deploy/requirements-xtts.lock "$STAGE/"
|
||||||
|
|
||||||
echo "== Übertrage Dateien nach ${TARGET}:/tmp/ai-profile-router/"
|
echo "== Übertrage Dateien nach ${TARGET}:/tmp/ai-profile-router/"
|
||||||
ssh -i "$SSH_KEY" "$TARGET" 'mkdir -p /tmp/ai-profile-router'
|
ssh -i "$SSH_KEY" "$TARGET" 'mkdir -p /tmp/ai-profile-router'
|
||||||
|
|||||||
+6
-10
@@ -51,13 +51,11 @@ if [ ! -x "$VENV/bin/python" ]; then
|
|||||||
echo "-- Erstelle Python-Venv in $VENV"
|
echo "-- Erstelle Python-Venv in $VENV"
|
||||||
python3 -m venv "$VENV"
|
python3 -m venv "$VENV"
|
||||||
fi
|
fi
|
||||||
echo "-- Installiere/aktualisiere Bild-Abhängigkeiten (torch, diffusers, ...)"
|
echo "-- Installiere festgeschriebene Bild-Abhängigkeiten"
|
||||||
"$VENV/bin/pip" install --quiet --upgrade pip
|
"$VENV/bin/pip" install --quiet --upgrade pip
|
||||||
"$VENV/bin/pip" install --quiet \
|
"$VENV/bin/pip" install --quiet torch==2.11.0+cu128 \
|
||||||
torch \
|
--index-url https://download.pytorch.org/whl/cu128
|
||||||
diffusers \
|
"$VENV/bin/pip" install --quiet -r "$DIR/requirements-image.lock"
|
||||||
transformers \
|
|
||||||
accelerate
|
|
||||||
|
|
||||||
# --- 4. FLUX-Modell (nur wenn noch nicht vorhanden) --------------------------
|
# --- 4. FLUX-Modell (nur wenn noch nicht vorhanden) --------------------------
|
||||||
if [ -f "$MODEL_DIR/model_index.json" ]; then
|
if [ -f "$MODEL_DIR/model_index.json" ]; then
|
||||||
@@ -89,12 +87,10 @@ if [ ! -x "$XTTS_VENV/bin/python" ]; then
|
|||||||
uv venv --python 3.11 "$XTTS_VENV"
|
uv venv --python 3.11 "$XTTS_VENV"
|
||||||
"$XTTS_VENV/bin/pip" install --quiet --upgrade pip
|
"$XTTS_VENV/bin/pip" install --quiet --upgrade pip
|
||||||
echo "-- Installiere CPU-only torch"
|
echo "-- Installiere CPU-only torch"
|
||||||
"$XTTS_VENV/bin/pip" install --quiet torch torchaudio \
|
"$XTTS_VENV/bin/pip" install --quiet torch==2.11.0+cpu torchaudio==2.11.0+cpu \
|
||||||
--index-url https://download.pytorch.org/whl/cpu
|
--index-url https://download.pytorch.org/whl/cpu
|
||||||
echo "-- Installiere Coqui TTS + Abhängigkeiten"
|
echo "-- Installiere Coqui TTS + Abhängigkeiten"
|
||||||
"$XTTS_VENV/bin/pip" install --quiet \
|
"$XTTS_VENV/bin/pip" install --quiet -r "$DIR/requirements-xtts.lock"
|
||||||
TTS==0.22.0 transformers==4.40.2 tokenizers==0.19.1 \
|
|
||||||
huggingface-hub==0.36.2 librosa soundfile
|
|
||||||
# PyTorch 2.6+ weights_only-Patch für Coqui TTS
|
# PyTorch 2.6+ weights_only-Patch für Coqui TTS
|
||||||
echo "-- Wende weights_only-Patch für Coqui TTS an"
|
echo "-- Wende weights_only-Patch für Coqui TTS an"
|
||||||
"$XTTS_VENV/bin/python" - <<'PY'
|
"$XTTS_VENV/bin/python" - <<'PY'
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ Environment=IMAGE_DIR=/opt/mike-ai/ai-profile-router/images
|
|||||||
Environment=IMAGE_WORKER_LOG=/opt/mike-ai/ai-profile-router/worker.log
|
Environment=IMAGE_WORKER_LOG=/opt/mike-ai/ai-profile-router/worker.log
|
||||||
Environment=IMAGE_GEN_TIMEOUT=600
|
Environment=IMAGE_GEN_TIMEOUT=600
|
||||||
Environment=IMAGE_VRAM_FREE_TIMEOUT=120
|
Environment=IMAGE_VRAM_FREE_TIMEOUT=120
|
||||||
Environment=LLAMA_SERVER_BIN=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server
|
Environment=LLAMA_SERVER_BIN=/opt/mike-ai/llama.cpp/build/bin/llama-server
|
||||||
Environment=VISION_MODEL=/opt/mike-ai/models/qwen3.8-27b/Qwen3.8-27B-Q3_K_M.gguf
|
Environment=VISION_MODEL=/opt/mike-ai/models/qwen3.8-27b/Qwen3.8-27B-Q3_K_M.gguf
|
||||||
Environment=VISION_MMPROJ=/opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf
|
Environment=VISION_MMPROJ=/opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf
|
||||||
Environment=VISION_CTX=32768
|
Environment=VISION_CTX=32768
|
||||||
|
|||||||
@@ -0,0 +1,6 @@
|
|||||||
|
diffusers @ git+https://github.com/huggingface/diffusers.git@11a82a15fe473ed974ff35111dd629b05fb1b3ed
|
||||||
|
transformers==5.15.0
|
||||||
|
accelerate==1.14.0
|
||||||
|
huggingface-hub==1.28.0
|
||||||
|
safetensors==0.8.0
|
||||||
|
Pillow==12.3.0
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
TTS==0.22.0
|
||||||
|
transformers==4.40.2
|
||||||
|
tokenizers==0.19.1
|
||||||
|
huggingface-hub==0.36.2
|
||||||
|
librosa==0.11.0
|
||||||
|
soundfile==0.14.0
|
||||||
@@ -23,12 +23,16 @@ is_running() {
|
|||||||
case "$CMD" in
|
case "$CMD" in
|
||||||
stop)
|
stop)
|
||||||
if is_running; then
|
if is_running; then
|
||||||
kill "$(cat "$PIDFILE")" 2>/dev/null || true
|
PID="$(cat "$PIDFILE")"
|
||||||
rm -f "$PIDFILE"
|
kill "$PID" 2>/dev/null || true
|
||||||
for _ in $(seq 1 50); do
|
for _ in $(seq 1 50); do
|
||||||
is_running || break
|
kill -0 "$PID" 2>/dev/null || break
|
||||||
sleep 0.1
|
sleep 0.1
|
||||||
done
|
done
|
||||||
|
if kill -0 "$PID" 2>/dev/null; then
|
||||||
|
kill -9 "$PID" 2>/dev/null || true
|
||||||
|
fi
|
||||||
|
rm -f "$PIDFILE"
|
||||||
fi
|
fi
|
||||||
exit 0
|
exit 0
|
||||||
;;
|
;;
|
||||||
|
|||||||
@@ -0,0 +1,81 @@
|
|||||||
|
# Architektur
|
||||||
|
|
||||||
|
## Ziel
|
||||||
|
|
||||||
|
Die Plattform stellt eine private, lokal betriebene OpenAI-kompatible API
|
||||||
|
bereit. Jede Komponente hat genau eine Aufgabe und kann unabhängig ersetzt
|
||||||
|
werden. Ein Neuaufbau darf keine Dateien vom alten Host voraussetzen, die nicht
|
||||||
|
in diesem Repository oder im Modellmanifest beschrieben sind.
|
||||||
|
|
||||||
|
## Komponenten
|
||||||
|
|
||||||
|
| Komponente | Port | Ausführung | Aufgabe |
|
||||||
|
|---|---:|---|---|
|
||||||
|
| AI Profile Router | 8081 | systemd, unprivilegiert empfohlen | zentrale Client-API und Orchestrierung |
|
||||||
|
| llama.cpp | 8080 | systemd | Textmodell, Tool Calling und MCP |
|
||||||
|
| Whisper | 8084, nur localhost | systemd | Speech-to-Text |
|
||||||
|
| XTTS | 8085, nur localhost | systemd | Text-to-Speech |
|
||||||
|
| TinySearch | 8000, nur localhost | Docker | kompakte Websuche |
|
||||||
|
| SearXNG | intern | Docker | Suchmaschinen-Metasuche |
|
||||||
|
| LLama-GUI | 5240, optional | systemd | manuelle Administration |
|
||||||
|
| Glances | lokal, optional | systemd | Systemmetriken |
|
||||||
|
|
||||||
|
## Request-Fluss
|
||||||
|
|
||||||
|
1. Ein Client verwendet ausschließlich Port 8081.
|
||||||
|
2. Der Router veröffentlicht `qwen-fast`, `qwen-medium` und `qwen-long`.
|
||||||
|
3. Passt das aktive Profil nicht zum virtuellen Modell, wird llama.cpp kontrolliert
|
||||||
|
mit dem passenden Profil neu gestartet.
|
||||||
|
4. Der Router wartet auf Modellname und erwartete Kontextgröße.
|
||||||
|
5. Erst dann wird die Anfrage an Port 8080 weitergeleitet.
|
||||||
|
|
||||||
|
## Profilprinzip
|
||||||
|
|
||||||
|
Die Profile sind vollständige systemd-Overrides. Ein Profilwechsel kopiert die
|
||||||
|
gewählte Datei atomar auf `override.conf`, lädt systemd neu und startet genau
|
||||||
|
einen llama.cpp-Dienst neu. Es gibt niemals mehrere Textmodelle gleichzeitig.
|
||||||
|
|
||||||
|
## GPU-Hotswap
|
||||||
|
|
||||||
|
Vision und Bildgenerierung teilen sich die RTX mit dem Textmodell. Der Router:
|
||||||
|
|
||||||
|
1. sperrt die GPU-Orchestrierung,
|
||||||
|
2. merkt sich das aktive Textprofil,
|
||||||
|
3. stoppt llama.cpp,
|
||||||
|
4. startet vorübergehend Vision oder FLUX,
|
||||||
|
5. beendet den Hilfsprozess vollständig,
|
||||||
|
6. stellt das ursprüngliche Textprofil wieder her,
|
||||||
|
7. prüft Modell und Kontext vor der Freigabe.
|
||||||
|
|
||||||
|
## Verzeichnislayout auf dem Zielhost
|
||||||
|
|
||||||
|
```text
|
||||||
|
/opt/mike-ai/
|
||||||
|
ai-profile-router/ Routercode und eigenes Venv
|
||||||
|
llama.cpp/ exakt ein produktiver Build
|
||||||
|
models/ Modelle nach Manifest
|
||||||
|
whisper.cpp/ Speech-to-Text Runtime
|
||||||
|
xtts/ XTTS Runtime und Cache
|
||||||
|
web-search/ Docker Compose für TinySearch/SearXNG
|
||||||
|
|
||||||
|
/etc/mike-ai/
|
||||||
|
mcp-servers.json lokale, geheime produktive Konfiguration
|
||||||
|
*.env Credentials, niemals im Git
|
||||||
|
|
||||||
|
/etc/systemd/system/
|
||||||
|
mike-ai-*.service
|
||||||
|
mike-ai-llama-ui.service.d/
|
||||||
|
override.conf
|
||||||
|
profile-fast.conf.disabled
|
||||||
|
profile-medium.conf.disabled
|
||||||
|
profile-long.conf.disabled
|
||||||
|
```
|
||||||
|
|
||||||
|
## Nicht Teil der Zielarchitektur
|
||||||
|
|
||||||
|
- parallele llama.cpp-Builds
|
||||||
|
- allgemeiner Shell-MCP im Standardprofil
|
||||||
|
- doppelte Unraid-MCPs
|
||||||
|
- RX-spezifische Dienste ohne eingebaute RX
|
||||||
|
- automatisch startende Benchmark-Dienste
|
||||||
|
- Modellkopien außerhalb des dokumentierten Modellverzeichnisses
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
# Komponentenverzeichnis
|
||||||
|
|
||||||
|
| Bestandteil | Quelle | Bestandteil dieses Repositories | Status |
|
||||||
|
|---|---|---|---|
|
||||||
|
| AI Profile Router | `router/` | vollständig | Kern |
|
||||||
|
| llama.cpp | ggml-org/llama.cpp, festgeschriebener Commit | Buildskript und Commit | Kern |
|
||||||
|
| Qwen-Profile | `platform/profiles/` | vollständig, Modelle ausgenommen | Kern |
|
||||||
|
| Websuche | TinySearch + SearXNG | Compose und sichere Grundkonfiguration | Kern |
|
||||||
|
| Web-MCP-Fassade | `platform/web-search/web_search_mcp.py` | vollständig | Kern |
|
||||||
|
| Home-Assistant-MCP | separates privates Repository | nur Integration dokumentiert | optional |
|
||||||
|
| ARR-MCP | separates privates Repository | nur Integration dokumentiert | optional |
|
||||||
|
| Unraid-MCP | separates Repository/Installation | read-only Integration dokumentiert | optional |
|
||||||
|
| Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional |
|
||||||
|
| XTTS-v2 | Coqui | Worker, Service und Lockdatei | optional |
|
||||||
|
| FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional |
|
||||||
|
| LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional |
|
||||||
|
| Glances | Distribution | nur Betriebsrolle dokumentiert | optional |
|
||||||
|
|
||||||
|
Separate MCP-Repositories werden nicht in dieses Repository kopiert. Ihre
|
||||||
|
Versionen sollen künftig in einem Release-Manifest referenziert werden. So
|
||||||
|
bleiben Zuständigkeiten klar und Updates können unabhängig getestet werden.
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
# Saubere Installation
|
||||||
|
|
||||||
|
Diese Anleitung beschreibt den Neuaufbau. Sie löscht oder migriert keine Daten
|
||||||
|
automatisch.
|
||||||
|
|
||||||
|
## 1. Voraussetzungen
|
||||||
|
|
||||||
|
- Debian 13 oder kompatibles Linux
|
||||||
|
- NVIDIA-Treiber und funktionierendes `nvidia-smi`
|
||||||
|
- Build-Werkzeuge, CMake, Git und CUDA Toolkit
|
||||||
|
- Python 3.13 für Router/FLUX und Python 3.11 für XTTS
|
||||||
|
- Docker plus Compose für Websuche
|
||||||
|
- ausreichend freier Speicher; mindestens 15 Prozent auf `/`
|
||||||
|
|
||||||
|
## 2. Benutzer und Verzeichnisse
|
||||||
|
|
||||||
|
Für produktive Dienste sollen eigene Systembenutzer verwendet werden. Der
|
||||||
|
Router benötigt kontrollierte Berechtigung zum Neustart des llama.cpp-Dienstes;
|
||||||
|
er sollte nicht dauerhaft als root laufen.
|
||||||
|
|
||||||
|
```text
|
||||||
|
/opt/mike-ai/models
|
||||||
|
/opt/mike-ai/ai-profile-router
|
||||||
|
/etc/mike-ai
|
||||||
|
```
|
||||||
|
|
||||||
|
Geheimnisse werden mit Modus `0600` unter `/etc/mike-ai` abgelegt.
|
||||||
|
|
||||||
|
## 3. llama.cpp bauen
|
||||||
|
|
||||||
|
`platform/llama/build-llama-cpp.sh` checkt exakt den in
|
||||||
|
`platform/llama/LLAMA_CPP_COMMIT` hinterlegten Commit aus. Vor einem Upgrade:
|
||||||
|
|
||||||
|
1. neuen Commit in einem separaten Build testen,
|
||||||
|
2. Standardbenchmark ausführen,
|
||||||
|
3. MCP-Grammatik und Tool Calls prüfen,
|
||||||
|
4. Commitdatei erst danach aktualisieren.
|
||||||
|
|
||||||
|
## 4. Modelle bereitstellen
|
||||||
|
|
||||||
|
Modelldateien werden nicht in Git gespeichert. Die erwarteten Rollen und
|
||||||
|
Zielpfade stehen in `platform/models/manifest.example.yaml`. Für die lokale
|
||||||
|
Installation wird daraus eine nicht eingecheckte `manifest.local.yaml` mit
|
||||||
|
SHA256-Prüfsummen erstellt.
|
||||||
|
|
||||||
|
## 5. llama.cpp-Dienst und Profile
|
||||||
|
|
||||||
|
Die Dateien aus `platform/systemd` und `platform/profiles` installieren. Danach:
|
||||||
|
|
||||||
|
```text
|
||||||
|
llama-profile fast
|
||||||
|
```
|
||||||
|
|
||||||
|
Der Befehl muss Port 8080 erst freigeben, wenn Modell und Kontext korrekt sind.
|
||||||
|
|
||||||
|
## 6. MCP-Konfiguration
|
||||||
|
|
||||||
|
`platform/mcp/mcp-servers.example.json` nach `/etc/mike-ai/mcp-servers.json`
|
||||||
|
kopieren und nur benötigte Server aktivieren. Zugangsdaten werden ausschließlich
|
||||||
|
über lokale Environment-Dateien oder einen Secret Broker referenziert.
|
||||||
|
|
||||||
|
## 7. Router installieren
|
||||||
|
|
||||||
|
Das bestehende `deploy/install.sh` installiert Router, Vision/Bild-Worker,
|
||||||
|
Whisper und XTTS. Vor produktiver Verwendung müssen Modellpfade in der
|
||||||
|
systemd-Datei gegen das lokale Manifest geprüft werden.
|
||||||
|
|
||||||
|
## 8. Websuche
|
||||||
|
|
||||||
|
TinySearch und SearXNG bleiben als einziges Docker-Teilsystem isoliert. Die
|
||||||
|
Suchdienste sollen nur an localhost gebunden werden; nur der Web-MCP greift
|
||||||
|
darauf zu.
|
||||||
|
|
||||||
|
## 9. Verifikation
|
||||||
|
|
||||||
|
`platform/checks/verify-platform.sh` kontrolliert:
|
||||||
|
|
||||||
|
- freien Plattenplatz,
|
||||||
|
- GPU und VRAM,
|
||||||
|
- aktive Dienste,
|
||||||
|
- Ports,
|
||||||
|
- Routermodelle und aktives Profil,
|
||||||
|
- llama.cpp-Health,
|
||||||
|
- unerwartete RX- und Benchmark-Dienste.
|
||||||
|
|
||||||
|
Erst nach erfolgreicher Prüfung werden Clients auf Port 8081 umgestellt.
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
# Migration vom bestehenden Host
|
||||||
|
|
||||||
|
## Behalten
|
||||||
|
|
||||||
|
- Qwen3.8-27B IQ4-MIX und das getestete MTP2-Profil
|
||||||
|
- IQ4_XS Pure für Medium
|
||||||
|
- Q3-Vision-Modell und passender `mmproj`
|
||||||
|
- festgeschriebener llama.cpp-Commit
|
||||||
|
- Router, Whisper, XTTS und Websuche
|
||||||
|
- spezialisierte MCPs nach Sicherheitsprofil
|
||||||
|
- relevante Benchmarkresultate
|
||||||
|
|
||||||
|
## Nicht übernehmen
|
||||||
|
|
||||||
|
- RX-470-Dienste
|
||||||
|
- doppelte Whisper-Server
|
||||||
|
- automatisch aktivierte Modellrennen und Benchmarks
|
||||||
|
- unvollständige Modelldownloads
|
||||||
|
- alte llama.cpp-/BeeLlama-Testbuilds
|
||||||
|
- alte systemd-Backups
|
||||||
|
- Caches und generierte Medien
|
||||||
|
- doppelte oder klar unterlegene Modelle
|
||||||
|
|
||||||
|
## Reihenfolge
|
||||||
|
|
||||||
|
1. Repositories und verschlüsselte Konfiguration sichern.
|
||||||
|
2. Modellmanifest mit Dateigrößen und SHA256 erstellen.
|
||||||
|
3. Neuen Host installieren und Speicherlayout festlegen.
|
||||||
|
4. NVIDIA-Treiber und CUDA verifizieren.
|
||||||
|
5. Festgeschriebenen llama.cpp-Commit bauen.
|
||||||
|
6. nur die benötigten Modelle übertragen und Hashes prüfen.
|
||||||
|
7. Fast-Profil ohne MCP starten und testen.
|
||||||
|
8. Medium und Long einzeln testen.
|
||||||
|
9. Router installieren und Profilwechsel testen.
|
||||||
|
10. Web, HA, ARR und Unraid nacheinander hinzufügen.
|
||||||
|
11. STT, TTS und Vision ergänzen.
|
||||||
|
12. Standardbenchmark und Sicherheitsprüfung ausführen.
|
||||||
|
13. Erst danach Clients umstellen.
|
||||||
|
|
||||||
|
Der alte Host bleibt bis zum bestandenen Abnahmetest unverändert und dient nur
|
||||||
|
als Referenz. Es werden keine Caches oder unbekannten Altverzeichnisse kopiert.
|
||||||
@@ -0,0 +1,57 @@
|
|||||||
|
# Betrieb
|
||||||
|
|
||||||
|
## Profile
|
||||||
|
|
||||||
|
| Profil | Virtuelles Modell | Kontext | Zweck |
|
||||||
|
|---|---|---:|---|
|
||||||
|
| Fast | `qwen-fast` | 73.728 | Alltag, Agenten, hohe Geschwindigkeit |
|
||||||
|
| Medium | `qwen-medium` | 94.208 | mehr Kontext, reine IQ4_XS-Variante |
|
||||||
|
| Long | `qwen-long` | 131.072 | lange Hermes-/MCP-Sitzungen |
|
||||||
|
|
||||||
|
Manuell wird mit `llama-profile fast|medium|long` gewechselt. Über HTTP stehen
|
||||||
|
`POST /fast`, `/medium` und `/long` zur Verfügung. Für eine spätere Version ist
|
||||||
|
`large` als Alias für `long` vorgesehen; bestehende Namen bleiben kompatibel.
|
||||||
|
|
||||||
|
## Clients
|
||||||
|
|
||||||
|
Clients verbinden sich mit:
|
||||||
|
|
||||||
|
```text
|
||||||
|
http://HOST:8081/v1
|
||||||
|
```
|
||||||
|
|
||||||
|
Sie sollen nicht direkt Port 8080 verwenden, weil sie sonst Profilumschaltung,
|
||||||
|
Vision, Bildgenerierung, STT und TTS umgehen.
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
- `GET /status`: Router, Profil, Upstream, aktive Jobs
|
||||||
|
- `GET /v1/models`: virtuelle Modelle
|
||||||
|
- llama.cpp-Metriken: Port 8080, nur im administrativen Netz freigeben
|
||||||
|
- systemd-Journal: nur Metadaten und Fehler prüfen; keine Promptinhalte sammeln
|
||||||
|
|
||||||
|
## Upgrade-Regel
|
||||||
|
|
||||||
|
Niemals Build, Quantisierung und Profil gleichzeitig ändern. Immer genau eine
|
||||||
|
Variable ändern und anschließend denselben Benchmark ausführen.
|
||||||
|
|
||||||
|
## Kapazitätsregeln
|
||||||
|
|
||||||
|
- Systempartition dauerhaft unter 85 Prozent halten.
|
||||||
|
- Mindestens 1 GiB Sicherheitsreserve für allgemeine GPU-Profile vorsehen;
|
||||||
|
experimentelle Max-GPU-Profile klar kennzeichnen.
|
||||||
|
- Nur ein Textmodell gleichzeitig laden.
|
||||||
|
- Benchmarks sind deaktivierte, manuell gestartete Jobs und keine Boot-Dienste.
|
||||||
|
|
||||||
|
## Backup
|
||||||
|
|
||||||
|
Gesichert werden:
|
||||||
|
|
||||||
|
- dieses Repository,
|
||||||
|
- lokale Modellmanifest-Datei mit Hashes, aber ohne Secrets,
|
||||||
|
- `/etc/mike-ai` verschlüsselt,
|
||||||
|
- systemd-Konfiguration,
|
||||||
|
- Benchmarkresultate.
|
||||||
|
|
||||||
|
Nicht gesichert werden müssen Build-Verzeichnisse, Venvs, Caches oder Modelle,
|
||||||
|
wenn Downloadquelle und Prüfsumme dokumentiert sind.
|
||||||
@@ -0,0 +1,59 @@
|
|||||||
|
# Sicherheitsmodell
|
||||||
|
|
||||||
|
## Grundsatz
|
||||||
|
|
||||||
|
Das lokale Modell erhält nur die Werkzeuge, die es für den aktuellen Modus
|
||||||
|
benötigt. Lokalität allein ersetzt keine Zugriffskontrolle.
|
||||||
|
|
||||||
|
## MCP-Profile
|
||||||
|
|
||||||
|
Empfohlene Trennung:
|
||||||
|
|
||||||
|
| Modus | Werkzeuge |
|
||||||
|
|---|---|
|
||||||
|
| Standard | Websuche, harmlose lokale Hilfsfunktionen |
|
||||||
|
| Home Assistant | HA-Administration plus Websuche |
|
||||||
|
| ARR | Sonarr/Radarr plus Websuche |
|
||||||
|
| Unraid Read-only | Diagnose, Logs, Status |
|
||||||
|
| Unraid Write | nur bewusst aktiviert, mit Vorschau und Approval Ticket |
|
||||||
|
|
||||||
|
## Nicht im Standardprofil
|
||||||
|
|
||||||
|
- allgemeine Shell
|
||||||
|
- `python3`, `ssh`, `scp` oder beliebiges `curl`
|
||||||
|
- Container erstellen, verändern oder löschen
|
||||||
|
- Registry-/Storage-Direktzugriff
|
||||||
|
- uneingeschränkte Dateisuche
|
||||||
|
|
||||||
|
## Secrets
|
||||||
|
|
||||||
|
- Keine Secrets in Git, Prompts, MCP-Schemas oder Logs.
|
||||||
|
- Konfiguration referenziert nur Namen lokaler Environment-Dateien.
|
||||||
|
- Dateien mit Secrets: Eigentümer root oder Dienstbenutzer, Modus `0600`.
|
||||||
|
- Tokens werden pro Dienst getrennt und minimal berechtigt.
|
||||||
|
- Ein Secret Broker oder Wrapper stellt Verbindungen her, ohne Tokens an das
|
||||||
|
Modell zurückzugeben.
|
||||||
|
|
||||||
|
## Netzwerk
|
||||||
|
|
||||||
|
- Port 8080 nur localhost oder administratives VLAN.
|
||||||
|
- Clients verwenden Port 8081.
|
||||||
|
- Whisper, XTTS, TinySearch und SearXNG nur localhost.
|
||||||
|
- Firewall erlaubt nur bekannte Quellnetze.
|
||||||
|
- Externe Suche erhält nur die tatsächliche Suchanfrage, keine Chat-Historie.
|
||||||
|
|
||||||
|
## Schreibaktionen
|
||||||
|
|
||||||
|
Jede destruktive oder persistente Aktion verwendet:
|
||||||
|
|
||||||
|
1. read-only Bestandsaufnahme,
|
||||||
|
2. exakte Vorschau,
|
||||||
|
3. an diese Vorschau gebundenes Approval Ticket,
|
||||||
|
4. unveränderte Ausführung,
|
||||||
|
5. anschließende Verifikation.
|
||||||
|
|
||||||
|
## Repository-Prüfung vor jedem Push
|
||||||
|
|
||||||
|
- Suche nach Token-, Passwort- und Private-Key-Mustern.
|
||||||
|
- Keine `.env`, Zertifikate, Logs, Bilder, Audio oder Modellartefakte.
|
||||||
|
- Keine echten internen API-Schlüssel in Beispielen.
|
||||||
Executable
+70
@@ -0,0 +1,70 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -uo pipefail
|
||||||
|
|
||||||
|
PASS=0
|
||||||
|
WARN=0
|
||||||
|
FAIL=0
|
||||||
|
|
||||||
|
pass() { printf 'PASS %s\n' "$*"; PASS=$((PASS + 1)); }
|
||||||
|
warn() { printf 'WARN %s\n' "$*"; WARN=$((WARN + 1)); }
|
||||||
|
fail() { printf 'FAIL %s\n' "$*"; FAIL=$((FAIL + 1)); }
|
||||||
|
|
||||||
|
echo "== Local AI platform verification =="
|
||||||
|
|
||||||
|
ROOT_USE="$(df -P / | awk 'NR==2 {gsub(/%/, "", $5); print $5}')"
|
||||||
|
if [[ -n "$ROOT_USE" && "$ROOT_USE" -lt 85 ]]; then
|
||||||
|
pass "Systempartition bei ${ROOT_USE}%"
|
||||||
|
else
|
||||||
|
fail "Systempartition bei ${ROOT_USE:-unbekannt}% (Ziel: unter 85%)"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if command -v nvidia-smi >/dev/null 2>&1; then
|
||||||
|
GPU="$(nvidia-smi --query-gpu=name,memory.total --format=csv,noheader 2>/dev/null | head -1)"
|
||||||
|
[[ -n "$GPU" ]] && pass "GPU erkannt: $GPU" || fail "nvidia-smi liefert keine GPU"
|
||||||
|
else
|
||||||
|
fail "nvidia-smi fehlt"
|
||||||
|
fi
|
||||||
|
|
||||||
|
for service in mike-ai-llama-ui mike-ai-profile-router; do
|
||||||
|
if systemctl is-active --quiet "$service"; then
|
||||||
|
pass "$service aktiv"
|
||||||
|
else
|
||||||
|
fail "$service nicht aktiv"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
for optional in mike-ai-whisper mike-ai-xtts mike-ai-web-search; do
|
||||||
|
if systemctl is-active --quiet "$optional"; then
|
||||||
|
pass "$optional aktiv"
|
||||||
|
else
|
||||||
|
warn "$optional nicht aktiv oder nicht installiert"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if curl -fsS --max-time 3 http://127.0.0.1:8080/health >/dev/null; then
|
||||||
|
pass "llama.cpp Health-Check"
|
||||||
|
else
|
||||||
|
fail "llama.cpp auf Port 8080 nicht gesund"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if STATUS="$(curl -fsS --max-time 3 http://127.0.0.1:8081/status 2>/dev/null)"; then
|
||||||
|
PROFILE="$(python3 -c 'import json,sys; print(json.load(sys.stdin).get("current_profile"))' <<<"$STATUS" 2>/dev/null)"
|
||||||
|
pass "Router erreichbar, Profil ${PROFILE:-unbekannt}"
|
||||||
|
else
|
||||||
|
fail "Router auf Port 8081 nicht erreichbar"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if systemctl list-unit-files --no-legend 2>/dev/null | grep -Eq '(vision-rx|whisper-rx|granite-rx).*enabled'; then
|
||||||
|
warn "Aktivierte RX-Altlast gefunden"
|
||||||
|
else
|
||||||
|
pass "Keine aktivierte RX-Altlast"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if systemctl list-unit-files --no-legend 2>/dev/null | grep -E '(benchmark|race).*enabled' >/dev/null; then
|
||||||
|
warn "Automatisch aktivierter Benchmark-/Race-Dienst gefunden"
|
||||||
|
else
|
||||||
|
pass "Keine automatisch aktivierten Benchmarks"
|
||||||
|
fi
|
||||||
|
|
||||||
|
printf '\nErgebnis: %d PASS, %d WARN, %d FAIL\n' "$PASS" "$WARN" "$FAIL"
|
||||||
|
[[ "$FAIL" -eq 0 ]]
|
||||||
Executable
+49
@@ -0,0 +1,49 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
if [[ $EUID -ne 0 ]]; then
|
||||||
|
echo "Dieses Skript muss als root ausgeführt werden." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||||
|
PLATFORM="$REPO_ROOT/platform"
|
||||||
|
PROFILE_TARGET=/opt/mike-ai/platform/profiles
|
||||||
|
DROPIN=/etc/systemd/system/mike-ai-llama-ui.service.d
|
||||||
|
WEB_TARGET=/opt/mike-ai/web-search
|
||||||
|
|
||||||
|
install -d -m 0755 "$PROFILE_TARGET" "$DROPIN" "$WEB_TARGET" /etc/mike-ai
|
||||||
|
install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \
|
||||||
|
/etc/systemd/system/mike-ai-llama-ui.service
|
||||||
|
install -m 0644 "$PLATFORM/systemd/mike-ai-web-search.service" \
|
||||||
|
/etc/systemd/system/mike-ai-web-search.service
|
||||||
|
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
|
||||||
|
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
|
||||||
|
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
|
||||||
|
install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf"
|
||||||
|
install -m 0644 "$PLATFORM/web-search/compose.yaml" "$WEB_TARGET/compose.yaml"
|
||||||
|
install -m 0644 "$PLATFORM/web-search/tinysearch_config.json" \
|
||||||
|
"$WEB_TARGET/tinysearch_config.json"
|
||||||
|
install -m 0755 "$PLATFORM/web-search/web_search_mcp.py" \
|
||||||
|
"$WEB_TARGET/web_search_mcp.py"
|
||||||
|
if [[ ! -e "$WEB_TARGET/searxng-settings.yml" ]]; then
|
||||||
|
install -m 0600 "$PLATFORM/web-search/searxng-settings.example.yml" \
|
||||||
|
"$WEB_TARGET/searxng-settings.yml.example"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! -e /etc/mike-ai/mcp-servers.json ]]; then
|
||||||
|
install -m 0600 "$PLATFORM/mcp/mcp-servers.example.json" \
|
||||||
|
/etc/mike-ai/mcp-servers.json.example
|
||||||
|
fi
|
||||||
|
|
||||||
|
systemctl daemon-reload
|
||||||
|
|
||||||
|
cat <<'EOF'
|
||||||
|
Kernkonfiguration installiert, aber noch nicht gestartet.
|
||||||
|
|
||||||
|
Vor dem Start:
|
||||||
|
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
|
||||||
|
2. /etc/mike-ai/mcp-servers.json mit minimalen Servern erstellen.
|
||||||
|
3. llama.cpp bauen.
|
||||||
|
4. Danach: llama-profile fast
|
||||||
|
EOF
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
4df29be4f4c3673f428170fda944a5b19f743bb8
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
# llama.cpp und Profile
|
||||||
|
|
||||||
|
Der produktive Build wird über `LLAMA_CPP_COMMIT` festgeschrieben. Damit ist
|
||||||
|
ein Neuaufbau unabhängig vom jeweils aktuellen Stand des Upstream-Master.
|
||||||
|
|
||||||
|
## Build
|
||||||
|
|
||||||
|
```text
|
||||||
|
sudo platform/llama/build-llama-cpp.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
Der Build aktiviert CUDA und den HTTP-Server. Änderungen am Commit werden erst
|
||||||
|
nach Standardbenchmark, Tool-Calling-Test und Kontexttest übernommen.
|
||||||
|
|
||||||
|
## Produktive Modelle
|
||||||
|
|
||||||
|
- Fast und Long: Qwen3.8-27B IQ4-MIX mit MTP2
|
||||||
|
- Medium: Qwen3.8-27B IQ4_XS Pure ohne MTP
|
||||||
|
- Vision-Hotswap: Qwen3.8-27B Q3_K_M plus BF16-mmproj
|
||||||
|
|
||||||
|
Die Dateien selbst sind nicht Bestandteil des Repositories. Pfade und Hashes
|
||||||
|
werden im lokalen Modellmanifest verwaltet.
|
||||||
|
|
||||||
|
## Profilinstallation
|
||||||
|
|
||||||
|
Die Vorlagen unter `platform/profiles` verwenden Umgebungsvariablen in einer
|
||||||
|
gemeinsamen Environment-Datei. Für die aktuelle produktive Installation können
|
||||||
|
sie alternativ als dokumentierte Referenz für vollständige systemd-Overrides
|
||||||
|
verwendet werden.
|
||||||
|
|
||||||
|
Der Profilwechsel erfolgt ausschließlich über `platform/scripts/llama-profile`.
|
||||||
Executable
+29
@@ -0,0 +1,29 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
ROOT_DIR="${ROOT_DIR:-/opt/mike-ai}"
|
||||||
|
SOURCE_DIR="${SOURCE_DIR:-$ROOT_DIR/llama.cpp}"
|
||||||
|
BUILD_DIR="${BUILD_DIR:-$SOURCE_DIR/build}"
|
||||||
|
REPO_URL="${REPO_URL:-https://github.com/ggml-org/llama.cpp.git}"
|
||||||
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
COMMIT="$(tr -d '[:space:]' < "$SCRIPT_DIR/LLAMA_CPP_COMMIT")"
|
||||||
|
|
||||||
|
if [[ $EUID -ne 0 ]]; then
|
||||||
|
echo "Dieses Skript muss als root ausgeführt werden." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ ! -d "$SOURCE_DIR/.git" ]]; then
|
||||||
|
git clone "$REPO_URL" "$SOURCE_DIR"
|
||||||
|
fi
|
||||||
|
|
||||||
|
git -C "$SOURCE_DIR" fetch --tags origin
|
||||||
|
git -C "$SOURCE_DIR" checkout --detach "$COMMIT"
|
||||||
|
|
||||||
|
cmake -S "$SOURCE_DIR" -B "$BUILD_DIR" \
|
||||||
|
-DGGML_CUDA=ON \
|
||||||
|
-DLLAMA_CURL=ON \
|
||||||
|
-DCMAKE_BUILD_TYPE=Release
|
||||||
|
cmake --build "$BUILD_DIR" --config Release --parallel "$(nproc)"
|
||||||
|
|
||||||
|
"$BUILD_DIR/bin/llama-server" --version
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# MCP-Architektur
|
||||||
|
|
||||||
|
Die produktive MCP-Konfiguration ist absichtlich nicht Bestandteil des Git-
|
||||||
|
Repositories, weil sie lokale Pfade und Zugangsdaten referenziert. Das Beispiel
|
||||||
|
zeigt nur die Struktur.
|
||||||
|
|
||||||
|
## Empfohlene Server
|
||||||
|
|
||||||
|
- `web`: Websuche über lokales TinySearch/SearXNG
|
||||||
|
- `homeassistant`: Administration mit eigenem, minimal berechtigtem Token
|
||||||
|
- `arr`: Sonarr/Radarr über spezialisierte Aktionen
|
||||||
|
- `unraid-readonly`: Diagnose ohne Schreiboperationen
|
||||||
|
|
||||||
|
## Getrennte Konfigurationen
|
||||||
|
|
||||||
|
Statt alle Werkzeuge ständig zu laden, werden mehrere Dateien empfohlen:
|
||||||
|
|
||||||
|
```text
|
||||||
|
/etc/mike-ai/mcp-standard.json
|
||||||
|
/etc/mike-ai/mcp-homeassistant.json
|
||||||
|
/etc/mike-ai/mcp-arr.json
|
||||||
|
/etc/mike-ai/mcp-unraid-readonly.json
|
||||||
|
/etc/mike-ai/mcp-unraid-write.json
|
||||||
|
```
|
||||||
|
|
||||||
|
Das jeweilige Profil verweist nur auf die benötigte Datei. Dadurch werden die
|
||||||
|
Tool-Schemas kleiner, das Kontextfenster bleibt frei und kleine Modelle müssen
|
||||||
|
weniger Werkzeuge unterscheiden.
|
||||||
|
|
||||||
|
Credentials werden von schmalen Wrapper-Programmen wie `run-arr-mcp` oder
|
||||||
|
`runraid` aus geschützten Environment-Dateien geladen. Das JSON selbst enthält
|
||||||
|
weder Werte noch Pfade zu einzelnen Tokens.
|
||||||
|
|
||||||
|
## Schreibzugriff
|
||||||
|
|
||||||
|
Schreibende Server gehören nicht in `mcp-standard.json`. Sie benötigen eine
|
||||||
|
Vorschau und ein an die exakte Änderung gebundenes Approval Ticket.
|
||||||
|
|
||||||
|
## Shell
|
||||||
|
|
||||||
|
Ein allgemeiner Shell-MCP ist nicht Teil der Zielplattform. Insbesondere
|
||||||
|
`python3`, `ssh`, `scp`, `curl` und `systemctl` dürfen nicht gemeinsam als
|
||||||
|
scheinbar harmlose Allowlist angeboten werden.
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
{
|
||||||
|
"mcpServers": {
|
||||||
|
"web": {
|
||||||
|
"command": "/usr/bin/python3",
|
||||||
|
"args": ["/opt/mike-ai/web-search/web_search_mcp.py"]
|
||||||
|
},
|
||||||
|
"homeassistant": {
|
||||||
|
"command": "/usr/local/bin/homeassistant-native-mcp",
|
||||||
|
"args": [],
|
||||||
|
"timeout_ms": 60000
|
||||||
|
},
|
||||||
|
"arr": {
|
||||||
|
"command": "/usr/local/bin/run-arr-mcp",
|
||||||
|
"args": [],
|
||||||
|
"timeout_ms": 30000
|
||||||
|
},
|
||||||
|
"unraid-readonly": {
|
||||||
|
"command": "/usr/local/bin/runraid",
|
||||||
|
"args": ["mcp"],
|
||||||
|
"timeout_ms": 60000
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,42 @@
|
|||||||
|
schema: 1
|
||||||
|
models:
|
||||||
|
qwen_fast_long:
|
||||||
|
role: primary-text-fast-and-long
|
||||||
|
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||||
|
file: Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
|
target: /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
|
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||||
|
qwen_medium:
|
||||||
|
role: primary-text-medium
|
||||||
|
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||||
|
file: qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
|
target: /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
|
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||||
|
qwen_vision:
|
||||||
|
role: temporary-vision-model
|
||||||
|
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||||
|
file: Qwen3.8-27B-Q3_K_M.gguf
|
||||||
|
target: /opt/mike-ai/models/qwen3.8-27b/Qwen3.8-27B-Q3_K_M.gguf
|
||||||
|
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||||
|
qwen_vision_projector:
|
||||||
|
role: vision-projector
|
||||||
|
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||||
|
file: mmproj-BF16.gguf
|
||||||
|
target: /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf
|
||||||
|
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||||
|
whisper:
|
||||||
|
role: speech-to-text
|
||||||
|
source: ggml-org/whisper.cpp
|
||||||
|
file: ggml-large-v3-turbo.bin
|
||||||
|
target: /opt/mike-ai/models/whisper/ggml-large-v3-turbo.bin
|
||||||
|
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||||
|
flux:
|
||||||
|
role: image-generation
|
||||||
|
source: black-forest-labs/FLUX.2-klein-base-4B
|
||||||
|
target: /opt/mike-ai/models/FLUX.2-klein-base-4B
|
||||||
|
revision: "PIN_EXACT_REVISION"
|
||||||
|
xtts:
|
||||||
|
role: text-to-speech
|
||||||
|
source: coqui/XTTS-v2
|
||||||
|
target: /opt/mike-ai/xtts/.cache
|
||||||
|
revision: "PIN_EXACT_REVISION"
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Local AI llama.cpp - Qwen Fast 72K MTP2
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
ExecStart=
|
||||||
|
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --alias qwen38-27b-iq4mix-72k-mtp2 --ctx-size 73728 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
ExecStart=
|
||||||
|
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
ExecStart=
|
||||||
|
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none
|
||||||
Executable
+35
@@ -0,0 +1,35 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
PROFILE_DIR="${PROFILE_DIR:-/etc/systemd/system/mike-ai-llama-ui.service.d}"
|
||||||
|
SOURCE_DIR="${SOURCE_DIR:-/opt/mike-ai/platform/profiles}"
|
||||||
|
SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
|
||||||
|
PROFILE="${1:-}"
|
||||||
|
|
||||||
|
case "$PROFILE" in
|
||||||
|
fast|medium|long) ;;
|
||||||
|
large) PROFILE=long ;;
|
||||||
|
*)
|
||||||
|
echo "Usage: llama-profile {fast|medium|long|large}" >&2
|
||||||
|
exit 2
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
SOURCE="$SOURCE_DIR/profile-$PROFILE.conf"
|
||||||
|
TARGET="$PROFILE_DIR/override.conf"
|
||||||
|
|
||||||
|
if [[ ! -r "$SOURCE" ]]; then
|
||||||
|
echo "Profildatei fehlt: $SOURCE" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
install -d -m 0755 "$PROFILE_DIR"
|
||||||
|
TEMP="$(mktemp "$PROFILE_DIR/.override.conf.XXXXXX")"
|
||||||
|
trap 'rm -f "$TEMP"' EXIT
|
||||||
|
install -m 0644 "$SOURCE" "$TEMP"
|
||||||
|
mv -f "$TEMP" "$TARGET"
|
||||||
|
trap - EXIT
|
||||||
|
|
||||||
|
systemctl daemon-reload
|
||||||
|
systemctl restart "$SERVICE"
|
||||||
|
echo "Profil '$PROFILE' wurde aktiviert."
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Local AI llama.cpp server
|
||||||
|
After=network-online.target
|
||||||
|
Wants=network-online.target
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=simple
|
||||||
|
# ExecStart wird vollständig durch das aktive Profil-Override definiert.
|
||||||
|
ExecStart=/usr/bin/false
|
||||||
|
Restart=on-failure
|
||||||
|
RestartSec=3
|
||||||
|
TimeoutStartSec=600
|
||||||
|
TimeoutStopSec=120
|
||||||
|
NoNewPrivileges=true
|
||||||
|
PrivateTmp=true
|
||||||
|
LimitNOFILE=1048576
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=multi-user.target
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Local AI Web Search (TinySearch + SearXNG)
|
||||||
|
Requires=docker.service
|
||||||
|
After=docker.service network-online.target
|
||||||
|
Before=mike-ai-llama-ui.service
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
RemainAfterExit=yes
|
||||||
|
WorkingDirectory=/opt/mike-ai/web-search
|
||||||
|
ExecStart=/usr/bin/docker compose up -d
|
||||||
|
ExecStop=/usr/bin/docker compose down
|
||||||
|
TimeoutStartSec=180
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=multi-user.target
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
# Lokale Websuche
|
||||||
|
|
||||||
|
SearXNG übernimmt die Metasuche. TinySearch normalisiert, crawlt und rankt die
|
||||||
|
Ergebnisse lokal. Nur die Web-MCP-Fassade wird dem Modell angeboten; die
|
||||||
|
generischen TinySearch-Werkzeuge bleiben intern.
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
1. `searxng-settings.example.yml` nach `searxng-settings.yml` kopieren.
|
||||||
|
2. `CHANGE_ME_GENERATE_RANDOM_SECRET` durch einen zufälligen Wert ersetzen.
|
||||||
|
3. `tinysearch_config.json` prüfen.
|
||||||
|
4. `docker compose up -d` ausführen.
|
||||||
|
5. TinySearch bleibt ausschließlich über `127.0.0.1:8000` erreichbar.
|
||||||
|
|
||||||
|
Die Containerimages sind per Digest festgeschrieben. Upgrades erfolgen nur
|
||||||
|
bewusst nach Test von Suche, Crawling, Quellenbindung und Kontextgröße.
|
||||||
|
|
||||||
|
`web_search_mcp.py` ist die kompakte, für kleinere Modelle optimierte Fassade.
|
||||||
|
Sie bietet nur `web_search`, `web_compare`, `web_shop` und `web_research` an
|
||||||
|
und verwendet für GitHub und Hugging Face zusätzlich strukturierte APIs.
|
||||||
|
|
||||||
|
## Modellfreundliche Vorgaben
|
||||||
|
|
||||||
|
- kurze Suche: maximal fünf Ergebnisse
|
||||||
|
- Recherche: maximal vier gecrawlte Seiten und acht Evidenz-Chunks
|
||||||
|
- höchstens zwei Chunks je Quelle
|
||||||
|
- Seitenlimit 6000 Tokens, Chunkziel 300 Tokens
|
||||||
|
- externe Inhalte immer als nicht vertrauenswürdig markieren
|
||||||
|
- GitHub und Hugging Face bevorzugt über strukturierte öffentliche APIs
|
||||||
|
- Preise erst nach Prüfung der tatsächlichen Händlerseite als bestätigt melden
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
name: local-ai-web-search
|
||||||
|
|
||||||
|
services:
|
||||||
|
searxng:
|
||||||
|
image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60
|
||||||
|
restart: unless-stopped
|
||||||
|
volumes:
|
||||||
|
- ./searxng-settings.yml:/etc/searxng/settings.yml:ro
|
||||||
|
networks: [search]
|
||||||
|
healthcheck:
|
||||||
|
test: ["CMD", "wget", "-q", "--spider", "http://127.0.0.1:8080/healthz"]
|
||||||
|
interval: 30s
|
||||||
|
timeout: 10s
|
||||||
|
retries: 5
|
||||||
|
start_period: 30s
|
||||||
|
security_opt: ["no-new-privileges:true"]
|
||||||
|
|
||||||
|
tinysearch:
|
||||||
|
image: marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c
|
||||||
|
restart: unless-stopped
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8000:8000"
|
||||||
|
shm_size: "1gb"
|
||||||
|
volumes:
|
||||||
|
- tinysearch-models:/data/models
|
||||||
|
- ./tinysearch_config.json:/config/tinysearch_config.json:ro
|
||||||
|
environment:
|
||||||
|
MCP_TRANSPORT: streamable-http
|
||||||
|
MCP_HOST: 0.0.0.0
|
||||||
|
MCP_PORT: 8000
|
||||||
|
TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json
|
||||||
|
TINYSEARCH_SEARCH_BACKEND: searxng
|
||||||
|
SEARXNG_URL: http://searxng:8080/search
|
||||||
|
depends_on:
|
||||||
|
searxng:
|
||||||
|
condition: service_healthy
|
||||||
|
networks: [search]
|
||||||
|
security_opt: ["no-new-privileges:true"]
|
||||||
|
|
||||||
|
networks:
|
||||||
|
search:
|
||||||
|
driver: bridge
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
tinysearch-models:
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
use_default_settings: true
|
||||||
|
|
||||||
|
general:
|
||||||
|
instance_name: "Local AI Search"
|
||||||
|
debug: false
|
||||||
|
|
||||||
|
search:
|
||||||
|
safe_search: 0
|
||||||
|
autocomplete: ""
|
||||||
|
default_lang: "de"
|
||||||
|
formats: [html, json]
|
||||||
|
|
||||||
|
server:
|
||||||
|
secret_key: "CHANGE_ME_GENERATE_RANDOM_SECRET"
|
||||||
|
limiter: false
|
||||||
|
image_proxy: false
|
||||||
|
bind_address: "0.0.0.0"
|
||||||
|
port: 8080
|
||||||
|
|
||||||
|
outgoing:
|
||||||
|
request_timeout: 8.0
|
||||||
|
max_request_timeout: 15.0
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
{
|
||||||
|
"search_backend": "searxng",
|
||||||
|
"search_backend_url": "http://searxng:8080/search",
|
||||||
|
"search_backend_fallback": true,
|
||||||
|
"search_region": "de-de",
|
||||||
|
"search_max_results": 5,
|
||||||
|
"scrape_max_tokens": 1200,
|
||||||
|
"search_top_k": 15,
|
||||||
|
"search_dense_weight": 0.5,
|
||||||
|
"search_max_results_to_keep": 4,
|
||||||
|
"chunk_dense_weight": 0.5,
|
||||||
|
"chunk_max_results_to_keep": 8,
|
||||||
|
"chunk_rank_oversample": 3,
|
||||||
|
"chunk_dedupe_jaccard_threshold": 0.92,
|
||||||
|
"chunk_max_per_source_url": 2,
|
||||||
|
"max_concurrent_crawls": 3,
|
||||||
|
"pipeline_timeout_seconds": 90.0,
|
||||||
|
"crawl_fit_markdown_mode": "bm25",
|
||||||
|
"crawl_fit_min_chars": 200,
|
||||||
|
"crawl_bm25_threshold": 1.5,
|
||||||
|
"crawl_bm25_language": "german",
|
||||||
|
"crawl_max_chunk_tokens": 300,
|
||||||
|
"crawl_overlap_tokens": 40,
|
||||||
|
"crawl_max_page_tokens": 6000,
|
||||||
|
"embedding_backend": "onnx",
|
||||||
|
"embedding_model": "fast",
|
||||||
|
"dense_document_embed_batch_size": 32,
|
||||||
|
"encoding_name": "embedding",
|
||||||
|
"blocked_domains": [],
|
||||||
|
"trace_path": ""
|
||||||
|
}
|
||||||
File diff suppressed because it is too large.
Load diff
@@ -108,7 +108,7 @@ IMAGE_VRAM_FREE_TIMEOUT = float(os.environ.get("IMAGE_VRAM_FREE_TIMEOUT", "90"))
|
|||||||
# temporären Hotswap aus: Hauptprofil raus → Q3+mmproj rein →
|
# temporären Hotswap aus: Hauptprofil raus → Q3+mmproj rein →
|
||||||
# Bild analysieren → Q3 raus → Hauptprofil rein (try/finally).
|
# Bild analysieren → Q3 raus → Hauptprofil rein (try/finally).
|
||||||
LLAMA_SERVER_BIN = os.environ.get(
|
LLAMA_SERVER_BIN = os.environ.get(
|
||||||
"LLAMA_SERVER_BIN", "/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server")
|
"LLAMA_SERVER_BIN", "/opt/mike-ai/llama.cpp/build/bin/llama-server")
|
||||||
VISION_MODEL = os.environ.get(
|
VISION_MODEL = os.environ.get(
|
||||||
"VISION_MODEL", "/opt/mike-ai/models/qwen3.8-27b/Qwen3.8-27B-Q3_K_M.gguf")
|
"VISION_MODEL", "/opt/mike-ai/models/qwen3.8-27b/Qwen3.8-27B-Q3_K_M.gguf")
|
||||||
VISION_MMPROJ = os.environ.get(
|
VISION_MMPROJ = os.environ.get(
|
||||||
|
|||||||
Reference in new issue
Block a user