Expand router into reproducible local AI platform
This commit is contained in:
Executable
+70
@@ -0,0 +1,70 @@
|
||||
#!/usr/bin/env bash
|
||||
set -uo pipefail
|
||||
|
||||
PASS=0
|
||||
WARN=0
|
||||
FAIL=0
|
||||
|
||||
pass() { printf 'PASS %s\n' "$*"; PASS=$((PASS + 1)); }
|
||||
warn() { printf 'WARN %s\n' "$*"; WARN=$((WARN + 1)); }
|
||||
fail() { printf 'FAIL %s\n' "$*"; FAIL=$((FAIL + 1)); }
|
||||
|
||||
echo "== Local AI platform verification =="
|
||||
|
||||
ROOT_USE="$(df -P / | awk 'NR==2 {gsub(/%/, "", $5); print $5}')"
|
||||
if [[ -n "$ROOT_USE" && "$ROOT_USE" -lt 85 ]]; then
|
||||
pass "Systempartition bei ${ROOT_USE}%"
|
||||
else
|
||||
fail "Systempartition bei ${ROOT_USE:-unbekannt}% (Ziel: unter 85%)"
|
||||
fi
|
||||
|
||||
if command -v nvidia-smi >/dev/null 2>&1; then
|
||||
GPU="$(nvidia-smi --query-gpu=name,memory.total --format=csv,noheader 2>/dev/null | head -1)"
|
||||
[[ -n "$GPU" ]] && pass "GPU erkannt: $GPU" || fail "nvidia-smi liefert keine GPU"
|
||||
else
|
||||
fail "nvidia-smi fehlt"
|
||||
fi
|
||||
|
||||
for service in mike-ai-llama-ui mike-ai-profile-router; do
|
||||
if systemctl is-active --quiet "$service"; then
|
||||
pass "$service aktiv"
|
||||
else
|
||||
fail "$service nicht aktiv"
|
||||
fi
|
||||
done
|
||||
|
||||
for optional in mike-ai-whisper mike-ai-xtts mike-ai-web-search; do
|
||||
if systemctl is-active --quiet "$optional"; then
|
||||
pass "$optional aktiv"
|
||||
else
|
||||
warn "$optional nicht aktiv oder nicht installiert"
|
||||
fi
|
||||
done
|
||||
|
||||
if curl -fsS --max-time 3 http://127.0.0.1:8080/health >/dev/null; then
|
||||
pass "llama.cpp Health-Check"
|
||||
else
|
||||
fail "llama.cpp auf Port 8080 nicht gesund"
|
||||
fi
|
||||
|
||||
if STATUS="$(curl -fsS --max-time 3 http://127.0.0.1:8081/status 2>/dev/null)"; then
|
||||
PROFILE="$(python3 -c 'import json,sys; print(json.load(sys.stdin).get("current_profile"))' <<<"$STATUS" 2>/dev/null)"
|
||||
pass "Router erreichbar, Profil ${PROFILE:-unbekannt}"
|
||||
else
|
||||
fail "Router auf Port 8081 nicht erreichbar"
|
||||
fi
|
||||
|
||||
if systemctl list-unit-files --no-legend 2>/dev/null | grep -Eq '(vision-rx|whisper-rx|granite-rx).*enabled'; then
|
||||
warn "Aktivierte RX-Altlast gefunden"
|
||||
else
|
||||
pass "Keine aktivierte RX-Altlast"
|
||||
fi
|
||||
|
||||
if systemctl list-unit-files --no-legend 2>/dev/null | grep -E '(benchmark|race).*enabled' >/dev/null; then
|
||||
warn "Automatisch aktivierter Benchmark-/Race-Dienst gefunden"
|
||||
else
|
||||
pass "Keine automatisch aktivierten Benchmarks"
|
||||
fi
|
||||
|
||||
printf '\nErgebnis: %d PASS, %d WARN, %d FAIL\n' "$PASS" "$WARN" "$FAIL"
|
||||
[[ "$FAIL" -eq 0 ]]
|
||||
Executable
+49
@@ -0,0 +1,49 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
if [[ $EUID -ne 0 ]]; then
|
||||
echo "Dieses Skript muss als root ausgeführt werden." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
PLATFORM="$REPO_ROOT/platform"
|
||||
PROFILE_TARGET=/opt/mike-ai/platform/profiles
|
||||
DROPIN=/etc/systemd/system/mike-ai-llama-ui.service.d
|
||||
WEB_TARGET=/opt/mike-ai/web-search
|
||||
|
||||
install -d -m 0755 "$PROFILE_TARGET" "$DROPIN" "$WEB_TARGET" /etc/mike-ai
|
||||
install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \
|
||||
/etc/systemd/system/mike-ai-llama-ui.service
|
||||
install -m 0644 "$PLATFORM/systemd/mike-ai-web-search.service" \
|
||||
/etc/systemd/system/mike-ai-web-search.service
|
||||
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
|
||||
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
|
||||
install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf"
|
||||
install -m 0644 "$PLATFORM/web-search/compose.yaml" "$WEB_TARGET/compose.yaml"
|
||||
install -m 0644 "$PLATFORM/web-search/tinysearch_config.json" \
|
||||
"$WEB_TARGET/tinysearch_config.json"
|
||||
install -m 0755 "$PLATFORM/web-search/web_search_mcp.py" \
|
||||
"$WEB_TARGET/web_search_mcp.py"
|
||||
if [[ ! -e "$WEB_TARGET/searxng-settings.yml" ]]; then
|
||||
install -m 0600 "$PLATFORM/web-search/searxng-settings.example.yml" \
|
||||
"$WEB_TARGET/searxng-settings.yml.example"
|
||||
fi
|
||||
|
||||
if [[ ! -e /etc/mike-ai/mcp-servers.json ]]; then
|
||||
install -m 0600 "$PLATFORM/mcp/mcp-servers.example.json" \
|
||||
/etc/mike-ai/mcp-servers.json.example
|
||||
fi
|
||||
|
||||
systemctl daemon-reload
|
||||
|
||||
cat <<'EOF'
|
||||
Kernkonfiguration installiert, aber noch nicht gestartet.
|
||||
|
||||
Vor dem Start:
|
||||
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
|
||||
2. /etc/mike-ai/mcp-servers.json mit minimalen Servern erstellen.
|
||||
3. llama.cpp bauen.
|
||||
4. Danach: llama-profile fast
|
||||
EOF
|
||||
@@ -0,0 +1 @@
|
||||
4df29be4f4c3673f428170fda944a5b19f743bb8
|
||||
@@ -0,0 +1,31 @@
|
||||
# llama.cpp und Profile
|
||||
|
||||
Der produktive Build wird über `LLAMA_CPP_COMMIT` festgeschrieben. Damit ist
|
||||
ein Neuaufbau unabhängig vom jeweils aktuellen Stand des Upstream-Master.
|
||||
|
||||
## Build
|
||||
|
||||
```text
|
||||
sudo platform/llama/build-llama-cpp.sh
|
||||
```
|
||||
|
||||
Der Build aktiviert CUDA und den HTTP-Server. Änderungen am Commit werden erst
|
||||
nach Standardbenchmark, Tool-Calling-Test und Kontexttest übernommen.
|
||||
|
||||
## Produktive Modelle
|
||||
|
||||
- Fast und Long: Qwen3.8-27B IQ4-MIX mit MTP2
|
||||
- Medium: Qwen3.8-27B IQ4_XS Pure ohne MTP
|
||||
- Vision-Hotswap: Qwen3.8-27B Q3_K_M plus BF16-mmproj
|
||||
|
||||
Die Dateien selbst sind nicht Bestandteil des Repositories. Pfade und Hashes
|
||||
werden im lokalen Modellmanifest verwaltet.
|
||||
|
||||
## Profilinstallation
|
||||
|
||||
Die Vorlagen unter `platform/profiles` verwenden Umgebungsvariablen in einer
|
||||
gemeinsamen Environment-Datei. Für die aktuelle produktive Installation können
|
||||
sie alternativ als dokumentierte Referenz für vollständige systemd-Overrides
|
||||
verwendet werden.
|
||||
|
||||
Der Profilwechsel erfolgt ausschließlich über `platform/scripts/llama-profile`.
|
||||
Executable
+29
@@ -0,0 +1,29 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
ROOT_DIR="${ROOT_DIR:-/opt/mike-ai}"
|
||||
SOURCE_DIR="${SOURCE_DIR:-$ROOT_DIR/llama.cpp}"
|
||||
BUILD_DIR="${BUILD_DIR:-$SOURCE_DIR/build}"
|
||||
REPO_URL="${REPO_URL:-https://github.com/ggml-org/llama.cpp.git}"
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
COMMIT="$(tr -d '[:space:]' < "$SCRIPT_DIR/LLAMA_CPP_COMMIT")"
|
||||
|
||||
if [[ $EUID -ne 0 ]]; then
|
||||
echo "Dieses Skript muss als root ausgeführt werden." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ ! -d "$SOURCE_DIR/.git" ]]; then
|
||||
git clone "$REPO_URL" "$SOURCE_DIR"
|
||||
fi
|
||||
|
||||
git -C "$SOURCE_DIR" fetch --tags origin
|
||||
git -C "$SOURCE_DIR" checkout --detach "$COMMIT"
|
||||
|
||||
cmake -S "$SOURCE_DIR" -B "$BUILD_DIR" \
|
||||
-DGGML_CUDA=ON \
|
||||
-DLLAMA_CURL=ON \
|
||||
-DCMAKE_BUILD_TYPE=Release
|
||||
cmake --build "$BUILD_DIR" --config Release --parallel "$(nproc)"
|
||||
|
||||
"$BUILD_DIR/bin/llama-server" --version
|
||||
@@ -0,0 +1,43 @@
|
||||
# MCP-Architektur
|
||||
|
||||
Die produktive MCP-Konfiguration ist absichtlich nicht Bestandteil des Git-
|
||||
Repositories, weil sie lokale Pfade und Zugangsdaten referenziert. Das Beispiel
|
||||
zeigt nur die Struktur.
|
||||
|
||||
## Empfohlene Server
|
||||
|
||||
- `web`: Websuche über lokales TinySearch/SearXNG
|
||||
- `homeassistant`: Administration mit eigenem, minimal berechtigtem Token
|
||||
- `arr`: Sonarr/Radarr über spezialisierte Aktionen
|
||||
- `unraid-readonly`: Diagnose ohne Schreiboperationen
|
||||
|
||||
## Getrennte Konfigurationen
|
||||
|
||||
Statt alle Werkzeuge ständig zu laden, werden mehrere Dateien empfohlen:
|
||||
|
||||
```text
|
||||
/etc/mike-ai/mcp-standard.json
|
||||
/etc/mike-ai/mcp-homeassistant.json
|
||||
/etc/mike-ai/mcp-arr.json
|
||||
/etc/mike-ai/mcp-unraid-readonly.json
|
||||
/etc/mike-ai/mcp-unraid-write.json
|
||||
```
|
||||
|
||||
Das jeweilige Profil verweist nur auf die benötigte Datei. Dadurch werden die
|
||||
Tool-Schemas kleiner, das Kontextfenster bleibt frei und kleine Modelle müssen
|
||||
weniger Werkzeuge unterscheiden.
|
||||
|
||||
Credentials werden von schmalen Wrapper-Programmen wie `run-arr-mcp` oder
|
||||
`runraid` aus geschützten Environment-Dateien geladen. Das JSON selbst enthält
|
||||
weder Werte noch Pfade zu einzelnen Tokens.
|
||||
|
||||
## Schreibzugriff
|
||||
|
||||
Schreibende Server gehören nicht in `mcp-standard.json`. Sie benötigen eine
|
||||
Vorschau und ein an die exakte Änderung gebundenes Approval Ticket.
|
||||
|
||||
## Shell
|
||||
|
||||
Ein allgemeiner Shell-MCP ist nicht Teil der Zielplattform. Insbesondere
|
||||
`python3`, `ssh`, `scp`, `curl` und `systemctl` dürfen nicht gemeinsam als
|
||||
scheinbar harmlose Allowlist angeboten werden.
|
||||
@@ -0,0 +1,23 @@
|
||||
{
|
||||
"mcpServers": {
|
||||
"web": {
|
||||
"command": "/usr/bin/python3",
|
||||
"args": ["/opt/mike-ai/web-search/web_search_mcp.py"]
|
||||
},
|
||||
"homeassistant": {
|
||||
"command": "/usr/local/bin/homeassistant-native-mcp",
|
||||
"args": [],
|
||||
"timeout_ms": 60000
|
||||
},
|
||||
"arr": {
|
||||
"command": "/usr/local/bin/run-arr-mcp",
|
||||
"args": [],
|
||||
"timeout_ms": 30000
|
||||
},
|
||||
"unraid-readonly": {
|
||||
"command": "/usr/local/bin/runraid",
|
||||
"args": ["mcp"],
|
||||
"timeout_ms": 60000
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
schema: 1
|
||||
models:
|
||||
qwen_fast_long:
|
||||
role: primary-text-fast-and-long
|
||||
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||
file: Qwen3.8-27B-IQ4-MIX.gguf
|
||||
target: /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||
qwen_medium:
|
||||
role: primary-text-medium
|
||||
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||
file: qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
target: /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||
qwen_vision:
|
||||
role: temporary-vision-model
|
||||
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||
file: Qwen3.8-27B-Q3_K_M.gguf
|
||||
target: /opt/mike-ai/models/qwen3.8-27b/Qwen3.8-27B-Q3_K_M.gguf
|
||||
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||
qwen_vision_projector:
|
||||
role: vision-projector
|
||||
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||
file: mmproj-BF16.gguf
|
||||
target: /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf
|
||||
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||
whisper:
|
||||
role: speech-to-text
|
||||
source: ggml-org/whisper.cpp
|
||||
file: ggml-large-v3-turbo.bin
|
||||
target: /opt/mike-ai/models/whisper/ggml-large-v3-turbo.bin
|
||||
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||
flux:
|
||||
role: image-generation
|
||||
source: black-forest-labs/FLUX.2-klein-base-4B
|
||||
target: /opt/mike-ai/models/FLUX.2-klein-base-4B
|
||||
revision: "PIN_EXACT_REVISION"
|
||||
xtts:
|
||||
role: text-to-speech
|
||||
source: coqui/XTTS-v2
|
||||
target: /opt/mike-ai/xtts/.cache
|
||||
revision: "PIN_EXACT_REVISION"
|
||||
@@ -0,0 +1,6 @@
|
||||
[Unit]
|
||||
Description=Local AI llama.cpp - Qwen Fast 72K MTP2
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --alias qwen38-27b-iq4mix-72k-mtp2 --ctx-size 73728 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
|
||||
@@ -0,0 +1,6 @@
|
||||
[Unit]
|
||||
Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
|
||||
@@ -0,0 +1,6 @@
|
||||
[Unit]
|
||||
Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none
|
||||
Executable
+35
@@ -0,0 +1,35 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
PROFILE_DIR="${PROFILE_DIR:-/etc/systemd/system/mike-ai-llama-ui.service.d}"
|
||||
SOURCE_DIR="${SOURCE_DIR:-/opt/mike-ai/platform/profiles}"
|
||||
SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
|
||||
PROFILE="${1:-}"
|
||||
|
||||
case "$PROFILE" in
|
||||
fast|medium|long) ;;
|
||||
large) PROFILE=long ;;
|
||||
*)
|
||||
echo "Usage: llama-profile {fast|medium|long|large}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
SOURCE="$SOURCE_DIR/profile-$PROFILE.conf"
|
||||
TARGET="$PROFILE_DIR/override.conf"
|
||||
|
||||
if [[ ! -r "$SOURCE" ]]; then
|
||||
echo "Profildatei fehlt: $SOURCE" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
install -d -m 0755 "$PROFILE_DIR"
|
||||
TEMP="$(mktemp "$PROFILE_DIR/.override.conf.XXXXXX")"
|
||||
trap 'rm -f "$TEMP"' EXIT
|
||||
install -m 0644 "$SOURCE" "$TEMP"
|
||||
mv -f "$TEMP" "$TARGET"
|
||||
trap - EXIT
|
||||
|
||||
systemctl daemon-reload
|
||||
systemctl restart "$SERVICE"
|
||||
echo "Profil '$PROFILE' wurde aktiviert."
|
||||
@@ -0,0 +1,19 @@
|
||||
[Unit]
|
||||
Description=Local AI llama.cpp server
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
# ExecStart wird vollständig durch das aktive Profil-Override definiert.
|
||||
ExecStart=/usr/bin/false
|
||||
Restart=on-failure
|
||||
RestartSec=3
|
||||
TimeoutStartSec=600
|
||||
TimeoutStopSec=120
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
LimitNOFILE=1048576
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
@@ -0,0 +1,16 @@
|
||||
[Unit]
|
||||
Description=Local AI Web Search (TinySearch + SearXNG)
|
||||
Requires=docker.service
|
||||
After=docker.service network-online.target
|
||||
Before=mike-ai-llama-ui.service
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
RemainAfterExit=yes
|
||||
WorkingDirectory=/opt/mike-ai/web-search
|
||||
ExecStart=/usr/bin/docker compose up -d
|
||||
ExecStop=/usr/bin/docker compose down
|
||||
TimeoutStartSec=180
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
@@ -0,0 +1,30 @@
|
||||
# Lokale Websuche
|
||||
|
||||
SearXNG übernimmt die Metasuche. TinySearch normalisiert, crawlt und rankt die
|
||||
Ergebnisse lokal. Nur die Web-MCP-Fassade wird dem Modell angeboten; die
|
||||
generischen TinySearch-Werkzeuge bleiben intern.
|
||||
|
||||
## Installation
|
||||
|
||||
1. `searxng-settings.example.yml` nach `searxng-settings.yml` kopieren.
|
||||
2. `CHANGE_ME_GENERATE_RANDOM_SECRET` durch einen zufälligen Wert ersetzen.
|
||||
3. `tinysearch_config.json` prüfen.
|
||||
4. `docker compose up -d` ausführen.
|
||||
5. TinySearch bleibt ausschließlich über `127.0.0.1:8000` erreichbar.
|
||||
|
||||
Die Containerimages sind per Digest festgeschrieben. Upgrades erfolgen nur
|
||||
bewusst nach Test von Suche, Crawling, Quellenbindung und Kontextgröße.
|
||||
|
||||
`web_search_mcp.py` ist die kompakte, für kleinere Modelle optimierte Fassade.
|
||||
Sie bietet nur `web_search`, `web_compare`, `web_shop` und `web_research` an
|
||||
und verwendet für GitHub und Hugging Face zusätzlich strukturierte APIs.
|
||||
|
||||
## Modellfreundliche Vorgaben
|
||||
|
||||
- kurze Suche: maximal fünf Ergebnisse
|
||||
- Recherche: maximal vier gecrawlte Seiten und acht Evidenz-Chunks
|
||||
- höchstens zwei Chunks je Quelle
|
||||
- Seitenlimit 6000 Tokens, Chunkziel 300 Tokens
|
||||
- externe Inhalte immer als nicht vertrauenswürdig markieren
|
||||
- GitHub und Hugging Face bevorzugt über strukturierte öffentliche APIs
|
||||
- Preise erst nach Prüfung der tatsächlichen Händlerseite als bestätigt melden
|
||||
@@ -0,0 +1,45 @@
|
||||
name: local-ai-web-search
|
||||
|
||||
services:
|
||||
searxng:
|
||||
image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- ./searxng-settings.yml:/etc/searxng/settings.yml:ro
|
||||
networks: [search]
|
||||
healthcheck:
|
||||
test: ["CMD", "wget", "-q", "--spider", "http://127.0.0.1:8080/healthz"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
start_period: 30s
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
|
||||
tinysearch:
|
||||
image: marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "127.0.0.1:8000:8000"
|
||||
shm_size: "1gb"
|
||||
volumes:
|
||||
- tinysearch-models:/data/models
|
||||
- ./tinysearch_config.json:/config/tinysearch_config.json:ro
|
||||
environment:
|
||||
MCP_TRANSPORT: streamable-http
|
||||
MCP_HOST: 0.0.0.0
|
||||
MCP_PORT: 8000
|
||||
TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json
|
||||
TINYSEARCH_SEARCH_BACKEND: searxng
|
||||
SEARXNG_URL: http://searxng:8080/search
|
||||
depends_on:
|
||||
searxng:
|
||||
condition: service_healthy
|
||||
networks: [search]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
|
||||
networks:
|
||||
search:
|
||||
driver: bridge
|
||||
|
||||
volumes:
|
||||
tinysearch-models:
|
||||
@@ -0,0 +1,22 @@
|
||||
use_default_settings: true
|
||||
|
||||
general:
|
||||
instance_name: "Local AI Search"
|
||||
debug: false
|
||||
|
||||
search:
|
||||
safe_search: 0
|
||||
autocomplete: ""
|
||||
default_lang: "de"
|
||||
formats: [html, json]
|
||||
|
||||
server:
|
||||
secret_key: "CHANGE_ME_GENERATE_RANDOM_SECRET"
|
||||
limiter: false
|
||||
image_proxy: false
|
||||
bind_address: "0.0.0.0"
|
||||
port: 8080
|
||||
|
||||
outgoing:
|
||||
request_timeout: 8.0
|
||||
max_request_timeout: 15.0
|
||||
@@ -0,0 +1,31 @@
|
||||
{
|
||||
"search_backend": "searxng",
|
||||
"search_backend_url": "http://searxng:8080/search",
|
||||
"search_backend_fallback": true,
|
||||
"search_region": "de-de",
|
||||
"search_max_results": 5,
|
||||
"scrape_max_tokens": 1200,
|
||||
"search_top_k": 15,
|
||||
"search_dense_weight": 0.5,
|
||||
"search_max_results_to_keep": 4,
|
||||
"chunk_dense_weight": 0.5,
|
||||
"chunk_max_results_to_keep": 8,
|
||||
"chunk_rank_oversample": 3,
|
||||
"chunk_dedupe_jaccard_threshold": 0.92,
|
||||
"chunk_max_per_source_url": 2,
|
||||
"max_concurrent_crawls": 3,
|
||||
"pipeline_timeout_seconds": 90.0,
|
||||
"crawl_fit_markdown_mode": "bm25",
|
||||
"crawl_fit_min_chars": 200,
|
||||
"crawl_bm25_threshold": 1.5,
|
||||
"crawl_bm25_language": "german",
|
||||
"crawl_max_chunk_tokens": 300,
|
||||
"crawl_overlap_tokens": 40,
|
||||
"crawl_max_page_tokens": 6000,
|
||||
"embedding_backend": "onnx",
|
||||
"embedding_model": "fast",
|
||||
"dense_document_embed_batch_size": 32,
|
||||
"encoding_name": "embedding",
|
||||
"blocked_domains": [],
|
||||
"trace_path": ""
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user