Synchronize repository with Athena deployment

This commit is contained in:
Mikei386
2026-09-13 20:01:36 +02:00
parent 040a2df48b
commit fce9900389
60 changed files with 2570 additions and 607 deletions
+78
View File
@@ -0,0 +1,78 @@
#!/usr/bin/env bash
# Encrypted off-host backup for data-disk and total-loss recovery.
set -Eeuo pipefail
umask 077
CONFIG=${DISASTER_BACKUP_CONFIG:-/etc/mike-ai/disaster-backup.env}
STATE=/var/lib/mike-ai-disaster-backup
log() { printf '\n==> %s\n' "$*"; }
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
[[ -r $CONFIG ]] || die "Konfiguration fehlt: $CONFIG"
# shellcheck disable=SC1090
source "$CONFIG"
[[ ${DISASTER_BACKUP_ENABLED:-false} == true ]] || die \
"Externes Backup ist noch nicht freigeschaltet (DISASTER_BACKUP_ENABLED=true)."
[[ -n ${RESTIC_REPOSITORY:-} ]] || die "RESTIC_REPOSITORY fehlt."
if [[ -n ${RESTIC_REQUIRE_MOUNT:-} ]]; then
mountpoint -q "$RESTIC_REQUIRE_MOUNT" || die \
"Externes Backupziel ist nicht eingehängt: $RESTIC_REQUIRE_MOUNT"
fi
[[ -n ${RESTIC_PASSWORD_FILE:-} && -r $RESTIC_PASSWORD_FILE ]] || die \
"RESTIC_PASSWORD_FILE fehlt oder ist nicht lesbar."
command -v restic >/dev/null || die "restic ist nicht installiert."
command -v docker >/dev/null || die "Docker ist nicht installiert."
exec 9>/run/lock/athena-disaster-backup.lock
flock -n 9 || die "Ein Disaster-Backup läuft bereits."
install -d -m 0700 "$STATE/latest"
log "Konsistentes Docker-Schnellbackup erzeugen"
docker inspect mike-ai-backup >/dev/null 2>&1 || die "mike-ai-backup fehlt."
docker exec mike-ai-backup backup
latest=$(readlink -f /data/docker-backups/athena-latest.tar.gz)
[[ -s $latest ]] || die "Lokales Docker-Backup wurde nicht erzeugt."
gzip -t "$latest" || die "Lokales Docker-Backup ist beschädigt."
install -m 0600 "$latest" "$STATE/latest/docker-state.tar.gz"
sha256sum "$STATE/latest/docker-state.tar.gz" >"$STATE/latest/docker-state.tar.gz.sha256"
log "Wiederaufbau-Metadaten erfassen"
{
printf 'created_utc=%s\n' "$(date -u +%FT%TZ)"
printf 'hostname=%s\n' "$(hostname)"
printf 'source_commit=%s\n' "$(git -C /opt/mike-ai/stack rev-parse HEAD 2>/dev/null || printf unknown)"
findmnt -rn -o SOURCE,UUID,FSTYPE,TARGET / /data 2>/dev/null || true
} >"$STATE/latest/manifest.txt"
find /data/models -type f -printf '%P\t%s\n' 2>/dev/null | sort \
>"$STATE/latest/model-manifest.tsv"
docker ps -a --format '{{.Names}}\t{{.Image}}\t{{.Status}}' \
>"$STATE/latest/container-manifest.tsv"
paths=(/etc/mike-ai /opt/mike-ai "$STATE/latest")
for path in \
/data/voice /data/music /data/audio /data/llama-dashboard \
/data/mike-ai-operator /data/benchmarks /data/model-benchmarks \
/data/image-comparison /data/backups /data/deploy-backups; do
[[ ! -e $path ]] || paths+=("$path")
done
tag=${RESTIC_TAG:-athena-disaster}
log "Verschlüsseltes externes Backup schreiben"
if ! restic snapshots >/dev/null 2>&1; then
log "Neues Restic-Repository initialisieren"
restic init
fi
restic backup --tag "$tag" "${paths[@]}"
log "Aufbewahrung anwenden"
restic forget --tag "$tag" \
--keep-daily "${RESTIC_KEEP_DAILY:-14}" \
--keep-weekly "${RESTIC_KEEP_WEEKLY:-8}" \
--keep-monthly "${RESTIC_KEEP_MONTHLY:-12}" --prune
log "Letzten Snapshot verifizieren"
restic snapshots --tag "$tag" --latest 1
restic check
printf 'ATHENA_DISASTER_BACKUP_OK\n'
@@ -0,0 +1,12 @@
[Unit]
Description=Encrypted off-host disaster backup for Athena
After=docker.service network-online.target
Wants=network-online.target
ConditionPathExists=/etc/mike-ai/disaster-backup.env
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/athena-disaster-backup
Nice=10
IOSchedulingClass=best-effort
IOSchedulingPriority=7
@@ -0,0 +1,11 @@
[Unit]
Description=Nightly Athena off-host disaster backup
[Timer]
OnCalendar=*-*-* 03:15:00
Persistent=true
RandomizedDelaySec=30m
Unit=athena-disaster-backup.service
[Install]
WantedBy=timers.target
+82
View File
@@ -0,0 +1,82 @@
#!/usr/bin/env bash
# Build a browser-downloadable, encrypted archive of irreplaceable Athena data.
set -Eeuo pipefail
umask 077
OUTPUT_DIR=${ATHENA_EXPORT_DIR:-/data/emergency-backups}
RECIPIENT_FILE=${ATHENA_AGE_RECIPIENT_FILE:-/etc/mike-ai/recovery.age-recipient}
STATE=/var/lib/mike-ai-disaster-backup/latest
KEEP=${ATHENA_EXPORT_KEEP:-5}
log() { printf '\n==> %s\n' "$*"; }
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
[[ -s $RECIPIENT_FILE ]] || die "Age-Empfänger fehlt: $RECIPIENT_FILE"
[[ $KEEP =~ ^[1-9][0-9]*$ ]] || die "ATHENA_EXPORT_KEEP muss positiv sein."
for command in age zstd tar docker sha256sum flock; do
command -v "$command" >/dev/null || die "$command fehlt."
done
exec 9>/run/lock/athena-export-backup.lock
flock -n 9 || die "Ein exportierbares Backup läuft bereits."
install -d -m 0755 "$OUTPUT_DIR"
install -d -m 0700 "$STATE"
log "Aktuellen Docker-Zustand sichern"
docker exec mike-ai-backup backup
latest=$(readlink -f /data/docker-backups/athena-latest.tar.gz)
[[ -s $latest ]] || die "Docker-Zustandsbackup fehlt."
gzip -t "$latest" || die "Docker-Zustandsbackup ist beschädigt."
install -m 0600 "$latest" "$STATE/docker-state.tar.gz"
stamp=$(date -u +%Y-%m-%dT%H-%M-%SZ)
name="athena-portable-$stamp.tar.zst.age"
partial="$OUTPUT_DIR/.$name.partial"
target="$OUTPUT_DIR/$name"
list=$(mktemp /tmp/athena-export-list.XXXXXX)
trap 'rm -f "$list" "$partial"' EXIT
add_path() {
local path=${1#/}
[[ ! -e /$path ]] || printf '%s\0' "$path" >>"$list"
}
# Reproducible model/HF caches are deliberately omitted. Everything below is
# either a host configuration, project source, user input or generated result.
add_path /etc/mike-ai
add_path /opt/mike-ai
add_path /var/lib/mike-ai-disaster-backup/latest
add_path /data/voice/applio/logs
add_path /data/voice/applio/datasets
add_path /data/voice/applio/config.json
add_path /data/voice/omnivoice/output
add_path /data/voice/xvc/output
add_path /data/voice/studio
add_path /data/music
add_path /data/audio
add_path /data/llama-dashboard
add_path /data/mike-ai-operator
add_path /data/benchmarks
add_path /data/model-benchmarks
add_path /data/image-comparison
add_path /data/backups
add_path /data/deploy-backups
log "Portables, verschlüsseltes Backup erzeugen"
tar --create --numeric-owner --acls --xattrs -C / --null --files-from="$list" \
| zstd -T0 -3 \
| age -R "$RECIPIENT_FILE" -o "$partial"
chmod 0644 "$partial"
mv "$partial" "$target"
sha256sum "$target" >"$target.sha256"
chmod 0644 "$target.sha256"
log "Nur die letzten $KEEP Generationen behalten"
mapfile -t old < <(find "$OUTPUT_DIR" -maxdepth 1 -type f \
-name 'athena-portable-*.tar.zst.age' -printf '%T@ %p\n' | sort -rn | tail -n +$((KEEP + 1)) | cut -d' ' -f2-)
for archive in "${old[@]}"; do
rm -f -- "$archive" "$archive.sha256"
done
printf 'ATHENA_EXPORT_BACKUP_OK file=%s bytes=%s\n' "$target" "$(stat -c %s "$target")"
@@ -0,0 +1,12 @@
[Unit]
Description=Create encrypted downloadable Athena recovery package
After=docker.service
Requires=docker.service
ConditionPathExists=/etc/mike-ai/recovery.age-recipient
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/athena-export-backup
Nice=10
IOSchedulingClass=best-effort
IOSchedulingPriority=7
@@ -0,0 +1,12 @@
[Unit]
Description=Create an Athena recovery package every five hours
[Timer]
OnBootSec=45m
OnUnitActiveSec=5h
Persistent=true
RandomizedDelaySec=10m
Unit=athena-export-backup.service
[Install]
WantedBy=timers.target
+3 -1
View File
@@ -13,9 +13,11 @@ RUN apt-get update && apt-get install -y --no-install-recommends python3.12-venv
"transformers==${TRANSFORMERS_VERSION}" \
"accelerate==${ACCELERATE_VERSION}" \
"huggingface-hub==${HF_HUB_VERSION}" \
"nvidia-modelopt==0.46.0" bitsandbytes \
sentencepiece protobuf safetensors pillow && \
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker
COPY image_worker.py /app/image_worker.py
COPY image_worker_9b.py /app/image_worker_9b.py
USER 10002:10002
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker.py"]
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker_9b.py"]
@@ -0,0 +1,282 @@
#!/usr/bin/env python3
"""Private FLUX.2 Klein 9B FP8 beta worker for Athena's two GPUs.
The FP8 diffusion transformer runs on the RTX 5080. A Qwen3-8B NF4 text
encoder runs on the RTX 3060 while the profile controller temporarily pauses
Qwen3-TTS. The transformer and encoder are released before VAE decoding so
the 1024px decoder has sufficient workspace on the RTX 5080.
"""
from __future__ import annotations
import gc
import json
import os
import signal
import time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from types import MethodType
HOST = os.environ.get("WORKER_HOST", "0.0.0.0")
PORT = int(os.environ.get("WORKER_PORT", "8086"))
TOKEN = os.environ.get("WORKER_TOKEN", "").strip()
COMPONENT_DIR = os.environ.get("FLUX_COMPONENT_DIR", "/models/components")
TRANSFORMER_FILE = os.environ.get(
"FLUX_TRANSFORMER_FILE", "/models/fp8/flux-2-klein-9b-fp8.safetensors")
OUTPUT_DIR = Path(os.environ.get("IMAGE_DIR", "/data/images")).resolve()
ACTIVE = False
os.environ.setdefault("DIFFUSERS_VERBOSITY", "error")
os.environ.setdefault("TRANSFORMERS_VERBOSITY", "error")
os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
if len(TOKEN) < 32:
raise RuntimeError("WORKER_TOKEN is missing or too short")
signal.signal(signal.SIGTERM, lambda *_: os._exit(0))
def _devices(torch):
if torch.cuda.device_count() != 2:
raise RuntimeError("FLUX 9B beta requires exactly two visible CUDA GPUs")
totals = {i: torch.cuda.get_device_properties(i).total_memory
for i in range(torch.cuda.device_count())}
transformer_index = max(totals, key=totals.get)
encoder_index = min(totals, key=totals.get)
return (transformer_index, encoder_index,
torch.device(f"cuda:{transformer_index}"),
torch.device(f"cuda:{encoder_index}"))
def _install_fp8_converter():
import diffusers.loaders.single_file_model as single_file_model
original = single_file_model.SINGLE_FILE_LOADABLE_CLASSES[
"Flux2Transformer2DModel"]["checkpoint_mapping_fn"]
scales = {}
double_map = {
"img_attn.proj": "attn.to_out.0",
"img_mlp.0": "ff.linear_in",
"img_mlp.2": "ff.linear_out",
"txt_attn.proj": "attn.to_add_out",
"txt_mlp.0": "ff_context.linear_in",
"txt_mlp.2": "ff_context.linear_out",
}
single_map = {
"linear1": "attn.to_qkv_mlp_proj",
"linear2": "attn.to_out",
}
def record(key, value):
parts = key.split(".")
scale_name, block = parts[-1], parts[1]
within = ".".join(parts[2:-1])
if parts[0] == "double_blocks":
if within == "img_attn.qkv":
targets = ("attn.to_q", "attn.to_k", "attn.to_v")
elif within == "txt_attn.qkv":
targets = ("attn.add_q_proj", "attn.add_k_proj",
"attn.add_v_proj")
else:
targets = (double_map[within],)
prefix = f"transformer_blocks.{block}"
elif parts[0] == "single_blocks":
targets = (single_map[within],)
prefix = f"single_transformer_blocks.{block}"
else:
raise ValueError(f"unexpected FP8 scale key: {key}")
for target in targets:
scales.setdefault(f"{prefix}.{target}", {})[scale_name] = value.clone()
def convert(checkpoint, **kwargs):
scales.clear()
for key in list(checkpoint):
if key.endswith((".input_scale", ".weight_scale")):
record(key, checkpoint.pop(key))
return original(checkpoint=checkpoint, **kwargs)
single_file_model.SINGLE_FILE_LOADABLE_CLASSES[
"Flux2Transformer2DModel"]["checkpoint_mapping_fn"] = convert
return scales
def _fp8_forward(torch, module, inputs):
shape = inputs.shape
input_fp8 = ((inputs / module._fp8_input_scale)
.clamp(torch.finfo(torch.float8_e4m3fn).min,
torch.finfo(torch.float8_e4m3fn).max)
.to(torch.float8_e4m3fn).reshape(-1, shape[-1]))
output = torch._scaled_mm(
input_fp8,
module.weight.reshape(-1, module.weight.shape[-1]).t(),
scale_a=module._fp8_input_scale,
scale_b=module._fp8_weight_scale,
bias=module.bias,
out_dtype=inputs.dtype,
use_fast_accum=True,
)
return output.reshape(*shape[:-1], output.shape[-1])
def generate(data: dict) -> dict:
global ACTIVE
import torch
from diffusers import (Flux2KleinPipeline, Flux2Transformer2DModel,
NVIDIAModelOptConfig)
from modelopt.torch.opt import enable_huggingface_checkpointing
from modelopt.torch.quantization.config import FP8_DEFAULT_CFG
from PIL import Image
from transformers import BitsAndBytesConfig, Qwen3ForCausalLM
prompt, filename = data.get("prompt"), data.get("filename")
if not isinstance(prompt, str) or not prompt.strip() or len(prompt) > 8000:
raise ValueError("invalid prompt")
if (not isinstance(filename, str) or Path(filename).name != filename
or not filename.endswith(".png")):
raise ValueError("invalid filename")
width, height = int(data.get("width", 1024)), int(data.get("height", 1024))
if (width, height) != (1024, 1024):
raise ValueError("FLUX 9B beta currently supports only 1024x1024")
if int(data.get("steps", 4)) != 4 or float(data.get("guidance", 1.0)) != 1.0:
raise ValueError("FLUX 9B beta requires steps=4 and guidance=1.0")
source_files = data.get("source_files") or []
if not isinstance(source_files, list) or len(source_files) > 4:
raise ValueError("invalid source image list")
source_images = []
for name in source_files:
if not isinstance(name, str) or Path(name).name != name:
raise ValueError("invalid source image filename")
source = (OUTPUT_DIR / name).resolve()
if source.parent != OUTPUT_DIR or not source.is_file():
raise ValueError("source image not found")
with Image.open(source) as opened:
source_images.append(opened.convert("RGB"))
started = time.monotonic()
ACTIVE = True
transformer = text_encoder = pipe = latent = decoded = image = None
try:
enable_huggingface_checkpointing()
scales = _install_fp8_converter()
tx_index, enc_index, tx_device, enc_device = _devices(torch)
quantization = NVIDIAModelOptConfig(
quant_type="FP8", weight_only=False,
modelopt_config=FP8_DEFAULT_CFG)
transformer = Flux2Transformer2DModel.from_single_file(
TRANSFORMER_FILE, config=COMPONENT_DIR, subfolder="transformer",
quantization_config=quantization, torch_dtype=torch.bfloat16,
device_map={"": tx_index}, local_files_only=True)
patched = 0
for module_name, module in transformer.named_modules():
if module_name not in scales:
continue
module.register_buffer("_fp8_input_scale",
scales[module_name]["input_scale"])
module.register_buffer("_fp8_weight_scale",
scales[module_name]["weight_scale"])
module.forward = MethodType(
lambda self, inputs: _fp8_forward(torch, self, inputs), module)
patched += 1
if patched != len(scales):
raise RuntimeError(f"patched only {patched} of {len(scales)} FP8 layers")
transformer.to(tx_device)
text_encoder = Qwen3ForCausalLM.from_pretrained(
os.path.join(COMPONENT_DIR, "text_encoder"),
torch_dtype=torch.bfloat16, low_cpu_mem_usage=True,
quantization_config=BitsAndBytesConfig(
load_in_4bit=True, bnb_4bit_quant_type="nf4",
bnb_4bit_compute_dtype=torch.bfloat16,
bnb_4bit_use_double_quant=True),
device_map={"": enc_index}, local_files_only=True)
pipe = Flux2KleinPipeline.from_pretrained(
COMPONENT_DIR, transformer=transformer, text_encoder=text_encoder,
torch_dtype=torch.bfloat16, local_files_only=True)
pipe.vae.enable_slicing()
pipe.vae.enable_tiling()
pipe.vae.to(tx_device)
loaded = time.monotonic() - started
prompt_embeds, _ = pipe.encode_prompt(
prompt.strip(), device=enc_device, max_sequence_length=128)
prompt_embeds = prompt_embeds.to(tx_device)
pipe.text_encoder = None
seed = data.get("seed")
generator = None if seed is None else torch.Generator(
device=tx_device).manual_seed(int(seed))
kwargs = {
"prompt": None, "prompt_embeds": prompt_embeds,
"height": height, "width": width, "num_inference_steps": 4,
"guidance_scale": 1.0, "generator": generator,
"output_type": "latent",
}
if source_images:
kwargs["image"] = (source_images[0] if len(source_images) == 1
else source_images)
latent = pipe(**kwargs).images
pipe.transformer = None
del transformer, text_encoder, prompt_embeds, generator
transformer = text_encoder = None
gc.collect()
torch.cuda.empty_cache()
latent = latent.to(device=tx_device, dtype=pipe.vae.dtype)
decoded = pipe.vae.decode(latent, return_dict=False)[0]
image = pipe.image_processor.postprocess(
decoded.detach(), output_type="pil")[0]
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
image.save(OUTPUT_DIR / filename)
return {"status": "ok", "filename": filename,
"seconds": round(time.monotonic() - started, 3),
"load_seconds": round(loaded, 3),
"model": "FLUX.2-klein-9B-fp8-beta"}
finally:
for value in (image, decoded, latent, pipe, text_encoder, transformer):
if value is not None:
del value
gc.collect()
torch.cuda.empty_cache()
ACTIVE = False
class Handler(BaseHTTPRequestHandler):
def log_message(self, fmt: str, *args: object) -> None:
print(f"[flux9b-beta] {self.client_address[0]} {fmt % args}", flush=True)
def reply(self, status: int, payload: dict) -> None:
body = json.dumps(payload, separators=(",", ":")).encode()
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def do_GET(self) -> None: # noqa: N802
if self.path == "/health":
self.reply(200, {"status": "ok", "model_loaded": ACTIVE,
"model": "FLUX.2-klein-9B-fp8-beta"})
else:
self.reply(404, {"error": "not found"})
def do_POST(self) -> None: # noqa: N802
if self.headers.get("Authorization", "") != f"Bearer {TOKEN}":
self.reply(401, {"error": "unauthorized"})
return
if self.path != "/generate":
self.reply(404, {"error": "not found"})
return
try:
length = int(self.headers.get("Content-Length", "0"))
if length < 2 or length > 16384:
raise ValueError("invalid request size")
self.reply(200, generate(json.loads(self.rfile.read(length))))
except Exception as exc:
print(f"[flux9b-beta] generation failed: {type(exc).__name__}: "
f"{str(exc)[:1000]}", flush=True)
self.reply(500, {"status": "error", "message": str(exc)})
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
-24
View File
@@ -1,24 +0,0 @@
FROM python:3.12-slim-bookworm
ARG PIPER_TTS_VERSION=1.6.0
RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates curl ffmpeg gosu \
&& python -m pip install --no-cache-dir "piper-tts==${PIPER_TTS_VERSION}" \
&& useradd --system --uid 10003 --home-dir /nonexistent --shell /usr/sbin/nologin piper \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
COPY piper_worker.py /app/piper_worker.py
COPY entrypoint.sh /usr/local/bin/mike-ai-piper-entrypoint
RUN chmod 0755 /usr/local/bin/mike-ai-piper-entrypoint
ENV PIPER_DATA_DIR=/data \
PIPER_VOICE=de_DE-thorsten-high \
PIPER_VOICE_ALIAS=alloy \
PIPER_HOST=0.0.0.0 \
PIPER_PORT=8085
VOLUME ["/data"]
EXPOSE 8085
ENTRYPOINT ["/usr/local/bin/mike-ai-piper-entrypoint"]
-15
View File
@@ -1,15 +0,0 @@
#!/bin/sh
set -eu
data_dir=${PIPER_DATA_DIR:-/data}
voice=${PIPER_VOICE:-de_DE-thorsten-high}
mkdir -p "$data_dir"
chown 10003:10003 "$data_dir"
if [ ! -s "$data_dir/$voice.onnx" ] || [ ! -s "$data_dir/$voice.onnx.json" ]; then
echo "Downloading Piper voice: $voice"
gosu piper python -m piper.download_voices --data-dir "$data_dir" "$voice"
fi
exec gosu piper python /app/piper_worker.py
-153
View File
@@ -1,153 +0,0 @@
#!/usr/bin/env python3
"""Small, private Piper worker for the Mike AI profile router.
The public OpenAI-compatible endpoint remains in the router. This worker only
accepts the narrow internal /status and /tts protocol and never logs input text.
"""
from __future__ import annotations
import io
import json
import os
import subprocess
import threading
import wave
from http import HTTPStatus
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from piper import PiperVoice, SynthesisConfig
DATA_DIR = Path(os.getenv("PIPER_DATA_DIR", "/data"))
VOICE_NAME = os.getenv("PIPER_VOICE", "de_DE-thorsten-high")
VOICE_ALIAS = os.getenv("PIPER_VOICE_ALIAS", "alloy")
HOST = os.getenv("PIPER_HOST", "0.0.0.0")
PORT = int(os.getenv("PIPER_PORT", "8085"))
MAX_TEXT_CHARS = int(os.getenv("PIPER_MAX_TEXT_CHARS", "8000"))
MAX_REQUEST_BYTES = int(os.getenv("PIPER_MAX_REQUEST_BYTES", "65536"))
VOICE_PATH = DATA_DIR / f"{VOICE_NAME}.onnx"
VOICE = PiperVoice.load(str(VOICE_PATH))
SYNTHESIS_LOCK = threading.Lock()
def synthesize_wav(text: str, speed: float) -> bytes:
"""Synthesize a complete WAV in memory without retaining the text."""
output = io.BytesIO()
config = SynthesisConfig(length_scale=1.0 / speed)
with SYNTHESIS_LOCK, wave.open(output, "wb") as wav_file:
VOICE.synthesize_wav(text, wav_file, syn_config=config)
return output.getvalue()
def wav_to_mp3(wav_bytes: bytes) -> bytes:
"""Convert Piper's WAV to the MP3 format Open WebUI requests by default."""
result = subprocess.run(
[
"ffmpeg", "-hide_banner", "-loglevel", "error",
"-f", "wav", "-i", "pipe:0",
"-codec:a", "libmp3lame", "-b:a", "96k",
"-f", "mp3", "pipe:1",
],
input=wav_bytes,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
timeout=120,
)
if result.returncode != 0:
raise RuntimeError("ffmpeg conversion failed")
return result.stdout
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def log_message(self, fmt: str, *args: object) -> None:
# Deliberately omit URLs and request bodies from the log.
print(f"piper-worker: {self.command} -> {args[1] if len(args) > 1 else '-'}")
def send_bytes(self, status: int, body: bytes, content_type: str) -> None:
self.send_response(status)
self.send_header("Content-Type", content_type)
self.send_header("Content-Length", str(len(body)))
self.send_header("Cache-Control", "no-store")
self.end_headers()
self.wfile.write(body)
def send_json(self, status: int, payload: dict) -> None:
self.send_bytes(
status,
json.dumps(payload, separators=(",", ":")).encode(),
"application/json",
)
def do_GET(self) -> None: # noqa: N802
if self.path != "/status":
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
self.send_json(
HTTPStatus.OK,
{
"ready": True,
"engine": "piper",
"model": VOICE_NAME,
"voices": [VOICE_ALIAS],
},
)
def do_POST(self) -> None: # noqa: N802
if self.path != "/tts":
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
try:
content_length = int(self.headers.get("Content-Length", "0"))
except ValueError:
content_length = 0
if content_length <= 0 or content_length > MAX_REQUEST_BYTES:
self.send_json(HTTPStatus.REQUEST_ENTITY_TOO_LARGE, {"error": "invalid request size"})
return
try:
request = json.loads(self.rfile.read(content_length))
text = request.get("text", "")
voice = request.get("voice", VOICE_ALIAS)
output_format = request.get("format", "mp3")
speed = float(request.get("speed", 1.0))
except (json.JSONDecodeError, TypeError, ValueError):
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid JSON request"})
return
if not isinstance(text, str) or not text.strip() or len(text) > MAX_TEXT_CHARS:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid text"})
return
if voice != VOICE_ALIAS:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
return
if output_format not in {"wav", "mp3"}:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
return
if not 0.5 <= speed <= 2.0:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
return
try:
audio = synthesize_wav(text.strip(), speed)
if output_format == "mp3":
audio = wav_to_mp3(audio)
content_type = "audio/mpeg"
else:
content_type = "audio/wav"
except (OSError, RuntimeError, subprocess.SubprocessError):
self.send_json(HTTPStatus.INTERNAL_SERVER_ERROR, {"error": "synthesis failed"})
return
self.send_bytes(HTTPStatus.OK, audio, content_type)
if __name__ == "__main__":
print(f"Piper worker ready: {VOICE_NAME} as {VOICE_ALIAS} on {HOST}:{PORT}")
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
+27 -13
View File
@@ -132,6 +132,25 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertIn("5 bis 6 Uhr", spoken)
self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken)
def test_qwen_speaks_aspect_ratios_as_ratios(self):
spoken = gateway.prepare_for_qwen_speech(
"Cover sind im Hochformat (2:3), Screenshots im Querformat "
"(16:9), ein Quadrat im Seitenverhältnis 2:2 und 4:3-Format."
)
self.assertIn("Hochformat (2 zu 3)", spoken)
self.assertIn("Querformat (16 zu 9)", spoken)
self.assertIn("Seitenverhältnis 2 zu 2", spoken)
self.assertIn("4 zu 3-Format", spoken)
def test_qwen_keeps_clock_times_distinct_from_aspect_ratios(self):
spoken = gateway.prepare_for_qwen_speech(
"Beginn um 16:09 Uhr, Fehler um 02:14; das Videoformat ist 16:9."
)
self.assertIn("16 Uhr 9", spoken)
self.assertNotIn("16 Uhr 9 Uhr", spoken)
self.assertIn("2 Uhr 14", spoken)
self.assertIn("Videoformat ist 16 zu 9", spoken)
def test_qwen_speaks_strict_date_ranges_as_calendar_dates(self):
spoken = gateway.prepare_for_qwen_speech(
"Neuigkeiten vom 04.–05.09. und Vergleich 04.09.–06.10.2026."
@@ -225,25 +244,20 @@ class LanguageSegmentationTests(unittest.TestCase):
self.assertTrue(all(len(part.split()) > 4 for _, part in segments))
class FallbackTests(unittest.TestCase):
class BackendFailureTests(unittest.TestCase):
def setUp(self):
self.original_xtts = gateway.synthesize_xtts
self.original_piper = gateway.synthesize_piper
self.original_qwen = gateway.synthesize_qwen
def tearDown(self):
gateway.synthesize_xtts = self.original_xtts
gateway.synthesize_piper = self.original_piper
gateway.synthesize_qwen = self.original_qwen
def test_piper_is_used_when_xtts_fails(self):
def test_qwen_failure_is_reported_without_fallback(self):
def fail(*_args):
raise RuntimeError("synthetic XTTS failure")
raise RuntimeError("synthetic Qwen failure")
gateway.synthesize_xtts = fail
gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav")
self.assertEqual(
gateway.synthesize("synthetic test", "wav", 1.0),
(b"piper", "audio/wav"),
)
gateway.synthesize_qwen = fail
with self.assertRaisesRegex(RuntimeError, "synthetic Qwen failure"):
gateway.synthesize("synthetic test", "wav", 1.0)
class AudioJoinTests(unittest.TestCase):
+166 -36
View File
@@ -1,5 +1,5 @@
#!/usr/bin/env python3
"""Private Qwen3-TTS-first gateway with a Piper fallback.
"""Private Qwen3-TTS gateway.
The gateway implements the narrow /status and /tts protocol already consumed
by the profile router. Request text is never logged or persisted.
@@ -8,6 +8,7 @@ by the profile router. Request text is never logged or persisted.
from __future__ import annotations
import io
import http.client
import json
import os
import re
@@ -17,6 +18,7 @@ import time
import unicodedata
import urllib.error
import urllib.request
import urllib.parse
import wave
from array import array
from http import HTTPStatus
@@ -31,7 +33,6 @@ QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
@@ -41,7 +42,6 @@ MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
# XTTS loses natural prosody when a sentence is synthesized as many tiny
# requests: every request starts a fresh utterance. Keep complete sentences
@@ -61,8 +61,7 @@ SPEAKER_LOCK = threading.Lock()
SPEAKER_CONDITIONING: dict | None = None
STATE = {
"last_backend": None,
"xtts_failures": 0,
"piper_fallbacks": 0,
"qwen_failures": 0,
"last_error": None,
}
@@ -267,6 +266,36 @@ def _spoken_ipv4(match: re.Match) -> str:
return " Punkt ".join(str(int(part)) for part in match.group(0).split("."))
def _normalize_aspect_ratios(text: str) -> str:
"""Speak colon notation as a ratio only when the surrounding text says so.
A bare ``16:09`` remains a clock time. This deliberately avoids a global
replacement of common ratios because ``16:9`` can also be a valid time.
"""
cue = (
r"(?:Seitenverh[aä]ltnis|Bildseitenverh[aä]ltnis|Bildformat|"
r"Videoformat|Hochformat|Querformat|Format|Aspect[- ]?Ratio)"
)
text = re.sub(
rf"\b({cue}\b(?:\s+(?:von|im|ist|betr[aä]gt))?\s*[\(\[]?\s*)"
rf"(\d{{1,3}})\s*:\s*(\d{{1,3}})",
lambda match: (
f"{match.group(1)}{int(match.group(2))} zu {int(match.group(3))}"
),
text,
flags=re.IGNORECASE,
)
return re.sub(
rf"\b(\d{{1,3}})\s*:\s*(\d{{1,3}})"
rf"(\s*[-‐‑‒–—−]?\s*{cue}\b)",
lambda match: (
f"{int(match.group(1))} zu {int(match.group(2))}{match.group(3)}"
),
text,
flags=re.IGNORECASE,
)
def normalize_for_german_speech(text: str) -> str:
"""Turn common visual notation into unambiguous spoken German."""
text = re.sub(r"\bv\.\s*a\.", "vor allem", text, flags=re.IGNORECASE)
@@ -279,10 +308,14 @@ def normalize_for_german_speech(text: str) -> str:
_spoken_ipv4,
text,
)
# A colon is ambiguous between an aspect ratio and a clock time. Resolve
# ratios first, but only when an explicit format cue is present.
text = _normalize_aspect_ratios(text)
text = re.sub(
r"\b([01]?\d|2[0-3]):([0-5]\d)\b",
r"\b([01]?\d|2[0-3]):([0-5]\d)\b(?:\s*Uhr\b)?",
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
text,
flags=re.IGNORECASE,
)
text = re.sub(
r"\b([01]?\d|2[0-3])\s*[-‐‑‒–—−]\s*"
@@ -702,18 +735,6 @@ def synthesize_xtts(text: str, output_format: str,
return _convert(_wav(_join_pcm(pcm_parts)), output_format, speed)
def synthesize_piper(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
upstream_format = "wav" if output_format == "pcm" else output_format
audio, content_type = _request(
f"{PIPER_URL}/tts",
payload={"text": text, "voice": "alloy", "speed": speed,
"format": upstream_format},
timeout=PIPER_TIMEOUT,
)
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
def synthesize_qwen(text: str, output_format: str,
speed: float) -> tuple[bytes, str]:
text = prepare_for_qwen_speech(text)
@@ -728,6 +749,47 @@ def synthesize_qwen(text: str, output_format: str,
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
def open_qwen_pcm_stream(text: str, chunk_size: int = 4) \
-> tuple[http.client.HTTPConnection, http.client.HTTPResponse]:
"""Open Qwen's native token-level PCM stream without buffering it.
The upstream emits headerless 24 kHz mono signed 16-bit little-endian
PCM. Keeping this response streaming is what lets playback begin while
the remainder of the sentence is still being synthesized.
"""
parsed = urllib.parse.urlparse(QWEN_TTS_URL)
if parsed.scheme != "http" or not parsed.hostname:
raise RuntimeError("QWEN_TTS_URL must be an http URL")
port = parsed.port or 80
prefix = parsed.path.rstrip("/")
payload = json.dumps({
"model": QWEN_TTS_MODEL,
"input": prepare_for_qwen_speech(text),
"voice": QWEN_TTS_VOICE,
"language": QWEN_TTS_LANGUAGE,
"chunk_size": chunk_size,
}, separators=(",", ":")).encode()
connection = http.client.HTTPConnection(
parsed.hostname, port, timeout=QWEN_TTS_TIMEOUT)
try:
connection.request(
"POST",
f"{prefix}/v1/audio/speech/pcm-stream",
body=payload,
headers={"Content-Type": "application/json",
"Accept": "application/octet-stream"},
)
response = connection.getresponse()
if response.status != HTTPStatus.OK:
message = response.read(512).decode(errors="replace")
raise RuntimeError(
f"Qwen PCM stream failed ({response.status}): {message}")
return connection, response
except Exception:
connection.close()
raise
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
if acquired:
@@ -737,22 +799,18 @@ def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]
STATE["last_backend"] = "qwen3-tts-1.7b"
STATE["last_error"] = None
return audio
except Exception as exc: # fallback must cover all Qwen failures
except Exception as exc:
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["qwen_failures"] += 1
STATE["last_error"] = type(exc).__name__
raise
finally:
SYNTHESIS_LOCK.release()
else:
with STATE_LOCK:
STATE["xtts_failures"] += 1
STATE["qwen_failures"] += 1
STATE["last_error"] = "queue-timeout"
audio = synthesize_piper(text, output_format, speed)
with STATE_LOCK:
STATE["last_backend"] = "piper"
STATE["piper_fallbacks"] += 1
return audio
raise RuntimeError("speech queue timeout")
class Handler(BaseHTTPRequestHandler):
@@ -779,25 +837,26 @@ class Handler(BaseHTTPRequestHandler):
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
primary_ready = _reachable(QWEN_TTS_URL, "/health")
fallback_ready = _reachable(PIPER_URL, "/status")
with STATE_LOCK:
state = dict(STATE)
# This endpoint is also the container liveness check. Qwen3-TTS is
# deliberately stopped in exclusive GPU modes such as Applio, so the
# gateway itself must stay healthy while reporting ready=false.
self.send_json(
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
HTTPStatus.OK,
{
"ready": fallback_ready,
"engine": "qwen3-tts-with-piper-fallback",
"ready": primary_ready,
"engine": "qwen3-tts",
"model": "Qwen3-TTS-12Hz-1.7B-Base",
"voices": [VOICE_ALIAS],
"speaker": QWEN_TTS_VOICE,
"primary_ready": primary_ready,
"fallback_ready": fallback_ready,
**state,
},
)
def do_POST(self) -> None: # noqa: N802
if self.path != "/tts":
if self.path not in {"/tts", "/tts/pcm-stream"}:
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
return
try:
@@ -810,7 +869,7 @@ class Handler(BaseHTTPRequestHandler):
return
try:
request = json.loads(self.rfile.read(length))
text = request.get("text", "")
text = request.get("input", request.get("text", ""))
voice = request.get("voice", VOICE_ALIAS)
output_format = request.get("format", "mp3")
speed = float(request.get("speed", 1.0))
@@ -829,6 +888,9 @@ class Handler(BaseHTTPRequestHandler):
if not 0.5 <= speed <= 2.0:
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
return
if self.path == "/tts/pcm-stream":
self._stream_qwen_pcm(text.strip(), request)
return
started = time.monotonic()
try:
audio, content_type = synthesize(text.strip(), output_format, speed)
@@ -836,13 +898,81 @@ class Handler(BaseHTTPRequestHandler):
with STATE_LOCK:
STATE["last_error"] = type(exc).__name__
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
{"error": "all local speech backends failed"})
{"error": "local Qwen3-TTS backend failed"})
return
print(f"tts-gateway: synthesized via {STATE['last_backend']} in "
f"{time.monotonic() - started:.2f}s")
self.send_bytes(HTTPStatus.OK, audio, content_type)
def _stream_qwen_pcm(self, text: str, request: dict) -> None:
"""Unframe Qwen's PCM frames and relay their audio immediately."""
try:
chunk_size = max(1, min(32, int(request.get("chunk_size", 4))))
except (TypeError, ValueError):
self.send_json(HTTPStatus.BAD_REQUEST,
{"error": "invalid chunk_size"})
return
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
if not acquired:
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
{"error": "speech queue timeout"})
return
connection = None
started = time.monotonic()
headers_sent = False
try:
connection, response = open_qwen_pcm_stream(text, chunk_size)
self.send_response(HTTPStatus.OK)
self.send_header("Content-Type", "application/octet-stream")
self.send_header("Cache-Control", "no-store")
self.send_header("Connection", "close")
self.end_headers()
headers_sent = True
first = True
while True:
frame_header = response.read(4)
if not frame_header:
break
if len(frame_header) != 4:
raise RuntimeError("truncated Qwen PCM frame header")
frame_length = int.from_bytes(frame_header, "big")
if frame_length == 0:
break
if frame_length > MAX_AUDIO_BYTES:
raise RuntimeError("Qwen PCM frame is too large")
remaining = frame_length
while remaining:
chunk = response.read(min(16384, remaining))
if not chunk:
raise RuntimeError("truncated Qwen PCM frame")
if first:
print("tts-gateway: first Qwen PCM chunk in "
f"{time.monotonic() - started:.2f}s")
first = False
self.wfile.write(chunk)
self.wfile.flush()
remaining -= len(chunk)
with STATE_LOCK:
STATE["last_backend"] = "qwen3-tts-1.7b-stream"
STATE["last_error"] = None
except Exception as exc:
with STATE_LOCK:
STATE["last_error"] = type(exc).__name__
# Once PCM started, simply close the truncated response. Sending
# JSON into the audio stream would produce loud corrupt samples.
if not headers_sent and not self.wfile.closed:
try:
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
{"error": "local PCM stream failed"})
except (OSError, BrokenPipeError):
pass
finally:
if connection is not None:
connection.close()
SYNTHESIS_LOCK.release()
self.close_connection = True
if __name__ == "__main__":
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
print(f"TTS gateway ready on {HOST}:{PORT}; backend=Qwen3-TTS; voice={QWEN_TTS_VOICE}")
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
-70
View File
@@ -1,70 +0,0 @@
#!/bin/sh
set -eu
target=/opt/hermes/cron/scheduler_delivery.py
broken='env.pop("HERMES_HOME", None)'
if [ -f "$target" ] && grep -Fq "$broken" "$target"; then
sed -i '/^[[:space:]]*env\.pop("HERMES_HOME", None)[[:space:]]*$/d' "$target"
echo "[cron-profile-root-fix] preserved HERMES_HOME for named-profile delivery"
else
echo "[cron-profile-root-fix] upstream code already fixed or layout changed; no action"
fi
cli_dir="${HERMES_HOME:-/opt/data}/.local/bin"
mkdir -p "$cli_dir"
ln -sfn /opt/hermes/.venv/bin/hermes "$cli_dir/hermes"
ln -sfn /opt/hermes/.venv/bin/hermes /usr/local/bin/hermes
# Hermes' generic sentence splitter treats periods in German abbreviations as
# sentence ends. It also submits every sentence as a separate generative TTS
# request, which creates long gaps with local Qwen3-TTS. Keep the first sentence
# immediate, then group following complete sentences into modest ~240-character
# chunks so playback remains responsive but substantially more continuous.
tts_target=/opt/hermes/tools/tts_streaming.py
if [ -f "$tts_target" ] && grep -Fq 'self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)' "$tts_target"; then
TTS_TARGET="$tts_target" python3 <<'PY'
import os
from pathlib import Path
p = Path(os.environ["TTS_TARGET"])
s = p.read_text()
base_feed = ' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)'
abbr_lines = [
' self.buf = re.sub(r"\\bca\\.(?=\\s)", "circa", self.buf, flags=re.IGNORECASE)',
' self.buf = re.sub(r"\\bv\\.\\s*a\\.(?=\\s)", "vor allem", self.buf, flags=re.IGNORECASE)',
' self.buf = re.sub(r"\\bmax\\.(?=\\s)", "maximal", self.buf, flags=re.IGNORECASE)',
]
for line in abbr_lines:
s = s.replace("\n" + line, "")
init_old = ' self.buf = ""'
init_extra = (
'\n self.followup_target_len = 240'
'\n self._emitted_first = False'
)
s = s.replace(init_old + init_extra, init_old)
threshold_old = ' if len(head.strip()) < self.min_len:'
threshold_new = (
' threshold = (self.min_len if not self._emitted_first '
'else self.followup_target_len)\n'
' if len(head.strip()) < threshold:'
)
s = s.replace(threshold_new, threshold_old)
append_old = ' out.append(head)'
append_new = append_old + '\n self._emitted_first = True'
s = s.replace(append_new, append_old)
s = s.replace(base_feed, base_feed + "\n" + "\n".join(abbr_lines), 1)
s = s.replace(init_old, init_old + init_extra, 1)
s = s.replace(threshold_old, threshold_new, 1)
s = s.replace(append_old, append_new, 1)
p.write_text(s)
PY
echo "[tts-streaming-fix] enabled German normalization and adaptive sentence grouping"
else
echo "[tts-streaming-fix] upstream code already fixed or layout changed; no action"
fi
+2 -3
View File
@@ -144,13 +144,12 @@ stt:
model: "base"
language: "de"
# Reuse Athena's OpenAI-compatible TTS route. It currently serves XTTS v2 with
# Annmarie Nele and transparently falls back to Piper when XTTS is unavailable.
# Reuse Athena's OpenAI-compatible Qwen3-TTS route.
tts:
provider: "openai"
speed: 1.0
openai:
model: "piper"
model: "qwen3-tts"
voice: "alloy"
speed: 1.0
base_url: "http://router:8081/v1"
@@ -1,120 +0,0 @@
---
name: athena-ai-profile-router
description: Use when operating the AI profile router on Athena.
metadata:
hermes:
editorial_name: "Athena AI-Profile-Router"
editorial_description: "Betrieb und Abfragen der lokalen KI-Plattform auf Athena: Profile, Status, Umschaltung."
---
# Athena AI-Profile-Router
Lokale KI-Plattform auf **Athena** (root@192.168.1.212, Debian 13, GPU-Host).
Repo: `/root/AI-Profile-Router` (Gitea: `michael/AI-Profile-Router`).
SSH: `ssh -i /opt/data/athena_key -o UserKnownHostsFile=/opt/data/.ssh/known_hosts -o IdentitiesOnly=yes root@192.168.1.212`
## Wichtigste Regel
**Der Host steht physisch in einer anderen Stadt. Es gibt keinen schnellen
Zugang.**
- **NIEMALS** herunterfahren (`shutdown`, `poweroff`, `halt`) oder neu
starten (`reboot`, `init 6`).
- Keine Aktion, die die Erreichbarkeit gefährdet: keine Änderungen an `lan0`,
Firewall-Default-Policies, `DOCKER-USER`-Regeln oder Routing ohne explizite
Freigabe und Rückfallplan; keine Reboot-Pflicht auslösenden Aktionen
(Kernel-, NVIDIA-Treiber-Installation, `apt full-upgrade` mit Kernel) ohne
ausdrückliche Freigabe; keine Abschaltung von WireGuard, SSH oder dem
Gateway-Container.
- Installer-Exit-Code 20 (NVIDIA-Treiber, Reboot nötig) wird nicht automatisch
nachgeholt — der Betreiber entscheidet.
- Nach jeder Netz-/Firewall-/WG-Änderung verifizieren: (1) SSH-Roundtrip,
(2) WG-Tunnel up, (3) Router `/health` und `/ready` = 200.
## Architektur (Kurzform)
- **Profile Router** (`mike-ai-router`, OpenAI-kompatible API, Port 8081):
einzige Client-Schnittstelle. Clients nutzen `http://<host>:8081/v1` mit
Bearer-Key aus `/etc/mike-ai/router-api-key`.
- **llama.cpp** (`mike-ai-llama-*`): ein Container pro Profil, immer exakt
einer aktiv; nur Docker-intern (Port 8080, nie direkt von Clients).
- **Profile Controller** (`mike-ai-profile-controller`): einziger Dienst mit
Docker-Socket; begrenzter Containerwechsel.
- **Open WebUI** (`mike-ai-open-webui`, Port 8080): einzige normale Oberfläche.
- **MCP-Tool-Stack** (`platform/mcp/`): getrennte Container für Web, HA, ARR,
Unraid — nur über internes `mike-ai-tools`-Netz.
- WireGuard-Isolation: KI-Dienste nur über WG-Adresse erreichbar,
fail-closed bei Tunnelausfall (Blackhole-Default-Route).
## Profile (Live-Stand 11.09.2026)
| Profil | Virtuelles Modell | Kontext | Zweck |
|---|---|---:|---|
| fast | `qwen-fast` | 76.800 | Alltag, Agenten, hohe Geschwindigkeit |
| medium | `qwen-medium` | 160.000 | mehr Kontext |
| large | `qwen-large` | 192.000 | lange Sitzungen |
| ultra | `qwen-ultra` | 262.144 | maximale Kontextlänge |
| uncensored | `qwen-uncensored` | 80.000 | unzensiert |
Aktives Profil: `GET /status` → `current_profile`.
Profilwechsel: `POST /fast`, `/medium`, `/large`, `/ultra`, `/uncensored`
(authentifiziert) oder manuell `llama-profile <name>` auf dem Host.
## Wichtige Befehle (per SSH auf Athena)
```bash
# Status (Router lebt, Profil, Upstream, Telemetrie)
KEY=$(cat /etc/mike-ai/router-api-key)
curl -s -H "Authorization: Bearer ***" http://172.30.20.3:8081/status
# Health / Ready (ohne Auth)
curl -s -o /dev/null -w '%{http_code}' http://172.30.20.3:8081/health # 200 = Router lebt
curl -s -o /dev/null -w '%{http_code}' http://172.30.20.3:8081/ready # 200 = Modell bereit
# Virtuelle Modelle
curl -s -H "Authorization: Bearer ***" http://172.30.20.3:8081/v1/models
# Profilwechsel
curl -s -X POST -H "Authorization: Bearer ***" http://172.30.20.3:8081/ultra
# Aktive Profile-Registry (Router-Container)
docker exec mike-ai-router cat /app/router_profiles.json
# Container-Status
docker ps --format '{{.Names}}\t{{.Status}}' | grep mike-ai
```
Router-IP auf dem internen `mike-ai_control`-Netz: `172.30.20.3`
(neu ermitteln: `docker inspect mike-ai-router --format '{{.NetworkSettings.Networks.mike-ai_control.IPAddress}}'`).
## Fehler- und Recovery-Verhalten
- `/health=200` + `/ready=503` → Router lebt, Modell lädt/wechselt noch.
- `429` → Parallelitätsgrenze (max. 16) erreicht; Client mit Backoff.
- Profilwechsel bricht ab, wenn laufende Chats den Drain-Timeout (600 s)
überschreiten — er beendet niemals absichtlich einen Chat.
- Nach Routerabsturz: letztes stabiles Profil aus atomarer Zustandsdatei
(`/var/lib/mike-ai-profile-router/state.json`) rekonstruiert.
- Upgrade-Regel: niemals Build, Quantisierung und Profil gleichzeitig ändern.
## Git / Repo
- Remote: `git@192.168.1.2:michael/AI-Profile-Router.git` (Gitea auf Unraid).
- Athena erreicht das Heimnetz **nur über WireGuard** — ist der WG-Tunnel
down, schlägt `git push` mit Connection timed out fehl. Nicht mit
Firewall-/Routing-Änderungen gegensteuern; Tunnel-Status prüfen und
melden. Commits bleiben lokal auf Athena, bis der Tunnel wieder up ist.
- Git-Identität auf Athena ist repo-lokal gesetzt (Mikei386).
## Sicherheitsregeln
- API-Key (`/etc/mike-ai/router-api-key`) nie in Chat, Repo oder URLs.
- Clients nie direkt Port 8080 (llama.cpp) — immer über Router 8081.
- `config/install.env` bleibt lokal (0600), wird nicht committet.
- Systempartition dauerhaft unter 85 % halten; nur ein Textmodell gleichzeitig.
## Backup
Gesichert: Repo, Modellmanifest (Hashes, ohne Secrets), `/etc/mike-ai`
verschlüsselt, systemd-Konfiguration, Benchmarkresultate.
Nicht nötig: Builds, Venvs, Caches, Modelle (Quelle + Prüfsumme dokumentiert).
@@ -0,0 +1,47 @@
# Athena architecture and modes
Use `ATHENA.md` as the short operational truth and
`docs/CONTAINER_INVENTORY.md` for the current mapping of container, model and
role. If live state disagrees with documentation, report the discrepancy and
correct the durable source when the user requested maintenance.
## Boundaries
- Hermes, chats, skills and portable specialist MCPs live on Unraid.
- Athena is the inference host. Its canonical checkout is
`/opt/mike-ai/stack`; models live under `/data/models`; local secrets and
runtime configuration live under `/etc/mike-ai` and never enter Git.
- The Profile Router is the single OpenAI-compatible address clients use.
- The Profile Controller is the only component allowed to orchestrate approved
model and specialist workers.
- The Athena Operator is host-bound and is the normal maintenance interface.
## Exclusive states
Athena has five mutually exclusive persistent modes: `llm`, `music`,
`separation`, `voice`, and `voicechange`. Image generation is a transactional
request: it temporarily pauses the active text profile and Qwen3-TTS, runs the
image worker, then restores the previous LLM state.
Only one heavy GPU path may be active. Do not manually start a second GPU
worker around the controller. The lightweight dashboard, router, controller,
gateway, UI, CPU-STT, backup and operator containers may remain active.
## Model profiles
- Fast: Qwen3.8-27B IQ4-MIX, 76,800 tokens.
- Medium: Qwen3.8-27B IQ4_XS-pure, 160,000 tokens, vision.
- Large: the same Q4 model, 192,000 tokens, vision.
- Ultra: the same Q4 model, 262,144 tokens, no vision projector.
- Uncensored: Abliterated Q4_K_M, 80,000 tokens, vision.
Medium, Large, Ultra and Beta distribute their runtime across both GPUs. Do not
infer allocation from model size alone; verify the profile's Compose arguments
and live VRAM. RTX 3060 also hosts Qwen3-TTS during normal LLM operation.
## What is and is not stale
The GPU workers use `restart: "no"` and are created once, then started on
demand. A stopped `mike-ai-llama-*`, image, music, separator, OmniVoice or X-VC
container is expected. A candidate is stale only after checking Compose,
labels, mounts, router/controller references, model paths and test history.
@@ -163,10 +163,10 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
cd "$STACK_DIR"
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
qwen3-tts piper tts-gateway router open-webui
qwen3-tts tts-gateway router open-webui
deadline=$((SECONDS + 180))
for container in mike-ai-qwen3-tts mike-ai-piper mike-ai-tts-gateway \
for container in mike-ai-qwen3-tts mike-ai-tts-gateway \
mike-ai-router mike-ai-open-webui; do
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
"$container" 2>/dev/null || true) == healthy ]]; do
+1 -1
View File
@@ -83,7 +83,7 @@ with con:
for key, value in (
("task.follow_up.enable", False),
("audio.tts.engine", "openai"),
("audio.tts.model", "piper"),
("audio.tts.model", "qwen3-tts"),
("audio.tts.voice", "alloy"),
("audio.tts.openai.api_base_url", "http://router:8081/v1"),
("web.search.enable", True),
+36
View File
@@ -0,0 +1,36 @@
#!/usr/bin/env bash
# Rebuild/create specialist containers after a bare-metal restore. GPU workers
# remain stopped; the core installer leaves Athena safely in LLM mode.
set -Eeuo pipefail
log() { printf '\n==> %s\n' "$*"; }
if [[ -r /etc/mike-ai/install.env ]]; then
set -a
# shellcheck disable=SC1091
source /etc/mike-ai/install.env
set +a
fi
export VOICE_GPU_UUID=${VOICE_GPU_UUID:-${IMAGE_GPU_DEVICES:-}}
export ACESTEP_GPU_UUID=${ACESTEP_GPU_UUID:-${IMAGE_GPU_DEVICES:-}}
export SEPARATOR_GPU_UUID=${SEPARATOR_GPU_UUID:-${IMAGE_GPU_DEVICES:-}}
create_project() {
local dir=$1 file=${2:-compose.yaml} profile=${3:-}
[[ -f $dir/$file ]] || { printf 'Übersprungen (fehlt): %s/%s\n' "$dir" "$file"; return 0; }
log "Spezialprojekt vorbereiten: $dir"
local args=(docker compose -f "$file")
[[ -z $profile ]] || args+=(--profile "$profile")
(cd "$dir" && "${args[@]}" pull --ignore-buildable && \
"${args[@]}" build && "${args[@]}" create)
}
docker network inspect mike-ai_frontend >/dev/null
create_project /opt/mike-ai/acestep-test compose.yaml music-test
create_project /opt/mike-ai/stem-separator
create_project /opt/mike-ai/omnivoice-studio
create_project /opt/mike-ai/xvc-studio
create_project /opt/mike-ai/stack/experiments/applio-rvc
create_project /opt/mike-ai/Mikes-Applio-UI compose.example.yaml
printf 'ATHENA_SPECIALISTS_REBUILT_OK\n'
+2 -2
View File
@@ -7,9 +7,9 @@ SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
PROFILE="${1:-}"
case "$PROFILE" in
fast|medium|beta1|large|ultra|uncensored) ;;
fast|medium|large|ultra|uncensored) ;;
*)
echo "Usage: llama-profile {fast|medium|beta1|large|ultra|uncensored}" >&2
echo "Usage: llama-profile {fast|medium|large|ultra|uncensored}" >&2
exit 2
;;
esac