Synchronize repository with Athena deployment
This commit is contained in:
Executable
+78
@@ -0,0 +1,78 @@
|
||||
#!/usr/bin/env bash
|
||||
# Encrypted off-host backup for data-disk and total-loss recovery.
|
||||
set -Eeuo pipefail
|
||||
umask 077
|
||||
|
||||
CONFIG=${DISASTER_BACKUP_CONFIG:-/etc/mike-ai/disaster-backup.env}
|
||||
STATE=/var/lib/mike-ai-disaster-backup
|
||||
|
||||
log() { printf '\n==> %s\n' "$*"; }
|
||||
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
|
||||
[[ -r $CONFIG ]] || die "Konfiguration fehlt: $CONFIG"
|
||||
# shellcheck disable=SC1090
|
||||
source "$CONFIG"
|
||||
[[ ${DISASTER_BACKUP_ENABLED:-false} == true ]] || die \
|
||||
"Externes Backup ist noch nicht freigeschaltet (DISASTER_BACKUP_ENABLED=true)."
|
||||
[[ -n ${RESTIC_REPOSITORY:-} ]] || die "RESTIC_REPOSITORY fehlt."
|
||||
if [[ -n ${RESTIC_REQUIRE_MOUNT:-} ]]; then
|
||||
mountpoint -q "$RESTIC_REQUIRE_MOUNT" || die \
|
||||
"Externes Backupziel ist nicht eingehängt: $RESTIC_REQUIRE_MOUNT"
|
||||
fi
|
||||
[[ -n ${RESTIC_PASSWORD_FILE:-} && -r $RESTIC_PASSWORD_FILE ]] || die \
|
||||
"RESTIC_PASSWORD_FILE fehlt oder ist nicht lesbar."
|
||||
command -v restic >/dev/null || die "restic ist nicht installiert."
|
||||
command -v docker >/dev/null || die "Docker ist nicht installiert."
|
||||
exec 9>/run/lock/athena-disaster-backup.lock
|
||||
flock -n 9 || die "Ein Disaster-Backup läuft bereits."
|
||||
|
||||
install -d -m 0700 "$STATE/latest"
|
||||
|
||||
log "Konsistentes Docker-Schnellbackup erzeugen"
|
||||
docker inspect mike-ai-backup >/dev/null 2>&1 || die "mike-ai-backup fehlt."
|
||||
docker exec mike-ai-backup backup
|
||||
latest=$(readlink -f /data/docker-backups/athena-latest.tar.gz)
|
||||
[[ -s $latest ]] || die "Lokales Docker-Backup wurde nicht erzeugt."
|
||||
gzip -t "$latest" || die "Lokales Docker-Backup ist beschädigt."
|
||||
install -m 0600 "$latest" "$STATE/latest/docker-state.tar.gz"
|
||||
sha256sum "$STATE/latest/docker-state.tar.gz" >"$STATE/latest/docker-state.tar.gz.sha256"
|
||||
|
||||
log "Wiederaufbau-Metadaten erfassen"
|
||||
{
|
||||
printf 'created_utc=%s\n' "$(date -u +%FT%TZ)"
|
||||
printf 'hostname=%s\n' "$(hostname)"
|
||||
printf 'source_commit=%s\n' "$(git -C /opt/mike-ai/stack rev-parse HEAD 2>/dev/null || printf unknown)"
|
||||
findmnt -rn -o SOURCE,UUID,FSTYPE,TARGET / /data 2>/dev/null || true
|
||||
} >"$STATE/latest/manifest.txt"
|
||||
find /data/models -type f -printf '%P\t%s\n' 2>/dev/null | sort \
|
||||
>"$STATE/latest/model-manifest.tsv"
|
||||
docker ps -a --format '{{.Names}}\t{{.Image}}\t{{.Status}}' \
|
||||
>"$STATE/latest/container-manifest.tsv"
|
||||
|
||||
paths=(/etc/mike-ai /opt/mike-ai "$STATE/latest")
|
||||
for path in \
|
||||
/data/voice /data/music /data/audio /data/llama-dashboard \
|
||||
/data/mike-ai-operator /data/benchmarks /data/model-benchmarks \
|
||||
/data/image-comparison /data/backups /data/deploy-backups; do
|
||||
[[ ! -e $path ]] || paths+=("$path")
|
||||
done
|
||||
|
||||
tag=${RESTIC_TAG:-athena-disaster}
|
||||
log "Verschlüsseltes externes Backup schreiben"
|
||||
if ! restic snapshots >/dev/null 2>&1; then
|
||||
log "Neues Restic-Repository initialisieren"
|
||||
restic init
|
||||
fi
|
||||
restic backup --tag "$tag" "${paths[@]}"
|
||||
|
||||
log "Aufbewahrung anwenden"
|
||||
restic forget --tag "$tag" \
|
||||
--keep-daily "${RESTIC_KEEP_DAILY:-14}" \
|
||||
--keep-weekly "${RESTIC_KEEP_WEEKLY:-8}" \
|
||||
--keep-monthly "${RESTIC_KEEP_MONTHLY:-12}" --prune
|
||||
|
||||
log "Letzten Snapshot verifizieren"
|
||||
restic snapshots --tag "$tag" --latest 1
|
||||
restic check
|
||||
printf 'ATHENA_DISASTER_BACKUP_OK\n'
|
||||
@@ -0,0 +1,12 @@
|
||||
[Unit]
|
||||
Description=Encrypted off-host disaster backup for Athena
|
||||
After=docker.service network-online.target
|
||||
Wants=network-online.target
|
||||
ConditionPathExists=/etc/mike-ai/disaster-backup.env
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/athena-disaster-backup
|
||||
Nice=10
|
||||
IOSchedulingClass=best-effort
|
||||
IOSchedulingPriority=7
|
||||
@@ -0,0 +1,11 @@
|
||||
[Unit]
|
||||
Description=Nightly Athena off-host disaster backup
|
||||
|
||||
[Timer]
|
||||
OnCalendar=*-*-* 03:15:00
|
||||
Persistent=true
|
||||
RandomizedDelaySec=30m
|
||||
Unit=athena-disaster-backup.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
Executable
+82
@@ -0,0 +1,82 @@
|
||||
#!/usr/bin/env bash
|
||||
# Build a browser-downloadable, encrypted archive of irreplaceable Athena data.
|
||||
set -Eeuo pipefail
|
||||
umask 077
|
||||
|
||||
OUTPUT_DIR=${ATHENA_EXPORT_DIR:-/data/emergency-backups}
|
||||
RECIPIENT_FILE=${ATHENA_AGE_RECIPIENT_FILE:-/etc/mike-ai/recovery.age-recipient}
|
||||
STATE=/var/lib/mike-ai-disaster-backup/latest
|
||||
KEEP=${ATHENA_EXPORT_KEEP:-5}
|
||||
|
||||
log() { printf '\n==> %s\n' "$*"; }
|
||||
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
|
||||
[[ -s $RECIPIENT_FILE ]] || die "Age-Empfänger fehlt: $RECIPIENT_FILE"
|
||||
[[ $KEEP =~ ^[1-9][0-9]*$ ]] || die "ATHENA_EXPORT_KEEP muss positiv sein."
|
||||
for command in age zstd tar docker sha256sum flock; do
|
||||
command -v "$command" >/dev/null || die "$command fehlt."
|
||||
done
|
||||
exec 9>/run/lock/athena-export-backup.lock
|
||||
flock -n 9 || die "Ein exportierbares Backup läuft bereits."
|
||||
|
||||
install -d -m 0755 "$OUTPUT_DIR"
|
||||
install -d -m 0700 "$STATE"
|
||||
|
||||
log "Aktuellen Docker-Zustand sichern"
|
||||
docker exec mike-ai-backup backup
|
||||
latest=$(readlink -f /data/docker-backups/athena-latest.tar.gz)
|
||||
[[ -s $latest ]] || die "Docker-Zustandsbackup fehlt."
|
||||
gzip -t "$latest" || die "Docker-Zustandsbackup ist beschädigt."
|
||||
install -m 0600 "$latest" "$STATE/docker-state.tar.gz"
|
||||
|
||||
stamp=$(date -u +%Y-%m-%dT%H-%M-%SZ)
|
||||
name="athena-portable-$stamp.tar.zst.age"
|
||||
partial="$OUTPUT_DIR/.$name.partial"
|
||||
target="$OUTPUT_DIR/$name"
|
||||
list=$(mktemp /tmp/athena-export-list.XXXXXX)
|
||||
trap 'rm -f "$list" "$partial"' EXIT
|
||||
|
||||
add_path() {
|
||||
local path=${1#/}
|
||||
[[ ! -e /$path ]] || printf '%s\0' "$path" >>"$list"
|
||||
}
|
||||
|
||||
# Reproducible model/HF caches are deliberately omitted. Everything below is
|
||||
# either a host configuration, project source, user input or generated result.
|
||||
add_path /etc/mike-ai
|
||||
add_path /opt/mike-ai
|
||||
add_path /var/lib/mike-ai-disaster-backup/latest
|
||||
add_path /data/voice/applio/logs
|
||||
add_path /data/voice/applio/datasets
|
||||
add_path /data/voice/applio/config.json
|
||||
add_path /data/voice/omnivoice/output
|
||||
add_path /data/voice/xvc/output
|
||||
add_path /data/voice/studio
|
||||
add_path /data/music
|
||||
add_path /data/audio
|
||||
add_path /data/llama-dashboard
|
||||
add_path /data/mike-ai-operator
|
||||
add_path /data/benchmarks
|
||||
add_path /data/model-benchmarks
|
||||
add_path /data/image-comparison
|
||||
add_path /data/backups
|
||||
add_path /data/deploy-backups
|
||||
|
||||
log "Portables, verschlüsseltes Backup erzeugen"
|
||||
tar --create --numeric-owner --acls --xattrs -C / --null --files-from="$list" \
|
||||
| zstd -T0 -3 \
|
||||
| age -R "$RECIPIENT_FILE" -o "$partial"
|
||||
chmod 0644 "$partial"
|
||||
mv "$partial" "$target"
|
||||
sha256sum "$target" >"$target.sha256"
|
||||
chmod 0644 "$target.sha256"
|
||||
|
||||
log "Nur die letzten $KEEP Generationen behalten"
|
||||
mapfile -t old < <(find "$OUTPUT_DIR" -maxdepth 1 -type f \
|
||||
-name 'athena-portable-*.tar.zst.age' -printf '%T@ %p\n' | sort -rn | tail -n +$((KEEP + 1)) | cut -d' ' -f2-)
|
||||
for archive in "${old[@]}"; do
|
||||
rm -f -- "$archive" "$archive.sha256"
|
||||
done
|
||||
|
||||
printf 'ATHENA_EXPORT_BACKUP_OK file=%s bytes=%s\n' "$target" "$(stat -c %s "$target")"
|
||||
@@ -0,0 +1,12 @@
|
||||
[Unit]
|
||||
Description=Create encrypted downloadable Athena recovery package
|
||||
After=docker.service
|
||||
Requires=docker.service
|
||||
ConditionPathExists=/etc/mike-ai/recovery.age-recipient
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/athena-export-backup
|
||||
Nice=10
|
||||
IOSchedulingClass=best-effort
|
||||
IOSchedulingPriority=7
|
||||
@@ -0,0 +1,12 @@
|
||||
[Unit]
|
||||
Description=Create an Athena recovery package every five hours
|
||||
|
||||
[Timer]
|
||||
OnBootSec=45m
|
||||
OnUnitActiveSec=5h
|
||||
Persistent=true
|
||||
RandomizedDelaySec=10m
|
||||
Unit=athena-export-backup.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -13,9 +13,11 @@ RUN apt-get update && apt-get install -y --no-install-recommends python3.12-venv
|
||||
"transformers==${TRANSFORMERS_VERSION}" \
|
||||
"accelerate==${ACCELERATE_VERSION}" \
|
||||
"huggingface-hub==${HF_HUB_VERSION}" \
|
||||
"nvidia-modelopt==0.46.0" bitsandbytes \
|
||||
sentencepiece protobuf safetensors pillow && \
|
||||
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker
|
||||
|
||||
COPY image_worker.py /app/image_worker.py
|
||||
COPY image_worker_9b.py /app/image_worker_9b.py
|
||||
USER 10002:10002
|
||||
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker.py"]
|
||||
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker_9b.py"]
|
||||
|
||||
@@ -0,0 +1,282 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Private FLUX.2 Klein 9B FP8 beta worker for Athena's two GPUs.
|
||||
|
||||
The FP8 diffusion transformer runs on the RTX 5080. A Qwen3-8B NF4 text
|
||||
encoder runs on the RTX 3060 while the profile controller temporarily pauses
|
||||
Qwen3-TTS. The transformer and encoder are released before VAE decoding so
|
||||
the 1024px decoder has sufficient workspace on the RTX 5080.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import gc
|
||||
import json
|
||||
import os
|
||||
import signal
|
||||
import time
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
from types import MethodType
|
||||
|
||||
HOST = os.environ.get("WORKER_HOST", "0.0.0.0")
|
||||
PORT = int(os.environ.get("WORKER_PORT", "8086"))
|
||||
TOKEN = os.environ.get("WORKER_TOKEN", "").strip()
|
||||
COMPONENT_DIR = os.environ.get("FLUX_COMPONENT_DIR", "/models/components")
|
||||
TRANSFORMER_FILE = os.environ.get(
|
||||
"FLUX_TRANSFORMER_FILE", "/models/fp8/flux-2-klein-9b-fp8.safetensors")
|
||||
OUTPUT_DIR = Path(os.environ.get("IMAGE_DIR", "/data/images")).resolve()
|
||||
ACTIVE = False
|
||||
|
||||
os.environ.setdefault("DIFFUSERS_VERBOSITY", "error")
|
||||
os.environ.setdefault("TRANSFORMERS_VERBOSITY", "error")
|
||||
os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")
|
||||
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
|
||||
|
||||
if len(TOKEN) < 32:
|
||||
raise RuntimeError("WORKER_TOKEN is missing or too short")
|
||||
|
||||
signal.signal(signal.SIGTERM, lambda *_: os._exit(0))
|
||||
|
||||
|
||||
def _devices(torch):
|
||||
if torch.cuda.device_count() != 2:
|
||||
raise RuntimeError("FLUX 9B beta requires exactly two visible CUDA GPUs")
|
||||
totals = {i: torch.cuda.get_device_properties(i).total_memory
|
||||
for i in range(torch.cuda.device_count())}
|
||||
transformer_index = max(totals, key=totals.get)
|
||||
encoder_index = min(totals, key=totals.get)
|
||||
return (transformer_index, encoder_index,
|
||||
torch.device(f"cuda:{transformer_index}"),
|
||||
torch.device(f"cuda:{encoder_index}"))
|
||||
|
||||
|
||||
def _install_fp8_converter():
|
||||
import diffusers.loaders.single_file_model as single_file_model
|
||||
|
||||
original = single_file_model.SINGLE_FILE_LOADABLE_CLASSES[
|
||||
"Flux2Transformer2DModel"]["checkpoint_mapping_fn"]
|
||||
scales = {}
|
||||
double_map = {
|
||||
"img_attn.proj": "attn.to_out.0",
|
||||
"img_mlp.0": "ff.linear_in",
|
||||
"img_mlp.2": "ff.linear_out",
|
||||
"txt_attn.proj": "attn.to_add_out",
|
||||
"txt_mlp.0": "ff_context.linear_in",
|
||||
"txt_mlp.2": "ff_context.linear_out",
|
||||
}
|
||||
single_map = {
|
||||
"linear1": "attn.to_qkv_mlp_proj",
|
||||
"linear2": "attn.to_out",
|
||||
}
|
||||
|
||||
def record(key, value):
|
||||
parts = key.split(".")
|
||||
scale_name, block = parts[-1], parts[1]
|
||||
within = ".".join(parts[2:-1])
|
||||
if parts[0] == "double_blocks":
|
||||
if within == "img_attn.qkv":
|
||||
targets = ("attn.to_q", "attn.to_k", "attn.to_v")
|
||||
elif within == "txt_attn.qkv":
|
||||
targets = ("attn.add_q_proj", "attn.add_k_proj",
|
||||
"attn.add_v_proj")
|
||||
else:
|
||||
targets = (double_map[within],)
|
||||
prefix = f"transformer_blocks.{block}"
|
||||
elif parts[0] == "single_blocks":
|
||||
targets = (single_map[within],)
|
||||
prefix = f"single_transformer_blocks.{block}"
|
||||
else:
|
||||
raise ValueError(f"unexpected FP8 scale key: {key}")
|
||||
for target in targets:
|
||||
scales.setdefault(f"{prefix}.{target}", {})[scale_name] = value.clone()
|
||||
|
||||
def convert(checkpoint, **kwargs):
|
||||
scales.clear()
|
||||
for key in list(checkpoint):
|
||||
if key.endswith((".input_scale", ".weight_scale")):
|
||||
record(key, checkpoint.pop(key))
|
||||
return original(checkpoint=checkpoint, **kwargs)
|
||||
|
||||
single_file_model.SINGLE_FILE_LOADABLE_CLASSES[
|
||||
"Flux2Transformer2DModel"]["checkpoint_mapping_fn"] = convert
|
||||
return scales
|
||||
|
||||
|
||||
def _fp8_forward(torch, module, inputs):
|
||||
shape = inputs.shape
|
||||
input_fp8 = ((inputs / module._fp8_input_scale)
|
||||
.clamp(torch.finfo(torch.float8_e4m3fn).min,
|
||||
torch.finfo(torch.float8_e4m3fn).max)
|
||||
.to(torch.float8_e4m3fn).reshape(-1, shape[-1]))
|
||||
output = torch._scaled_mm(
|
||||
input_fp8,
|
||||
module.weight.reshape(-1, module.weight.shape[-1]).t(),
|
||||
scale_a=module._fp8_input_scale,
|
||||
scale_b=module._fp8_weight_scale,
|
||||
bias=module.bias,
|
||||
out_dtype=inputs.dtype,
|
||||
use_fast_accum=True,
|
||||
)
|
||||
return output.reshape(*shape[:-1], output.shape[-1])
|
||||
|
||||
|
||||
def generate(data: dict) -> dict:
|
||||
global ACTIVE
|
||||
import torch
|
||||
from diffusers import (Flux2KleinPipeline, Flux2Transformer2DModel,
|
||||
NVIDIAModelOptConfig)
|
||||
from modelopt.torch.opt import enable_huggingface_checkpointing
|
||||
from modelopt.torch.quantization.config import FP8_DEFAULT_CFG
|
||||
from PIL import Image
|
||||
from transformers import BitsAndBytesConfig, Qwen3ForCausalLM
|
||||
|
||||
prompt, filename = data.get("prompt"), data.get("filename")
|
||||
if not isinstance(prompt, str) or not prompt.strip() or len(prompt) > 8000:
|
||||
raise ValueError("invalid prompt")
|
||||
if (not isinstance(filename, str) or Path(filename).name != filename
|
||||
or not filename.endswith(".png")):
|
||||
raise ValueError("invalid filename")
|
||||
width, height = int(data.get("width", 1024)), int(data.get("height", 1024))
|
||||
if (width, height) != (1024, 1024):
|
||||
raise ValueError("FLUX 9B beta currently supports only 1024x1024")
|
||||
if int(data.get("steps", 4)) != 4 or float(data.get("guidance", 1.0)) != 1.0:
|
||||
raise ValueError("FLUX 9B beta requires steps=4 and guidance=1.0")
|
||||
|
||||
source_files = data.get("source_files") or []
|
||||
if not isinstance(source_files, list) or len(source_files) > 4:
|
||||
raise ValueError("invalid source image list")
|
||||
source_images = []
|
||||
for name in source_files:
|
||||
if not isinstance(name, str) or Path(name).name != name:
|
||||
raise ValueError("invalid source image filename")
|
||||
source = (OUTPUT_DIR / name).resolve()
|
||||
if source.parent != OUTPUT_DIR or not source.is_file():
|
||||
raise ValueError("source image not found")
|
||||
with Image.open(source) as opened:
|
||||
source_images.append(opened.convert("RGB"))
|
||||
|
||||
started = time.monotonic()
|
||||
ACTIVE = True
|
||||
transformer = text_encoder = pipe = latent = decoded = image = None
|
||||
try:
|
||||
enable_huggingface_checkpointing()
|
||||
scales = _install_fp8_converter()
|
||||
tx_index, enc_index, tx_device, enc_device = _devices(torch)
|
||||
quantization = NVIDIAModelOptConfig(
|
||||
quant_type="FP8", weight_only=False,
|
||||
modelopt_config=FP8_DEFAULT_CFG)
|
||||
transformer = Flux2Transformer2DModel.from_single_file(
|
||||
TRANSFORMER_FILE, config=COMPONENT_DIR, subfolder="transformer",
|
||||
quantization_config=quantization, torch_dtype=torch.bfloat16,
|
||||
device_map={"": tx_index}, local_files_only=True)
|
||||
patched = 0
|
||||
for module_name, module in transformer.named_modules():
|
||||
if module_name not in scales:
|
||||
continue
|
||||
module.register_buffer("_fp8_input_scale",
|
||||
scales[module_name]["input_scale"])
|
||||
module.register_buffer("_fp8_weight_scale",
|
||||
scales[module_name]["weight_scale"])
|
||||
module.forward = MethodType(
|
||||
lambda self, inputs: _fp8_forward(torch, self, inputs), module)
|
||||
patched += 1
|
||||
if patched != len(scales):
|
||||
raise RuntimeError(f"patched only {patched} of {len(scales)} FP8 layers")
|
||||
transformer.to(tx_device)
|
||||
|
||||
text_encoder = Qwen3ForCausalLM.from_pretrained(
|
||||
os.path.join(COMPONENT_DIR, "text_encoder"),
|
||||
torch_dtype=torch.bfloat16, low_cpu_mem_usage=True,
|
||||
quantization_config=BitsAndBytesConfig(
|
||||
load_in_4bit=True, bnb_4bit_quant_type="nf4",
|
||||
bnb_4bit_compute_dtype=torch.bfloat16,
|
||||
bnb_4bit_use_double_quant=True),
|
||||
device_map={"": enc_index}, local_files_only=True)
|
||||
pipe = Flux2KleinPipeline.from_pretrained(
|
||||
COMPONENT_DIR, transformer=transformer, text_encoder=text_encoder,
|
||||
torch_dtype=torch.bfloat16, local_files_only=True)
|
||||
pipe.vae.enable_slicing()
|
||||
pipe.vae.enable_tiling()
|
||||
pipe.vae.to(tx_device)
|
||||
loaded = time.monotonic() - started
|
||||
|
||||
prompt_embeds, _ = pipe.encode_prompt(
|
||||
prompt.strip(), device=enc_device, max_sequence_length=128)
|
||||
prompt_embeds = prompt_embeds.to(tx_device)
|
||||
pipe.text_encoder = None
|
||||
seed = data.get("seed")
|
||||
generator = None if seed is None else torch.Generator(
|
||||
device=tx_device).manual_seed(int(seed))
|
||||
kwargs = {
|
||||
"prompt": None, "prompt_embeds": prompt_embeds,
|
||||
"height": height, "width": width, "num_inference_steps": 4,
|
||||
"guidance_scale": 1.0, "generator": generator,
|
||||
"output_type": "latent",
|
||||
}
|
||||
if source_images:
|
||||
kwargs["image"] = (source_images[0] if len(source_images) == 1
|
||||
else source_images)
|
||||
latent = pipe(**kwargs).images
|
||||
|
||||
pipe.transformer = None
|
||||
del transformer, text_encoder, prompt_embeds, generator
|
||||
transformer = text_encoder = None
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
latent = latent.to(device=tx_device, dtype=pipe.vae.dtype)
|
||||
decoded = pipe.vae.decode(latent, return_dict=False)[0]
|
||||
image = pipe.image_processor.postprocess(
|
||||
decoded.detach(), output_type="pil")[0]
|
||||
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||
image.save(OUTPUT_DIR / filename)
|
||||
return {"status": "ok", "filename": filename,
|
||||
"seconds": round(time.monotonic() - started, 3),
|
||||
"load_seconds": round(loaded, 3),
|
||||
"model": "FLUX.2-klein-9B-fp8-beta"}
|
||||
finally:
|
||||
for value in (image, decoded, latent, pipe, text_encoder, transformer):
|
||||
if value is not None:
|
||||
del value
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
ACTIVE = False
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
def log_message(self, fmt: str, *args: object) -> None:
|
||||
print(f"[flux9b-beta] {self.client_address[0]} {fmt % args}", flush=True)
|
||||
|
||||
def reply(self, status: int, payload: dict) -> None:
|
||||
body = json.dumps(payload, separators=(",", ":")).encode()
|
||||
self.send_response(status)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802
|
||||
if self.path == "/health":
|
||||
self.reply(200, {"status": "ok", "model_loaded": ACTIVE,
|
||||
"model": "FLUX.2-klein-9B-fp8-beta"})
|
||||
else:
|
||||
self.reply(404, {"error": "not found"})
|
||||
|
||||
def do_POST(self) -> None: # noqa: N802
|
||||
if self.headers.get("Authorization", "") != f"Bearer {TOKEN}":
|
||||
self.reply(401, {"error": "unauthorized"})
|
||||
return
|
||||
if self.path != "/generate":
|
||||
self.reply(404, {"error": "not found"})
|
||||
return
|
||||
try:
|
||||
length = int(self.headers.get("Content-Length", "0"))
|
||||
if length < 2 or length > 16384:
|
||||
raise ValueError("invalid request size")
|
||||
self.reply(200, generate(json.loads(self.rfile.read(length))))
|
||||
except Exception as exc:
|
||||
print(f"[flux9b-beta] generation failed: {type(exc).__name__}: "
|
||||
f"{str(exc)[:1000]}", flush=True)
|
||||
self.reply(500, {"status": "error", "message": str(exc)})
|
||||
|
||||
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
@@ -1,24 +0,0 @@
|
||||
FROM python:3.12-slim-bookworm
|
||||
|
||||
ARG PIPER_TTS_VERSION=1.6.0
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends ca-certificates curl ffmpeg gosu \
|
||||
&& python -m pip install --no-cache-dir "piper-tts==${PIPER_TTS_VERSION}" \
|
||||
&& useradd --system --uid 10003 --home-dir /nonexistent --shell /usr/sbin/nologin piper \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
COPY piper_worker.py /app/piper_worker.py
|
||||
COPY entrypoint.sh /usr/local/bin/mike-ai-piper-entrypoint
|
||||
RUN chmod 0755 /usr/local/bin/mike-ai-piper-entrypoint
|
||||
|
||||
ENV PIPER_DATA_DIR=/data \
|
||||
PIPER_VOICE=de_DE-thorsten-high \
|
||||
PIPER_VOICE_ALIAS=alloy \
|
||||
PIPER_HOST=0.0.0.0 \
|
||||
PIPER_PORT=8085
|
||||
|
||||
VOLUME ["/data"]
|
||||
EXPOSE 8085
|
||||
ENTRYPOINT ["/usr/local/bin/mike-ai-piper-entrypoint"]
|
||||
@@ -1,15 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
|
||||
data_dir=${PIPER_DATA_DIR:-/data}
|
||||
voice=${PIPER_VOICE:-de_DE-thorsten-high}
|
||||
|
||||
mkdir -p "$data_dir"
|
||||
chown 10003:10003 "$data_dir"
|
||||
|
||||
if [ ! -s "$data_dir/$voice.onnx" ] || [ ! -s "$data_dir/$voice.onnx.json" ]; then
|
||||
echo "Downloading Piper voice: $voice"
|
||||
gosu piper python -m piper.download_voices --data-dir "$data_dir" "$voice"
|
||||
fi
|
||||
|
||||
exec gosu piper python /app/piper_worker.py
|
||||
@@ -1,153 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Small, private Piper worker for the Mike AI profile router.
|
||||
|
||||
The public OpenAI-compatible endpoint remains in the router. This worker only
|
||||
accepts the narrow internal /status and /tts protocol and never logs input text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import threading
|
||||
import wave
|
||||
from http import HTTPStatus
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
|
||||
from piper import PiperVoice, SynthesisConfig
|
||||
|
||||
|
||||
DATA_DIR = Path(os.getenv("PIPER_DATA_DIR", "/data"))
|
||||
VOICE_NAME = os.getenv("PIPER_VOICE", "de_DE-thorsten-high")
|
||||
VOICE_ALIAS = os.getenv("PIPER_VOICE_ALIAS", "alloy")
|
||||
HOST = os.getenv("PIPER_HOST", "0.0.0.0")
|
||||
PORT = int(os.getenv("PIPER_PORT", "8085"))
|
||||
MAX_TEXT_CHARS = int(os.getenv("PIPER_MAX_TEXT_CHARS", "8000"))
|
||||
MAX_REQUEST_BYTES = int(os.getenv("PIPER_MAX_REQUEST_BYTES", "65536"))
|
||||
|
||||
VOICE_PATH = DATA_DIR / f"{VOICE_NAME}.onnx"
|
||||
VOICE = PiperVoice.load(str(VOICE_PATH))
|
||||
SYNTHESIS_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def synthesize_wav(text: str, speed: float) -> bytes:
|
||||
"""Synthesize a complete WAV in memory without retaining the text."""
|
||||
output = io.BytesIO()
|
||||
config = SynthesisConfig(length_scale=1.0 / speed)
|
||||
with SYNTHESIS_LOCK, wave.open(output, "wb") as wav_file:
|
||||
VOICE.synthesize_wav(text, wav_file, syn_config=config)
|
||||
return output.getvalue()
|
||||
|
||||
|
||||
def wav_to_mp3(wav_bytes: bytes) -> bytes:
|
||||
"""Convert Piper's WAV to the MP3 format Open WebUI requests by default."""
|
||||
result = subprocess.run(
|
||||
[
|
||||
"ffmpeg", "-hide_banner", "-loglevel", "error",
|
||||
"-f", "wav", "-i", "pipe:0",
|
||||
"-codec:a", "libmp3lame", "-b:a", "96k",
|
||||
"-f", "mp3", "pipe:1",
|
||||
],
|
||||
input=wav_bytes,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
check=False,
|
||||
timeout=120,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
raise RuntimeError("ffmpeg conversion failed")
|
||||
return result.stdout
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
protocol_version = "HTTP/1.1"
|
||||
|
||||
def log_message(self, fmt: str, *args: object) -> None:
|
||||
# Deliberately omit URLs and request bodies from the log.
|
||||
print(f"piper-worker: {self.command} -> {args[1] if len(args) > 1 else '-'}")
|
||||
|
||||
def send_bytes(self, status: int, body: bytes, content_type: str) -> None:
|
||||
self.send_response(status)
|
||||
self.send_header("Content-Type", content_type)
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.send_header("Cache-Control", "no-store")
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def send_json(self, status: int, payload: dict) -> None:
|
||||
self.send_bytes(
|
||||
status,
|
||||
json.dumps(payload, separators=(",", ":")).encode(),
|
||||
"application/json",
|
||||
)
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802
|
||||
if self.path != "/status":
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
self.send_json(
|
||||
HTTPStatus.OK,
|
||||
{
|
||||
"ready": True,
|
||||
"engine": "piper",
|
||||
"model": VOICE_NAME,
|
||||
"voices": [VOICE_ALIAS],
|
||||
},
|
||||
)
|
||||
|
||||
def do_POST(self) -> None: # noqa: N802
|
||||
if self.path != "/tts":
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
|
||||
try:
|
||||
content_length = int(self.headers.get("Content-Length", "0"))
|
||||
except ValueError:
|
||||
content_length = 0
|
||||
if content_length <= 0 or content_length > MAX_REQUEST_BYTES:
|
||||
self.send_json(HTTPStatus.REQUEST_ENTITY_TOO_LARGE, {"error": "invalid request size"})
|
||||
return
|
||||
|
||||
try:
|
||||
request = json.loads(self.rfile.read(content_length))
|
||||
text = request.get("text", "")
|
||||
voice = request.get("voice", VOICE_ALIAS)
|
||||
output_format = request.get("format", "mp3")
|
||||
speed = float(request.get("speed", 1.0))
|
||||
except (json.JSONDecodeError, TypeError, ValueError):
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid JSON request"})
|
||||
return
|
||||
|
||||
if not isinstance(text, str) or not text.strip() or len(text) > MAX_TEXT_CHARS:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid text"})
|
||||
return
|
||||
if voice != VOICE_ALIAS:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unknown voice"})
|
||||
return
|
||||
if output_format not in {"wav", "mp3"}:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "unsupported format"})
|
||||
return
|
||||
if not 0.5 <= speed <= 2.0:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
|
||||
return
|
||||
|
||||
try:
|
||||
audio = synthesize_wav(text.strip(), speed)
|
||||
if output_format == "mp3":
|
||||
audio = wav_to_mp3(audio)
|
||||
content_type = "audio/mpeg"
|
||||
else:
|
||||
content_type = "audio/wav"
|
||||
except (OSError, RuntimeError, subprocess.SubprocessError):
|
||||
self.send_json(HTTPStatus.INTERNAL_SERVER_ERROR, {"error": "synthesis failed"})
|
||||
return
|
||||
|
||||
self.send_bytes(HTTPStatus.OK, audio, content_type)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"Piper worker ready: {VOICE_NAME} as {VOICE_ALIAS} on {HOST}:{PORT}")
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
@@ -132,6 +132,25 @@ class LanguageSegmentationTests(unittest.TestCase):
|
||||
self.assertIn("5 bis 6 Uhr", spoken)
|
||||
self.assertIn("Wind maximal 16 Kilometer pro Stunde", spoken)
|
||||
|
||||
def test_qwen_speaks_aspect_ratios_as_ratios(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
"Cover sind im Hochformat (2:3), Screenshots im Querformat "
|
||||
"(16:9), ein Quadrat im Seitenverhältnis 2:2 und 4:3-Format."
|
||||
)
|
||||
self.assertIn("Hochformat (2 zu 3)", spoken)
|
||||
self.assertIn("Querformat (16 zu 9)", spoken)
|
||||
self.assertIn("Seitenverhältnis 2 zu 2", spoken)
|
||||
self.assertIn("4 zu 3-Format", spoken)
|
||||
|
||||
def test_qwen_keeps_clock_times_distinct_from_aspect_ratios(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
"Beginn um 16:09 Uhr, Fehler um 02:14; das Videoformat ist 16:9."
|
||||
)
|
||||
self.assertIn("16 Uhr 9", spoken)
|
||||
self.assertNotIn("16 Uhr 9 Uhr", spoken)
|
||||
self.assertIn("2 Uhr 14", spoken)
|
||||
self.assertIn("Videoformat ist 16 zu 9", spoken)
|
||||
|
||||
def test_qwen_speaks_strict_date_ranges_as_calendar_dates(self):
|
||||
spoken = gateway.prepare_for_qwen_speech(
|
||||
"Neuigkeiten vom 04.–05.09. und Vergleich 04.09.–06.10.2026."
|
||||
@@ -225,25 +244,20 @@ class LanguageSegmentationTests(unittest.TestCase):
|
||||
self.assertTrue(all(len(part.split()) > 4 for _, part in segments))
|
||||
|
||||
|
||||
class FallbackTests(unittest.TestCase):
|
||||
class BackendFailureTests(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.original_xtts = gateway.synthesize_xtts
|
||||
self.original_piper = gateway.synthesize_piper
|
||||
self.original_qwen = gateway.synthesize_qwen
|
||||
|
||||
def tearDown(self):
|
||||
gateway.synthesize_xtts = self.original_xtts
|
||||
gateway.synthesize_piper = self.original_piper
|
||||
gateway.synthesize_qwen = self.original_qwen
|
||||
|
||||
def test_piper_is_used_when_xtts_fails(self):
|
||||
def test_qwen_failure_is_reported_without_fallback(self):
|
||||
def fail(*_args):
|
||||
raise RuntimeError("synthetic XTTS failure")
|
||||
raise RuntimeError("synthetic Qwen failure")
|
||||
|
||||
gateway.synthesize_xtts = fail
|
||||
gateway.synthesize_piper = lambda *_args: (b"piper", "audio/wav")
|
||||
self.assertEqual(
|
||||
gateway.synthesize("synthetic test", "wav", 1.0),
|
||||
(b"piper", "audio/wav"),
|
||||
)
|
||||
gateway.synthesize_qwen = fail
|
||||
with self.assertRaisesRegex(RuntimeError, "synthetic Qwen failure"):
|
||||
gateway.synthesize("synthetic test", "wav", 1.0)
|
||||
|
||||
|
||||
class AudioJoinTests(unittest.TestCase):
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Private Qwen3-TTS-first gateway with a Piper fallback.
|
||||
"""Private Qwen3-TTS gateway.
|
||||
|
||||
The gateway implements the narrow /status and /tts protocol already consumed
|
||||
by the profile router. Request text is never logged or persisted.
|
||||
@@ -8,6 +8,7 @@ by the profile router. Request text is never logged or persisted.
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import http.client
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
@@ -17,6 +18,7 @@ import time
|
||||
import unicodedata
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
import urllib.parse
|
||||
import wave
|
||||
from array import array
|
||||
from http import HTTPStatus
|
||||
@@ -31,7 +33,6 @@ QWEN_TTS_VOICE = os.getenv("QWEN_TTS_VOICE", "serena")
|
||||
QWEN_TTS_LANGUAGE = os.getenv("QWEN_TTS_LANGUAGE", "German")
|
||||
QWEN_TTS_TIMEOUT = float(os.getenv("QWEN_TTS_TIMEOUT", "120"))
|
||||
XTTS_URL = os.getenv("XTTS_URL", "http://xtts:80").rstrip("/")
|
||||
PIPER_URL = os.getenv("PIPER_URL", "http://piper:8085").rstrip("/")
|
||||
VOICE_ALIAS = os.getenv("TTS_VOICE_ALIAS", "alloy")
|
||||
XTTS_SPEAKER = os.getenv("XTTS_SPEAKER", "Annmarie Nele")
|
||||
DEFAULT_LANGUAGE = os.getenv("TTS_DEFAULT_LANGUAGE", "de")
|
||||
@@ -41,7 +42,6 @@ MAX_TEXT_CHARS = int(os.getenv("TTS_MAX_TEXT_CHARS", "8000"))
|
||||
MAX_REQUEST_BYTES = int(os.getenv("TTS_MAX_REQUEST_BYTES", "65536"))
|
||||
MAX_AUDIO_BYTES = int(os.getenv("TTS_MAX_AUDIO_BYTES", str(64 * 1024 * 1024)))
|
||||
XTTS_TIMEOUT = float(os.getenv("XTTS_TIMEOUT", "120"))
|
||||
PIPER_TIMEOUT = float(os.getenv("PIPER_TIMEOUT", "120"))
|
||||
QUEUE_TIMEOUT = float(os.getenv("XTTS_QUEUE_TIMEOUT", "15"))
|
||||
# XTTS loses natural prosody when a sentence is synthesized as many tiny
|
||||
# requests: every request starts a fresh utterance. Keep complete sentences
|
||||
@@ -61,8 +61,7 @@ SPEAKER_LOCK = threading.Lock()
|
||||
SPEAKER_CONDITIONING: dict | None = None
|
||||
STATE = {
|
||||
"last_backend": None,
|
||||
"xtts_failures": 0,
|
||||
"piper_fallbacks": 0,
|
||||
"qwen_failures": 0,
|
||||
"last_error": None,
|
||||
}
|
||||
|
||||
@@ -267,6 +266,36 @@ def _spoken_ipv4(match: re.Match) -> str:
|
||||
return " Punkt ".join(str(int(part)) for part in match.group(0).split("."))
|
||||
|
||||
|
||||
def _normalize_aspect_ratios(text: str) -> str:
|
||||
"""Speak colon notation as a ratio only when the surrounding text says so.
|
||||
|
||||
A bare ``16:09`` remains a clock time. This deliberately avoids a global
|
||||
replacement of common ratios because ``16:9`` can also be a valid time.
|
||||
"""
|
||||
cue = (
|
||||
r"(?:Seitenverh[aä]ltnis|Bildseitenverh[aä]ltnis|Bildformat|"
|
||||
r"Videoformat|Hochformat|Querformat|Format|Aspect[- ]?Ratio)"
|
||||
)
|
||||
text = re.sub(
|
||||
rf"\b({cue}\b(?:\s+(?:von|im|ist|betr[aä]gt))?\s*[\(\[]?\s*)"
|
||||
rf"(\d{{1,3}})\s*:\s*(\d{{1,3}})",
|
||||
lambda match: (
|
||||
f"{match.group(1)}{int(match.group(2))} zu {int(match.group(3))}"
|
||||
),
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
return re.sub(
|
||||
rf"\b(\d{{1,3}})\s*:\s*(\d{{1,3}})"
|
||||
rf"(\s*[-‐‑‒–—−]?\s*{cue}\b)",
|
||||
lambda match: (
|
||||
f"{int(match.group(1))} zu {int(match.group(2))}{match.group(3)}"
|
||||
),
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def normalize_for_german_speech(text: str) -> str:
|
||||
"""Turn common visual notation into unambiguous spoken German."""
|
||||
text = re.sub(r"\bv\.\s*a\.", "vor allem", text, flags=re.IGNORECASE)
|
||||
@@ -279,10 +308,14 @@ def normalize_for_german_speech(text: str) -> str:
|
||||
_spoken_ipv4,
|
||||
text,
|
||||
)
|
||||
# A colon is ambiguous between an aspect ratio and a clock time. Resolve
|
||||
# ratios first, but only when an explicit format cue is present.
|
||||
text = _normalize_aspect_ratios(text)
|
||||
text = re.sub(
|
||||
r"\b([01]?\d|2[0-3]):([0-5]\d)\b",
|
||||
r"\b([01]?\d|2[0-3]):([0-5]\d)\b(?:\s*Uhr\b)?",
|
||||
lambda match: f"{int(match.group(1))} Uhr {int(match.group(2))}",
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
text = re.sub(
|
||||
r"\b([01]?\d|2[0-3])\s*[-‐‑‒–—−]\s*"
|
||||
@@ -702,18 +735,6 @@ def synthesize_xtts(text: str, output_format: str,
|
||||
return _convert(_wav(_join_pcm(pcm_parts)), output_format, speed)
|
||||
|
||||
|
||||
def synthesize_piper(text: str, output_format: str,
|
||||
speed: float) -> tuple[bytes, str]:
|
||||
upstream_format = "wav" if output_format == "pcm" else output_format
|
||||
audio, content_type = _request(
|
||||
f"{PIPER_URL}/tts",
|
||||
payload={"text": text, "voice": "alloy", "speed": speed,
|
||||
"format": upstream_format},
|
||||
timeout=PIPER_TIMEOUT,
|
||||
)
|
||||
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
|
||||
|
||||
|
||||
def synthesize_qwen(text: str, output_format: str,
|
||||
speed: float) -> tuple[bytes, str]:
|
||||
text = prepare_for_qwen_speech(text)
|
||||
@@ -728,6 +749,47 @@ def synthesize_qwen(text: str, output_format: str,
|
||||
return _convert(audio, "pcm", 1.0) if output_format == "pcm" else (audio, content_type)
|
||||
|
||||
|
||||
def open_qwen_pcm_stream(text: str, chunk_size: int = 4) \
|
||||
-> tuple[http.client.HTTPConnection, http.client.HTTPResponse]:
|
||||
"""Open Qwen's native token-level PCM stream without buffering it.
|
||||
|
||||
The upstream emits headerless 24 kHz mono signed 16-bit little-endian
|
||||
PCM. Keeping this response streaming is what lets playback begin while
|
||||
the remainder of the sentence is still being synthesized.
|
||||
"""
|
||||
parsed = urllib.parse.urlparse(QWEN_TTS_URL)
|
||||
if parsed.scheme != "http" or not parsed.hostname:
|
||||
raise RuntimeError("QWEN_TTS_URL must be an http URL")
|
||||
port = parsed.port or 80
|
||||
prefix = parsed.path.rstrip("/")
|
||||
payload = json.dumps({
|
||||
"model": QWEN_TTS_MODEL,
|
||||
"input": prepare_for_qwen_speech(text),
|
||||
"voice": QWEN_TTS_VOICE,
|
||||
"language": QWEN_TTS_LANGUAGE,
|
||||
"chunk_size": chunk_size,
|
||||
}, separators=(",", ":")).encode()
|
||||
connection = http.client.HTTPConnection(
|
||||
parsed.hostname, port, timeout=QWEN_TTS_TIMEOUT)
|
||||
try:
|
||||
connection.request(
|
||||
"POST",
|
||||
f"{prefix}/v1/audio/speech/pcm-stream",
|
||||
body=payload,
|
||||
headers={"Content-Type": "application/json",
|
||||
"Accept": "application/octet-stream"},
|
||||
)
|
||||
response = connection.getresponse()
|
||||
if response.status != HTTPStatus.OK:
|
||||
message = response.read(512).decode(errors="replace")
|
||||
raise RuntimeError(
|
||||
f"Qwen PCM stream failed ({response.status}): {message}")
|
||||
return connection, response
|
||||
except Exception:
|
||||
connection.close()
|
||||
raise
|
||||
|
||||
|
||||
def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]:
|
||||
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
|
||||
if acquired:
|
||||
@@ -737,22 +799,18 @@ def synthesize(text: str, output_format: str, speed: float) -> tuple[bytes, str]
|
||||
STATE["last_backend"] = "qwen3-tts-1.7b"
|
||||
STATE["last_error"] = None
|
||||
return audio
|
||||
except Exception as exc: # fallback must cover all Qwen failures
|
||||
except Exception as exc:
|
||||
with STATE_LOCK:
|
||||
STATE["xtts_failures"] += 1
|
||||
STATE["qwen_failures"] += 1
|
||||
STATE["last_error"] = type(exc).__name__
|
||||
raise
|
||||
finally:
|
||||
SYNTHESIS_LOCK.release()
|
||||
else:
|
||||
with STATE_LOCK:
|
||||
STATE["xtts_failures"] += 1
|
||||
STATE["qwen_failures"] += 1
|
||||
STATE["last_error"] = "queue-timeout"
|
||||
|
||||
audio = synthesize_piper(text, output_format, speed)
|
||||
with STATE_LOCK:
|
||||
STATE["last_backend"] = "piper"
|
||||
STATE["piper_fallbacks"] += 1
|
||||
return audio
|
||||
raise RuntimeError("speech queue timeout")
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
@@ -779,25 +837,26 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
primary_ready = _reachable(QWEN_TTS_URL, "/health")
|
||||
fallback_ready = _reachable(PIPER_URL, "/status")
|
||||
with STATE_LOCK:
|
||||
state = dict(STATE)
|
||||
# This endpoint is also the container liveness check. Qwen3-TTS is
|
||||
# deliberately stopped in exclusive GPU modes such as Applio, so the
|
||||
# gateway itself must stay healthy while reporting ready=false.
|
||||
self.send_json(
|
||||
HTTPStatus.OK if fallback_ready else HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
HTTPStatus.OK,
|
||||
{
|
||||
"ready": fallback_ready,
|
||||
"engine": "qwen3-tts-with-piper-fallback",
|
||||
"ready": primary_ready,
|
||||
"engine": "qwen3-tts",
|
||||
"model": "Qwen3-TTS-12Hz-1.7B-Base",
|
||||
"voices": [VOICE_ALIAS],
|
||||
"speaker": QWEN_TTS_VOICE,
|
||||
"primary_ready": primary_ready,
|
||||
"fallback_ready": fallback_ready,
|
||||
**state,
|
||||
},
|
||||
)
|
||||
|
||||
def do_POST(self) -> None: # noqa: N802
|
||||
if self.path != "/tts":
|
||||
if self.path not in {"/tts", "/tts/pcm-stream"}:
|
||||
self.send_json(HTTPStatus.NOT_FOUND, {"error": "not found"})
|
||||
return
|
||||
try:
|
||||
@@ -810,7 +869,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
return
|
||||
try:
|
||||
request = json.loads(self.rfile.read(length))
|
||||
text = request.get("text", "")
|
||||
text = request.get("input", request.get("text", ""))
|
||||
voice = request.get("voice", VOICE_ALIAS)
|
||||
output_format = request.get("format", "mp3")
|
||||
speed = float(request.get("speed", 1.0))
|
||||
@@ -829,6 +888,9 @@ class Handler(BaseHTTPRequestHandler):
|
||||
if not 0.5 <= speed <= 2.0:
|
||||
self.send_json(HTTPStatus.BAD_REQUEST, {"error": "invalid speed"})
|
||||
return
|
||||
if self.path == "/tts/pcm-stream":
|
||||
self._stream_qwen_pcm(text.strip(), request)
|
||||
return
|
||||
started = time.monotonic()
|
||||
try:
|
||||
audio, content_type = synthesize(text.strip(), output_format, speed)
|
||||
@@ -836,13 +898,81 @@ class Handler(BaseHTTPRequestHandler):
|
||||
with STATE_LOCK:
|
||||
STATE["last_error"] = type(exc).__name__
|
||||
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
{"error": "all local speech backends failed"})
|
||||
{"error": "local Qwen3-TTS backend failed"})
|
||||
return
|
||||
print(f"tts-gateway: synthesized via {STATE['last_backend']} in "
|
||||
f"{time.monotonic() - started:.2f}s")
|
||||
self.send_bytes(HTTPStatus.OK, audio, content_type)
|
||||
|
||||
def _stream_qwen_pcm(self, text: str, request: dict) -> None:
|
||||
"""Unframe Qwen's PCM frames and relay their audio immediately."""
|
||||
try:
|
||||
chunk_size = max(1, min(32, int(request.get("chunk_size", 4))))
|
||||
except (TypeError, ValueError):
|
||||
self.send_json(HTTPStatus.BAD_REQUEST,
|
||||
{"error": "invalid chunk_size"})
|
||||
return
|
||||
acquired = SYNTHESIS_LOCK.acquire(timeout=QUEUE_TIMEOUT)
|
||||
if not acquired:
|
||||
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
{"error": "speech queue timeout"})
|
||||
return
|
||||
connection = None
|
||||
started = time.monotonic()
|
||||
headers_sent = False
|
||||
try:
|
||||
connection, response = open_qwen_pcm_stream(text, chunk_size)
|
||||
self.send_response(HTTPStatus.OK)
|
||||
self.send_header("Content-Type", "application/octet-stream")
|
||||
self.send_header("Cache-Control", "no-store")
|
||||
self.send_header("Connection", "close")
|
||||
self.end_headers()
|
||||
headers_sent = True
|
||||
first = True
|
||||
while True:
|
||||
frame_header = response.read(4)
|
||||
if not frame_header:
|
||||
break
|
||||
if len(frame_header) != 4:
|
||||
raise RuntimeError("truncated Qwen PCM frame header")
|
||||
frame_length = int.from_bytes(frame_header, "big")
|
||||
if frame_length == 0:
|
||||
break
|
||||
if frame_length > MAX_AUDIO_BYTES:
|
||||
raise RuntimeError("Qwen PCM frame is too large")
|
||||
remaining = frame_length
|
||||
while remaining:
|
||||
chunk = response.read(min(16384, remaining))
|
||||
if not chunk:
|
||||
raise RuntimeError("truncated Qwen PCM frame")
|
||||
if first:
|
||||
print("tts-gateway: first Qwen PCM chunk in "
|
||||
f"{time.monotonic() - started:.2f}s")
|
||||
first = False
|
||||
self.wfile.write(chunk)
|
||||
self.wfile.flush()
|
||||
remaining -= len(chunk)
|
||||
with STATE_LOCK:
|
||||
STATE["last_backend"] = "qwen3-tts-1.7b-stream"
|
||||
STATE["last_error"] = None
|
||||
except Exception as exc:
|
||||
with STATE_LOCK:
|
||||
STATE["last_error"] = type(exc).__name__
|
||||
# Once PCM started, simply close the truncated response. Sending
|
||||
# JSON into the audio stream would produce loud corrupt samples.
|
||||
if not headers_sent and not self.wfile.closed:
|
||||
try:
|
||||
self.send_json(HTTPStatus.SERVICE_UNAVAILABLE,
|
||||
{"error": "local PCM stream failed"})
|
||||
except (OSError, BrokenPipeError):
|
||||
pass
|
||||
finally:
|
||||
if connection is not None:
|
||||
connection.close()
|
||||
SYNTHESIS_LOCK.release()
|
||||
self.close_connection = True
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(f"TTS gateway ready on {HOST}:{PORT}; primary={QWEN_TTS_VOICE}; fallback=Piper")
|
||||
print(f"TTS gateway ready on {HOST}:{PORT}; backend=Qwen3-TTS; voice={QWEN_TTS_VOICE}")
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
|
||||
@@ -1,70 +0,0 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
|
||||
target=/opt/hermes/cron/scheduler_delivery.py
|
||||
broken='env.pop("HERMES_HOME", None)'
|
||||
|
||||
if [ -f "$target" ] && grep -Fq "$broken" "$target"; then
|
||||
sed -i '/^[[:space:]]*env\.pop("HERMES_HOME", None)[[:space:]]*$/d' "$target"
|
||||
echo "[cron-profile-root-fix] preserved HERMES_HOME for named-profile delivery"
|
||||
else
|
||||
echo "[cron-profile-root-fix] upstream code already fixed or layout changed; no action"
|
||||
fi
|
||||
|
||||
cli_dir="${HERMES_HOME:-/opt/data}/.local/bin"
|
||||
mkdir -p "$cli_dir"
|
||||
ln -sfn /opt/hermes/.venv/bin/hermes "$cli_dir/hermes"
|
||||
ln -sfn /opt/hermes/.venv/bin/hermes /usr/local/bin/hermes
|
||||
|
||||
# Hermes' generic sentence splitter treats periods in German abbreviations as
|
||||
# sentence ends. It also submits every sentence as a separate generative TTS
|
||||
# request, which creates long gaps with local Qwen3-TTS. Keep the first sentence
|
||||
# immediate, then group following complete sentences into modest ~240-character
|
||||
# chunks so playback remains responsive but substantially more continuous.
|
||||
tts_target=/opt/hermes/tools/tts_streaming.py
|
||||
if [ -f "$tts_target" ] && grep -Fq 'self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)' "$tts_target"; then
|
||||
TTS_TARGET="$tts_target" python3 <<'PY'
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
p = Path(os.environ["TTS_TARGET"])
|
||||
s = p.read_text()
|
||||
|
||||
base_feed = ' self.buf = _THINK_BLOCK_RE.sub("", self.buf + delta)'
|
||||
abbr_lines = [
|
||||
' self.buf = re.sub(r"\\bca\\.(?=\\s)", "circa", self.buf, flags=re.IGNORECASE)',
|
||||
' self.buf = re.sub(r"\\bv\\.\\s*a\\.(?=\\s)", "vor allem", self.buf, flags=re.IGNORECASE)',
|
||||
' self.buf = re.sub(r"\\bmax\\.(?=\\s)", "maximal", self.buf, flags=re.IGNORECASE)',
|
||||
]
|
||||
for line in abbr_lines:
|
||||
s = s.replace("\n" + line, "")
|
||||
|
||||
init_old = ' self.buf = ""'
|
||||
init_extra = (
|
||||
'\n self.followup_target_len = 240'
|
||||
'\n self._emitted_first = False'
|
||||
)
|
||||
s = s.replace(init_old + init_extra, init_old)
|
||||
|
||||
threshold_old = ' if len(head.strip()) < self.min_len:'
|
||||
threshold_new = (
|
||||
' threshold = (self.min_len if not self._emitted_first '
|
||||
'else self.followup_target_len)\n'
|
||||
' if len(head.strip()) < threshold:'
|
||||
)
|
||||
s = s.replace(threshold_new, threshold_old)
|
||||
|
||||
append_old = ' out.append(head)'
|
||||
append_new = append_old + '\n self._emitted_first = True'
|
||||
s = s.replace(append_new, append_old)
|
||||
|
||||
s = s.replace(base_feed, base_feed + "\n" + "\n".join(abbr_lines), 1)
|
||||
s = s.replace(init_old, init_old + init_extra, 1)
|
||||
s = s.replace(threshold_old, threshold_new, 1)
|
||||
s = s.replace(append_old, append_new, 1)
|
||||
p.write_text(s)
|
||||
PY
|
||||
echo "[tts-streaming-fix] enabled German normalization and adaptive sentence grouping"
|
||||
else
|
||||
echo "[tts-streaming-fix] upstream code already fixed or layout changed; no action"
|
||||
fi
|
||||
@@ -144,13 +144,12 @@ stt:
|
||||
model: "base"
|
||||
language: "de"
|
||||
|
||||
# Reuse Athena's OpenAI-compatible TTS route. It currently serves XTTS v2 with
|
||||
# Annmarie Nele and transparently falls back to Piper when XTTS is unavailable.
|
||||
# Reuse Athena's OpenAI-compatible Qwen3-TTS route.
|
||||
tts:
|
||||
provider: "openai"
|
||||
speed: 1.0
|
||||
openai:
|
||||
model: "piper"
|
||||
model: "qwen3-tts"
|
||||
voice: "alloy"
|
||||
speed: 1.0
|
||||
base_url: "http://router:8081/v1"
|
||||
|
||||
@@ -1,120 +0,0 @@
|
||||
---
|
||||
name: athena-ai-profile-router
|
||||
description: Use when operating the AI profile router on Athena.
|
||||
metadata:
|
||||
hermes:
|
||||
editorial_name: "Athena AI-Profile-Router"
|
||||
editorial_description: "Betrieb und Abfragen der lokalen KI-Plattform auf Athena: Profile, Status, Umschaltung."
|
||||
---
|
||||
|
||||
# Athena AI-Profile-Router
|
||||
|
||||
Lokale KI-Plattform auf **Athena** (root@192.168.1.212, Debian 13, GPU-Host).
|
||||
Repo: `/root/AI-Profile-Router` (Gitea: `michael/AI-Profile-Router`).
|
||||
SSH: `ssh -i /opt/data/athena_key -o UserKnownHostsFile=/opt/data/.ssh/known_hosts -o IdentitiesOnly=yes root@192.168.1.212`
|
||||
|
||||
## Wichtigste Regel
|
||||
|
||||
**Der Host steht physisch in einer anderen Stadt. Es gibt keinen schnellen
|
||||
Zugang.**
|
||||
|
||||
- **NIEMALS** herunterfahren (`shutdown`, `poweroff`, `halt`) oder neu
|
||||
starten (`reboot`, `init 6`).
|
||||
- Keine Aktion, die die Erreichbarkeit gefährdet: keine Änderungen an `lan0`,
|
||||
Firewall-Default-Policies, `DOCKER-USER`-Regeln oder Routing ohne explizite
|
||||
Freigabe und Rückfallplan; keine Reboot-Pflicht auslösenden Aktionen
|
||||
(Kernel-, NVIDIA-Treiber-Installation, `apt full-upgrade` mit Kernel) ohne
|
||||
ausdrückliche Freigabe; keine Abschaltung von WireGuard, SSH oder dem
|
||||
Gateway-Container.
|
||||
- Installer-Exit-Code 20 (NVIDIA-Treiber, Reboot nötig) wird nicht automatisch
|
||||
nachgeholt — der Betreiber entscheidet.
|
||||
- Nach jeder Netz-/Firewall-/WG-Änderung verifizieren: (1) SSH-Roundtrip,
|
||||
(2) WG-Tunnel up, (3) Router `/health` und `/ready` = 200.
|
||||
|
||||
## Architektur (Kurzform)
|
||||
|
||||
- **Profile Router** (`mike-ai-router`, OpenAI-kompatible API, Port 8081):
|
||||
einzige Client-Schnittstelle. Clients nutzen `http://<host>:8081/v1` mit
|
||||
Bearer-Key aus `/etc/mike-ai/router-api-key`.
|
||||
- **llama.cpp** (`mike-ai-llama-*`): ein Container pro Profil, immer exakt
|
||||
einer aktiv; nur Docker-intern (Port 8080, nie direkt von Clients).
|
||||
- **Profile Controller** (`mike-ai-profile-controller`): einziger Dienst mit
|
||||
Docker-Socket; begrenzter Containerwechsel.
|
||||
- **Open WebUI** (`mike-ai-open-webui`, Port 8080): einzige normale Oberfläche.
|
||||
- **MCP-Tool-Stack** (`platform/mcp/`): getrennte Container für Web, HA, ARR,
|
||||
Unraid — nur über internes `mike-ai-tools`-Netz.
|
||||
- WireGuard-Isolation: KI-Dienste nur über WG-Adresse erreichbar,
|
||||
fail-closed bei Tunnelausfall (Blackhole-Default-Route).
|
||||
|
||||
## Profile (Live-Stand 11.09.2026)
|
||||
|
||||
| Profil | Virtuelles Modell | Kontext | Zweck |
|
||||
|---|---|---:|---|
|
||||
| fast | `qwen-fast` | 76.800 | Alltag, Agenten, hohe Geschwindigkeit |
|
||||
| medium | `qwen-medium` | 160.000 | mehr Kontext |
|
||||
| large | `qwen-large` | 192.000 | lange Sitzungen |
|
||||
| ultra | `qwen-ultra` | 262.144 | maximale Kontextlänge |
|
||||
| uncensored | `qwen-uncensored` | 80.000 | unzensiert |
|
||||
|
||||
Aktives Profil: `GET /status` → `current_profile`.
|
||||
Profilwechsel: `POST /fast`, `/medium`, `/large`, `/ultra`, `/uncensored`
|
||||
(authentifiziert) oder manuell `llama-profile <name>` auf dem Host.
|
||||
|
||||
## Wichtige Befehle (per SSH auf Athena)
|
||||
|
||||
```bash
|
||||
# Status (Router lebt, Profil, Upstream, Telemetrie)
|
||||
KEY=$(cat /etc/mike-ai/router-api-key)
|
||||
curl -s -H "Authorization: Bearer ***" http://172.30.20.3:8081/status
|
||||
|
||||
# Health / Ready (ohne Auth)
|
||||
curl -s -o /dev/null -w '%{http_code}' http://172.30.20.3:8081/health # 200 = Router lebt
|
||||
curl -s -o /dev/null -w '%{http_code}' http://172.30.20.3:8081/ready # 200 = Modell bereit
|
||||
|
||||
# Virtuelle Modelle
|
||||
curl -s -H "Authorization: Bearer ***" http://172.30.20.3:8081/v1/models
|
||||
|
||||
# Profilwechsel
|
||||
curl -s -X POST -H "Authorization: Bearer ***" http://172.30.20.3:8081/ultra
|
||||
|
||||
# Aktive Profile-Registry (Router-Container)
|
||||
docker exec mike-ai-router cat /app/router_profiles.json
|
||||
|
||||
# Container-Status
|
||||
docker ps --format '{{.Names}}\t{{.Status}}' | grep mike-ai
|
||||
```
|
||||
|
||||
Router-IP auf dem internen `mike-ai_control`-Netz: `172.30.20.3`
|
||||
(neu ermitteln: `docker inspect mike-ai-router --format '{{.NetworkSettings.Networks.mike-ai_control.IPAddress}}'`).
|
||||
|
||||
## Fehler- und Recovery-Verhalten
|
||||
|
||||
- `/health=200` + `/ready=503` → Router lebt, Modell lädt/wechselt noch.
|
||||
- `429` → Parallelitätsgrenze (max. 16) erreicht; Client mit Backoff.
|
||||
- Profilwechsel bricht ab, wenn laufende Chats den Drain-Timeout (600 s)
|
||||
überschreiten — er beendet niemals absichtlich einen Chat.
|
||||
- Nach Routerabsturz: letztes stabiles Profil aus atomarer Zustandsdatei
|
||||
(`/var/lib/mike-ai-profile-router/state.json`) rekonstruiert.
|
||||
- Upgrade-Regel: niemals Build, Quantisierung und Profil gleichzeitig ändern.
|
||||
|
||||
## Git / Repo
|
||||
|
||||
- Remote: `git@192.168.1.2:michael/AI-Profile-Router.git` (Gitea auf Unraid).
|
||||
- Athena erreicht das Heimnetz **nur über WireGuard** — ist der WG-Tunnel
|
||||
down, schlägt `git push` mit Connection timed out fehl. Nicht mit
|
||||
Firewall-/Routing-Änderungen gegensteuern; Tunnel-Status prüfen und
|
||||
melden. Commits bleiben lokal auf Athena, bis der Tunnel wieder up ist.
|
||||
- Git-Identität auf Athena ist repo-lokal gesetzt (Mikei386).
|
||||
|
||||
## Sicherheitsregeln
|
||||
|
||||
- API-Key (`/etc/mike-ai/router-api-key`) nie in Chat, Repo oder URLs.
|
||||
- Clients nie direkt Port 8080 (llama.cpp) — immer über Router 8081.
|
||||
- `config/install.env` bleibt lokal (0600), wird nicht committet.
|
||||
- Systempartition dauerhaft unter 85 % halten; nur ein Textmodell gleichzeitig.
|
||||
|
||||
## Backup
|
||||
|
||||
Gesichert: Repo, Modellmanifest (Hashes, ohne Secrets), `/etc/mike-ai`
|
||||
verschlüsselt, systemd-Konfiguration, Benchmarkresultate.
|
||||
Nicht nötig: Builds, Venvs, Caches, Modelle (Quelle + Prüfsumme dokumentiert).
|
||||
@@ -0,0 +1,47 @@
|
||||
# Athena architecture and modes
|
||||
|
||||
Use `ATHENA.md` as the short operational truth and
|
||||
`docs/CONTAINER_INVENTORY.md` for the current mapping of container, model and
|
||||
role. If live state disagrees with documentation, report the discrepancy and
|
||||
correct the durable source when the user requested maintenance.
|
||||
|
||||
## Boundaries
|
||||
|
||||
- Hermes, chats, skills and portable specialist MCPs live on Unraid.
|
||||
- Athena is the inference host. Its canonical checkout is
|
||||
`/opt/mike-ai/stack`; models live under `/data/models`; local secrets and
|
||||
runtime configuration live under `/etc/mike-ai` and never enter Git.
|
||||
- The Profile Router is the single OpenAI-compatible address clients use.
|
||||
- The Profile Controller is the only component allowed to orchestrate approved
|
||||
model and specialist workers.
|
||||
- The Athena Operator is host-bound and is the normal maintenance interface.
|
||||
|
||||
## Exclusive states
|
||||
|
||||
Athena has five mutually exclusive persistent modes: `llm`, `music`,
|
||||
`separation`, `voice`, and `voicechange`. Image generation is a transactional
|
||||
request: it temporarily pauses the active text profile and Qwen3-TTS, runs the
|
||||
image worker, then restores the previous LLM state.
|
||||
|
||||
Only one heavy GPU path may be active. Do not manually start a second GPU
|
||||
worker around the controller. The lightweight dashboard, router, controller,
|
||||
gateway, UI, CPU-STT, backup and operator containers may remain active.
|
||||
|
||||
## Model profiles
|
||||
|
||||
- Fast: Qwen3.8-27B IQ4-MIX, 76,800 tokens.
|
||||
- Medium: Qwen3.8-27B IQ4_XS-pure, 160,000 tokens, vision.
|
||||
- Large: the same Q4 model, 192,000 tokens, vision.
|
||||
- Ultra: the same Q4 model, 262,144 tokens, no vision projector.
|
||||
- Uncensored: Abliterated Q4_K_M, 80,000 tokens, vision.
|
||||
|
||||
Medium, Large, Ultra and Beta distribute their runtime across both GPUs. Do not
|
||||
infer allocation from model size alone; verify the profile's Compose arguments
|
||||
and live VRAM. RTX 3060 also hosts Qwen3-TTS during normal LLM operation.
|
||||
|
||||
## What is and is not stale
|
||||
|
||||
The GPU workers use `restart: "no"` and are created once, then started on
|
||||
demand. A stopped `mike-ai-llama-*`, image, music, separator, OmniVoice or X-VC
|
||||
container is expected. A candidate is stale only after checking Compose,
|
||||
labels, mounts, router/controller references, model paths and test history.
|
||||
@@ -163,10 +163,10 @@ log "Tool-Container mit der neuen Stackdefinition aktivieren"
|
||||
log "Router und OpenWebUI mit den wiederhergestellten Schlüsseln neu erstellen"
|
||||
cd "$STACK_DIR"
|
||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --force-recreate \
|
||||
qwen3-tts piper tts-gateway router open-webui
|
||||
qwen3-tts tts-gateway router open-webui
|
||||
|
||||
deadline=$((SECONDS + 180))
|
||||
for container in mike-ai-qwen3-tts mike-ai-piper mike-ai-tts-gateway \
|
||||
for container in mike-ai-qwen3-tts mike-ai-tts-gateway \
|
||||
mike-ai-router mike-ai-open-webui; do
|
||||
until [[ $(docker inspect -f '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
|
||||
"$container" 2>/dev/null || true) == healthy ]]; do
|
||||
|
||||
@@ -83,7 +83,7 @@ with con:
|
||||
for key, value in (
|
||||
("task.follow_up.enable", False),
|
||||
("audio.tts.engine", "openai"),
|
||||
("audio.tts.model", "piper"),
|
||||
("audio.tts.model", "qwen3-tts"),
|
||||
("audio.tts.voice", "alloy"),
|
||||
("audio.tts.openai.api_base_url", "http://router:8081/v1"),
|
||||
("web.search.enable", True),
|
||||
|
||||
Executable
+36
@@ -0,0 +1,36 @@
|
||||
#!/usr/bin/env bash
|
||||
# Rebuild/create specialist containers after a bare-metal restore. GPU workers
|
||||
# remain stopped; the core installer leaves Athena safely in LLM mode.
|
||||
set -Eeuo pipefail
|
||||
|
||||
log() { printf '\n==> %s\n' "$*"; }
|
||||
|
||||
if [[ -r /etc/mike-ai/install.env ]]; then
|
||||
set -a
|
||||
# shellcheck disable=SC1091
|
||||
source /etc/mike-ai/install.env
|
||||
set +a
|
||||
fi
|
||||
export VOICE_GPU_UUID=${VOICE_GPU_UUID:-${IMAGE_GPU_DEVICES:-}}
|
||||
export ACESTEP_GPU_UUID=${ACESTEP_GPU_UUID:-${IMAGE_GPU_DEVICES:-}}
|
||||
export SEPARATOR_GPU_UUID=${SEPARATOR_GPU_UUID:-${IMAGE_GPU_DEVICES:-}}
|
||||
|
||||
create_project() {
|
||||
local dir=$1 file=${2:-compose.yaml} profile=${3:-}
|
||||
[[ -f $dir/$file ]] || { printf 'Übersprungen (fehlt): %s/%s\n' "$dir" "$file"; return 0; }
|
||||
log "Spezialprojekt vorbereiten: $dir"
|
||||
local args=(docker compose -f "$file")
|
||||
[[ -z $profile ]] || args+=(--profile "$profile")
|
||||
(cd "$dir" && "${args[@]}" pull --ignore-buildable && \
|
||||
"${args[@]}" build && "${args[@]}" create)
|
||||
}
|
||||
|
||||
docker network inspect mike-ai_frontend >/dev/null
|
||||
create_project /opt/mike-ai/acestep-test compose.yaml music-test
|
||||
create_project /opt/mike-ai/stem-separator
|
||||
create_project /opt/mike-ai/omnivoice-studio
|
||||
create_project /opt/mike-ai/xvc-studio
|
||||
create_project /opt/mike-ai/stack/experiments/applio-rvc
|
||||
create_project /opt/mike-ai/Mikes-Applio-UI compose.example.yaml
|
||||
|
||||
printf 'ATHENA_SPECIALISTS_REBUILT_OK\n'
|
||||
@@ -7,9 +7,9 @@ SERVICE="${LLAMA_SERVICE:-mike-ai-llama-ui.service}"
|
||||
PROFILE="${1:-}"
|
||||
|
||||
case "$PROFILE" in
|
||||
fast|medium|beta1|large|ultra|uncensored) ;;
|
||||
fast|medium|large|ultra|uncensored) ;;
|
||||
*)
|
||||
echo "Usage: llama-profile {fast|medium|beta1|large|ultra|uncensored}" >&2
|
||||
echo "Usage: llama-profile {fast|medium|large|ultra|uncensored}" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
Reference in New Issue
Block a user