From 17f1a08d7dd592980a84bbbaa48e293156286ffc Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Wed, 9 Sep 2026 15:15:10 +0200 Subject: [PATCH] Document OmniVoice cloning gate --- docs/TESTED_MODELS.md | 3 +- experiments/omnivoice-cloning/Dockerfile | 28 +++++++++++++++ experiments/omnivoice-cloning/README.md | 18 ++++++++++ experiments/omnivoice-cloning/compose.yaml | 40 ++++++++++++++++++++++ 4 files changed, 88 insertions(+), 1 deletion(-) create mode 100644 experiments/omnivoice-cloning/Dockerfile create mode 100644 experiments/omnivoice-cloning/README.md create mode 100644 experiments/omnivoice-cloning/compose.yaml diff --git a/docs/TESTED_MODELS.md b/docs/TESTED_MODELS.md index 01a148e..2d56ebd 100644 --- a/docs/TESTED_MODELS.md +++ b/docs/TESTED_MODELS.md @@ -56,7 +56,8 @@ Titelgenerierung und Kontextkompression in Hermes. | 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** | | 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** mit Piper-Fallback | | 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** | -| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Produktions-HTTP-API mit offizieller 8,6-s-Sprachprobe: Warmstart 12,216 s, Wandlung 2,342 s, Spitzen-VRAM 5.605,6 MiB; gültiges 24-kHz-WAV mit 412.878 Byte; Profilanlage, Download und automatische Job-Bereinigung geprüft | **technischer Ende-zu-Ende-Test bestanden**, privater Voice-Studio-Betatest; Gewichte CC BY-NC-ND 4.0 und daher nicht kommerziell einsetzen | +| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen**; Container ersetzt, Image und Daten vorerst nur als Rollback erhalten | +| 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen | ## Musikgenerierung diff --git a/experiments/omnivoice-cloning/Dockerfile b/experiments/omnivoice-cloning/Dockerfile new file mode 100644 index 0000000..349ceb3 --- /dev/null +++ b/experiments/omnivoice-cloning/Dockerfile @@ -0,0 +1,28 @@ +FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04 + +ARG OMNIVOICE_VERSION=0.2.1 +ARG OMNIVOICE_TRITON_VERSION=0.1.0 + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + ca-certificates curl ffmpeg libsndfile1 python3 python3-pip python3-venv \ + && rm -rf /var/lib/apt/lists/* + +RUN python3 -m venv /opt/venv +ENV PATH="/opt/venv/bin:${PATH}" + +RUN python -m pip install --no-cache-dir --upgrade pip \ + && python -m pip install --no-cache-dir \ + --index-url https://download.pytorch.org/whl/cu128 \ + "torch==2.8.0" "torchaudio==2.8.0" \ + && python -m pip install --no-cache-dir \ + "omnivoice==${OMNIVOICE_VERSION}" \ + "omnivoice-triton==${OMNIVOICE_TRITON_VERSION}" \ + "num2words>=0.5.14" + +EXPOSE 8008 + +HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \ + CMD curl -fsS http://127.0.0.1:8008/ >/dev/null || exit 1 + +CMD ["omnivoice-demo", "--ip", "0.0.0.0", "--port", "8008"] diff --git a/experiments/omnivoice-cloning/README.md b/experiments/omnivoice-cloning/README.md new file mode 100644 index 0000000..0b74e64 --- /dev/null +++ b/experiments/omnivoice-cloning/README.md @@ -0,0 +1,18 @@ +# OmniVoice cloning gate on Athena + +This is the isolated quality gate for `k2-fsa/OmniVoice` 0.2.1. It exposes +the upstream Gradio demo through Athena's existing private Voice Studio route. +The image also contains `omnivoice-triton` 0.1.0 for a later measured +base-versus-optimized benchmark; the upstream UI deliberately starts in the +unmodified reference mode so kernel changes cannot contaminate the first +listening test. + +- Model weights: CC-BY-NC +- Code: Apache-2.0 +- Private URL: `http://192.168.1.212:8008` +- Persistent cache: `/data/voice/omnivoice/huggingface` +- GPU allocator workaround: `expandable_segments:True` + +Use a clean 3–10 second reference and provide its exact transcript. German +target text should be written out normally; avoid raw abbreviations and digits +in the first quality test. diff --git a/experiments/omnivoice-cloning/compose.yaml b/experiments/omnivoice-cloning/compose.yaml new file mode 100644 index 0000000..015036b --- /dev/null +++ b/experiments/omnivoice-cloning/compose.yaml @@ -0,0 +1,40 @@ +services: + voice-studio: + build: . + image: mike-ai/omnivoice-studio:0.2.1 + container_name: mike-ai-voice-studio + restart: "no" + labels: + # Kept compatible with the current controller during the A/B gate. + com.mike-ai.voice-worker: vevo2 + environment: + NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID} + NVIDIA_DRIVER_CAPABILITIES: compute,utility + HF_HOME: /models/huggingface + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + ports: + - "127.0.0.1:8008:8008" + volumes: + - /data/voice/omnivoice/huggingface:/models/huggingface + - /data/voice/omnivoice/output:/output + healthcheck: + test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/ >/dev/null"] + interval: 5s + timeout: 3s + start_period: 600s + retries: 3 + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"] + capabilities: [gpu] + networks: + frontend: + aliases: [voice-studio] + +networks: + frontend: + name: mike-ai_frontend + external: true