Document OmniVoice cloning gate
This commit is contained in:
@@ -56,7 +56,8 @@ Titelgenerierung und Kontextkompression in Hermes.
|
|||||||
| 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** |
|
| 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** |
|
||||||
| 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** mit Piper-Fallback |
|
| 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** mit Piper-Fallback |
|
||||||
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
|
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
|
||||||
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Produktions-HTTP-API mit offizieller 8,6-s-Sprachprobe: Warmstart 12,216 s, Wandlung 2,342 s, Spitzen-VRAM 5.605,6 MiB; gültiges 24-kHz-WAV mit 412.878 Byte; Profilanlage, Download und automatische Job-Bereinigung geprüft | **technischer Ende-zu-Ende-Test bestanden**, privater Voice-Studio-Betatest; Gewichte CC BY-NC-ND 4.0 und daher nicht kommerziell einsetzen |
|
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen**; Container ersetzt, Image und Daten vorerst nur als Rollback erhalten |
|
||||||
|
| 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen |
|
||||||
|
|
||||||
## Musikgenerierung
|
## Musikgenerierung
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,28 @@
|
|||||||
|
FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04
|
||||||
|
|
||||||
|
ARG OMNIVOICE_VERSION=0.2.1
|
||||||
|
ARG OMNIVOICE_TRITON_VERSION=0.1.0
|
||||||
|
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install -y --no-install-recommends \
|
||||||
|
ca-certificates curl ffmpeg libsndfile1 python3 python3-pip python3-venv \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
RUN python3 -m venv /opt/venv
|
||||||
|
ENV PATH="/opt/venv/bin:${PATH}"
|
||||||
|
|
||||||
|
RUN python -m pip install --no-cache-dir --upgrade pip \
|
||||||
|
&& python -m pip install --no-cache-dir \
|
||||||
|
--index-url https://download.pytorch.org/whl/cu128 \
|
||||||
|
"torch==2.8.0" "torchaudio==2.8.0" \
|
||||||
|
&& python -m pip install --no-cache-dir \
|
||||||
|
"omnivoice==${OMNIVOICE_VERSION}" \
|
||||||
|
"omnivoice-triton==${OMNIVOICE_TRITON_VERSION}" \
|
||||||
|
"num2words>=0.5.14"
|
||||||
|
|
||||||
|
EXPOSE 8008
|
||||||
|
|
||||||
|
HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \
|
||||||
|
CMD curl -fsS http://127.0.0.1:8008/ >/dev/null || exit 1
|
||||||
|
|
||||||
|
CMD ["omnivoice-demo", "--ip", "0.0.0.0", "--port", "8008"]
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
# OmniVoice cloning gate on Athena
|
||||||
|
|
||||||
|
This is the isolated quality gate for `k2-fsa/OmniVoice` 0.2.1. It exposes
|
||||||
|
the upstream Gradio demo through Athena's existing private Voice Studio route.
|
||||||
|
The image also contains `omnivoice-triton` 0.1.0 for a later measured
|
||||||
|
base-versus-optimized benchmark; the upstream UI deliberately starts in the
|
||||||
|
unmodified reference mode so kernel changes cannot contaminate the first
|
||||||
|
listening test.
|
||||||
|
|
||||||
|
- Model weights: CC-BY-NC
|
||||||
|
- Code: Apache-2.0
|
||||||
|
- Private URL: `http://192.168.1.212:8008`
|
||||||
|
- Persistent cache: `/data/voice/omnivoice/huggingface`
|
||||||
|
- GPU allocator workaround: `expandable_segments:True`
|
||||||
|
|
||||||
|
Use a clean 3–10 second reference and provide its exact transcript. German
|
||||||
|
target text should be written out normally; avoid raw abbreviations and digits
|
||||||
|
in the first quality test.
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
services:
|
||||||
|
voice-studio:
|
||||||
|
build: .
|
||||||
|
image: mike-ai/omnivoice-studio:0.2.1
|
||||||
|
container_name: mike-ai-voice-studio
|
||||||
|
restart: "no"
|
||||||
|
labels:
|
||||||
|
# Kept compatible with the current controller during the A/B gate.
|
||||||
|
com.mike-ai.voice-worker: vevo2
|
||||||
|
environment:
|
||||||
|
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
|
||||||
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
|
HF_HOME: /models/huggingface
|
||||||
|
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8008:8008"
|
||||||
|
volumes:
|
||||||
|
- /data/voice/omnivoice/huggingface:/models/huggingface
|
||||||
|
- /data/voice/omnivoice/output:/output
|
||||||
|
healthcheck:
|
||||||
|
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/ >/dev/null"]
|
||||||
|
interval: 5s
|
||||||
|
timeout: 3s
|
||||||
|
start_period: 600s
|
||||||
|
retries: 3
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
reservations:
|
||||||
|
devices:
|
||||||
|
- driver: nvidia
|
||||||
|
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
|
||||||
|
capabilities: [gpu]
|
||||||
|
networks:
|
||||||
|
frontend:
|
||||||
|
aliases: [voice-studio]
|
||||||
|
|
||||||
|
networks:
|
||||||
|
frontend:
|
||||||
|
name: mike-ai_frontend
|
||||||
|
external: true
|
||||||
Reference in New Issue
Block a user