Document OmniVoice cloning gate
This commit is contained in:
1 parent
68d02f32bd
commit
17f1a08d7d
4 files changed
+88
-1
No files matched your search
@@ -56,7 +56,8 @@ Titelgenerierung und Kontextkompression in Hermes.
|
||||
| 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** |
|
||||
| 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** mit Piper-Fallback |
|
||||
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
|
||||
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Produktions-HTTP-API mit offizieller 8,6-s-Sprachprobe: Warmstart 12,216 s, Wandlung 2,342 s, Spitzen-VRAM 5.605,6 MiB; gültiges 24-kHz-WAV mit 412.878 Byte; Profilanlage, Download und automatische Job-Bereinigung geprüft | **technischer Ende-zu-Ende-Test bestanden**, privater Voice-Studio-Betatest; Gewichte CC BY-NC-ND 4.0 und daher nicht kommerziell einsetzen |
|
||||
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen**; Container ersetzt, Image und Daten vorerst nur als Rollback erhalten |
|
||||
| 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen |
|
||||
|
||||
## Musikgenerierung
|
||||
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04
|
||||
|
||||
ARG OMNIVOICE_VERSION=0.2.1
|
||||
ARG OMNIVOICE_TRITON_VERSION=0.1.0
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl ffmpeg libsndfile1 python3 python3-pip python3-venv \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN python3 -m venv /opt/venv
|
||||
ENV PATH="/opt/venv/bin:${PATH}"
|
||||
|
||||
RUN python -m pip install --no-cache-dir --upgrade pip \
|
||||
&& python -m pip install --no-cache-dir \
|
||||
--index-url https://download.pytorch.org/whl/cu128 \
|
||||
"torch==2.8.0" "torchaudio==2.8.0" \
|
||||
&& python -m pip install --no-cache-dir \
|
||||
"omnivoice==${OMNIVOICE_VERSION}" \
|
||||
"omnivoice-triton==${OMNIVOICE_TRITON_VERSION}" \
|
||||
"num2words>=0.5.14"
|
||||
|
||||
EXPOSE 8008
|
||||
|
||||
HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \
|
||||
CMD curl -fsS http://127.0.0.1:8008/ >/dev/null || exit 1
|
||||
|
||||
CMD ["omnivoice-demo", "--ip", "0.0.0.0", "--port", "8008"]
|
||||
@@ -0,0 +1,18 @@
|
||||
# OmniVoice cloning gate on Athena
|
||||
|
||||
This is the isolated quality gate for `k2-fsa/OmniVoice` 0.2.1. It exposes
|
||||
the upstream Gradio demo through Athena's existing private Voice Studio route.
|
||||
The image also contains `omnivoice-triton` 0.1.0 for a later measured
|
||||
base-versus-optimized benchmark; the upstream UI deliberately starts in the
|
||||
unmodified reference mode so kernel changes cannot contaminate the first
|
||||
listening test.
|
||||
|
||||
- Model weights: CC-BY-NC
|
||||
- Code: Apache-2.0
|
||||
- Private URL: `http://192.168.1.212:8008`
|
||||
- Persistent cache: `/data/voice/omnivoice/huggingface`
|
||||
- GPU allocator workaround: `expandable_segments:True`
|
||||
|
||||
Use a clean 3–10 second reference and provide its exact transcript. German
|
||||
target text should be written out normally; avoid raw abbreviations and digits
|
||||
in the first quality test.
|
||||
@@ -0,0 +1,40 @@
|
||||
services:
|
||||
voice-studio:
|
||||
build: .
|
||||
image: mike-ai/omnivoice-studio:0.2.1
|
||||
container_name: mike-ai-voice-studio
|
||||
restart: "no"
|
||||
labels:
|
||||
# Kept compatible with the current controller during the A/B gate.
|
||||
com.mike-ai.voice-worker: vevo2
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
HF_HOME: /models/huggingface
|
||||
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
|
||||
ports:
|
||||
- "127.0.0.1:8008:8008"
|
||||
volumes:
|
||||
- /data/voice/omnivoice/huggingface:/models/huggingface
|
||||
- /data/voice/omnivoice/output:/output
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/ >/dev/null"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
start_period: 600s
|
||||
retries: 3
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
|
||||
capabilities: [gpu]
|
||||
networks:
|
||||
frontend:
|
||||
aliases: [voice-studio]
|
||||
|
||||
networks:
|
||||
frontend:
|
||||
name: mike-ai_frontend
|
||||
external: true
|
||||
Reference in new issue
Block a user