Add X-VC voice conversion mode

This commit is contained in:
Mikei386
2026-09-09 16:14:10 +02:00
parent 17f1a08d7d
commit 87a2ae5704
15 changed files with 544 additions and 28 deletions
+58 -11
View File
@@ -111,6 +111,7 @@ ENABLE_MUSIC_MODE = os.environ.get(
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600"))
VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600"))
VOICE_CHANGE_START_TIMEOUT = float(os.environ.get("VOICE_CHANGE_START_TIMEOUT", "600"))
# Optional worker APIs. The clean Docker baseline deliberately ships only
# text/multimodal chat; absent workers must fail explicitly instead of trying
@@ -383,6 +384,27 @@ def _voice_worker_health() -> str:
return "unknown"
def _voice_change_worker_state() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_change_worker", "missing"))
except Exception as exc:
log.warning("Voice-Change-Worker-Status nicht verfügbar: %s", exc)
return "unknown"
def _voice_change_worker_health() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_change_health", "unknown"))
except Exception:
return "unknown"
def _wait_music_ready() -> None:
deadline = time.monotonic() + MUSIC_START_TIMEOUT
while time.monotonic() < deadline:
@@ -419,10 +441,24 @@ def _wait_voice_ready() -> None:
and status.get("voice_health") == "healthy"):
return
if status.get("voice_health") == "unhealthy":
raise RuntimeError("Vevo2-Container ist unhealthy")
raise RuntimeError("OmniVoice-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"Vevo2 nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
f"OmniVoice nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
def _wait_voice_change_ready() -> None:
deadline = time.monotonic() + VOICE_CHANGE_START_TIMEOUT
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get("voice_change_worker") == "running"
and status.get("voice_change_health") == "healthy"):
return
if status.get("voice_change_health") == "unhealthy":
raise RuntimeError("X-VC-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"X-VC nach {VOICE_CHANGE_START_TIMEOUT:.0f} s nicht bereit")
def _special_worker(mode: str) -> tuple[str, str, callable]:
@@ -432,6 +468,9 @@ def _special_worker(mode: str) -> tuple[str, str, callable]:
return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready
if mode == "voice":
return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready
if mode == "voicechange":
return ("/workers/voice-change/start", _voice_change_worker_state(),
_wait_voice_change_ready)
raise ValueError(f"unbekannter Spezialmodus: {mode}")
@@ -439,11 +478,11 @@ def set_operating_mode(mode: str) -> dict:
"""Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
raise ValueError("Modus muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
with STATE.lock:
STATE.mode_error = None
if mode in {"music", "separation", "voice"}:
if mode in {"music", "separation", "voice", "voicechange"}:
path, worker_state, wait_ready = _special_worker(mode)
if STATE.mode == mode and worker_state == "running":
return {"status": "ok", "mode": mode, "changed": False}
@@ -485,6 +524,7 @@ def set_operating_mode(mode: str) -> dict:
_profile_controller_request("POST", "/workers/music/stop")
_profile_controller_request("POST", "/workers/separator/stop")
_profile_controller_request("POST", "/workers/voice/stop")
_profile_controller_request("POST", "/workers/voice-change/stop")
_restore_qwen(profile)
STATE.mode = "llm"
STATE.mode_phase = "ready"
@@ -503,8 +543,8 @@ def schedule_operating_mode(mode: str) -> tuple[bool, str]:
"""Start a transition in the background so chat/UI acknowledgement is instant."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
raise ValueError("Modus muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
with STATE.lock:
if STATE.mode_phase not in {"ready", "error"}:
return False, STATE.mode_phase
@@ -541,6 +581,7 @@ def _control_command(data: dict, path: str) -> str | None:
return command if command in {"/athena music", "/athena stems",
"/athena separation", "/athena llm",
"/athena voice",
"/athena voicechange", "/athena changer",
"/athena status"} else None
@@ -2018,6 +2059,8 @@ class Handler(BaseHTTPRequestHandler):
"separator_health": _separator_worker_health(),
"voice_worker": _voice_worker_state(),
"voice_health": _voice_worker_health(),
"voice_change_worker": _voice_change_worker_state(),
"voice_change_health": _voice_change_worker_health(),
"return_profile": state.get("return_profile"),
"last_error": STATE.mode_error,
"enabled": ENABLE_MUSIC_MODE,
@@ -2027,8 +2070,8 @@ class Handler(BaseHTTPRequestHandler):
try:
data = json.loads(self._read_body() or b"{}")
mode = data.get("mode") if isinstance(data, dict) else None
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation' oder 'voice' sein")
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
started, phase = schedule_operating_mode(mode)
self._send_json(202 if started else 200, {
"status": "accepted" if started else "ok",
@@ -2692,11 +2735,13 @@ class Handler(BaseHTTPRequestHandler):
f"Phase: {mode['phase']}. Musik-Worker: "
f"{mode['music_worker']}. Stem-Separator: "
f"{mode['separator_worker']}. Voice Studio: "
f"{mode['voice_worker']}. LLM-Profil: {profile or 'entladen'}.")
f"{mode['voice_worker']}. Voice Changer: "
f"{mode['voice_change_worker']}. LLM-Profil: {profile or 'entladen'}.")
else:
target = ("music" if command == "/athena music" else
"separation" if command in {"/athena stems", "/athena separation"}
else "voice" if command == "/athena voice"
else "voicechange" if command in {"/athena voicechange", "/athena changer"}
else "llm")
try:
started, phase = schedule_operating_mode(target)
@@ -2707,6 +2752,8 @@ class Handler(BaseHTTPRequestHandler):
if target == "separation" else
"Voice Studio wird gestartet. LLM und TTS werden entladen."
if target == "voice" else
"Voice Changer wird gestartet. LLM und TTS werden entladen."
if target == "voicechange" else
"Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.")
else:
text = (f"Athena ist bereits im {target.upper()}-Modus "
@@ -2999,7 +3046,7 @@ def _startup_reconcile() -> None:
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
special_mode = previous.get("mode")
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice"}:
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice", "voicechange"}:
STATE.mode = special_mode
STATE.mode_phase = f"starting-{special_mode}"
_set_qwen_unavailable(True)