Synchronize Athena operating modes with live deployment
This commit is contained in:
1 parent
9fe51390f6
commit
040a2df48b
5 files changed
+1349
-198
No files matched your search
+645
-40
@@ -2,31 +2,30 @@
|
||||
"""AI Profile Router – OpenAI-kompatibler Proxy vor llama.cpp.
|
||||
|
||||
Leitet OpenAI-kompatible Requests transparent an den lokalen llama.cpp-Server
|
||||
weiter (Streaming, Tool Calls, JSON) und schaltet zwischen sechs festen
|
||||
weiter (Streaming, Tool Calls, JSON) und schaltet zwischen fünf festen
|
||||
Profilen um:
|
||||
|
||||
Profil Kontext
|
||||
------ --------
|
||||
fast 76800
|
||||
medium 160000
|
||||
beta1 192000
|
||||
large 192000
|
||||
ultra 262144
|
||||
uncensored 80000
|
||||
|
||||
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-beta-1, qwen-large,
|
||||
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large,
|
||||
qwen-ultra, qwen-uncensored
|
||||
Kommandos: POST /fast, /medium, /beta1, /large, /ultra,
|
||||
Kommandos: POST /fast, /medium, /large, /ultra,
|
||||
/uncensored
|
||||
GET /status (Zustand)
|
||||
|
||||
Bildgenerierung und Editing (FLUX.2-klein-4B):
|
||||
Bildgenerierung und Editing (FLUX.2 Klein 9B FP8 beta):
|
||||
POST /v1/images/generations (OpenAI-kompatibel)
|
||||
POST /v1/images/edits (lokal, Referenzbilder)
|
||||
GET /images (Liste)
|
||||
GET /images/<datei> (PNG-Download)
|
||||
|
||||
Sprachausgabe (XTTS-v2, multilingual, CPU-only):
|
||||
Sprachausgabe (Qwen3-TTS auf RTX 3060):
|
||||
POST /v1/audio/speech (OpenAI-kompatibel)
|
||||
GET /v1/audio/voices (verfügbare Stimmen)
|
||||
|
||||
@@ -40,7 +39,8 @@ Der Router leitet /v1/audio/speech und /v1/audio/transcriptions
|
||||
per HTTP an die Worker weiter.
|
||||
|
||||
Der Router agiert als Modell-Orchestrator: vor der Generierung wird
|
||||
llama.cpp gestoppt, der Bild-Worker lädt FLUX.2, generiert/bearbeitet und entlädt
|
||||
llama.cpp und Qwen3-TTS gestoppt, der Bild-Worker lädt FLUX.2 und den
|
||||
Text-Encoder auf getrennte GPUs, generiert/bearbeitet und entlädt
|
||||
das Modell wieder; danach wird das vorherige Qwen-Profil wiederher-
|
||||
gestellt und erst dann geantwortet (try/finally – Qwen wird auch bei
|
||||
Fehlgeschlagener Generierung wiederhergestellt).
|
||||
@@ -105,6 +105,15 @@ PROFILE_DIR = os.environ.get(
|
||||
PROFILE_CONTROL_URL = os.environ.get("PROFILE_CONTROL_URL", "").rstrip("/")
|
||||
PROFILE_CONTROL_TOKEN_FILE = os.environ.get(
|
||||
"PROFILE_CONTROL_TOKEN_FILE", "/run/secrets/controller-token")
|
||||
ENABLE_MUSIC_MODE = os.environ.get(
|
||||
"ENABLE_MUSIC_MODE", "false").lower() in {"1", "true", "yes"}
|
||||
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
|
||||
YUE2_START_TIMEOUT = float(os.environ.get("YUE2_START_TIMEOUT", "600"))
|
||||
SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600"))
|
||||
VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600"))
|
||||
VOICE_CHANGE_START_TIMEOUT = float(os.environ.get("VOICE_CHANGE_START_TIMEOUT", "600"))
|
||||
APPLIO_START_TIMEOUT = float(os.environ.get("APPLIO_START_TIMEOUT", "900"))
|
||||
TRELLIS_START_TIMEOUT = float(os.environ.get("TRELLIS_START_TIMEOUT", "900"))
|
||||
|
||||
# Optional worker APIs. The clean Docker baseline deliberately ships only
|
||||
# text/multimodal chat; absent workers must fail explicitly instead of trying
|
||||
@@ -126,7 +135,7 @@ DEFAULT_REASONING_EFFORT = os.environ.get(
|
||||
GLOBAL_SYSTEM_POLICY_FILE = os.environ.get(
|
||||
"GLOBAL_SYSTEM_POLICY_FILE", "").strip()
|
||||
|
||||
# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 4B) ---
|
||||
# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 9B FP8) ---
|
||||
LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service")
|
||||
SYSTEMCTL_BIN = os.environ.get("SYSTEMCTL_BIN", "systemctl")
|
||||
IMAGE_WORKER = os.environ.get(
|
||||
@@ -135,7 +144,8 @@ IMAGE_PYTHON = os.environ.get(
|
||||
"IMAGE_PYTHON", "/opt/mike-ai/ai-profile-router/venv/bin/python")
|
||||
IMAGE_WORKER_URL = os.environ.get("IMAGE_WORKER_URL", "").rstrip("/")
|
||||
IMAGE_WORKER_TOKEN = os.environ.get("IMAGE_WORKER_TOKEN", "").strip()
|
||||
IMAGE_MODEL_NAME = os.environ.get("IMAGE_MODEL_NAME", "FLUX.2-klein-4B")
|
||||
IMAGE_MODEL_NAME = os.environ.get(
|
||||
"IMAGE_MODEL_NAME", "FLUX.2-klein-9B-fp8-beta")
|
||||
IMAGE_DIR = os.environ.get(
|
||||
"IMAGE_DIR", "/opt/mike-ai/ai-profile-router/images")
|
||||
IMAGE_WORKER_LOG = os.environ.get(
|
||||
@@ -163,20 +173,20 @@ IMAGE_SIZES = {
|
||||
"1920x1088": (1920, 1088),
|
||||
"1088x1920": (1088, 1920),
|
||||
}
|
||||
# Das destillierte FLUX.2-klein-4B ist auf vier Schritte ausgelegt.
|
||||
# Das destillierte FLUX.2 Klein 9B ist auf vier Schritte ausgelegt.
|
||||
IMAGE_QUALITY = {"standard": 4, "high": 4}
|
||||
IMAGE_DEFAULT_QUALITY = "standard"
|
||||
IMAGE_MAX_N = 4
|
||||
|
||||
# --- Sprachausgabe (austauschbarer interner TTS-Worker, CPU-only) ---
|
||||
# --- Sprachausgabe (Qwen3-TTS über das interne Normalisierungs-Gateway) ---
|
||||
TTS_WORKER_URL = os.environ.get("TTS_WORKER_URL", "http://127.0.0.1:8085")
|
||||
TTS_TIMEOUT = float(os.environ.get("TTS_TIMEOUT", "300")) # s, pro Synthese
|
||||
TTS_CONNECT_TIMEOUT = float(os.environ.get("TTS_CONNECT_TIMEOUT", "5"))
|
||||
TTS_MODEL = os.environ.get("TTS_MODEL", "xtts-v2")
|
||||
TTS_MODEL = os.environ.get("TTS_MODEL", "qwen3-tts")
|
||||
TTS_VOICES = tuple(v.strip() for v in os.environ.get(
|
||||
"TTS_VOICES", "claribel").split(",") if v.strip())
|
||||
"TTS_VOICES", "alloy").split(",") if v.strip())
|
||||
TTS_DEFAULT_VOICE = os.environ.get(
|
||||
"TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "claribel")
|
||||
"TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "alloy")
|
||||
TTS_FORMATS = ("mp3", "wav", "pcm")
|
||||
TTS_DEFAULT_FORMAT = "mp3"
|
||||
|
||||
@@ -254,6 +264,7 @@ class _ImageState:
|
||||
self.last_error: str | None = None
|
||||
self.last_image: str | None = None
|
||||
self.last_seconds: float | None = None
|
||||
self.current_model: str | None = None
|
||||
|
||||
|
||||
IMAGE_PHASES = (
|
||||
@@ -279,6 +290,9 @@ class _State:
|
||||
self.qwen_unavailable = True
|
||||
self.active_chats = 0
|
||||
self.avail_lock = threading.Lock()
|
||||
self.mode = "llm"
|
||||
self.mode_phase = "ready"
|
||||
self.mode_error: str | None = None
|
||||
|
||||
|
||||
STATE = _State()
|
||||
@@ -309,6 +323,329 @@ def _set_qwen_unavailable(unavailable: bool) -> None:
|
||||
STATE.qwen_unavailable = unavailable
|
||||
|
||||
|
||||
def _music_worker_state() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"music_worker", "missing"))
|
||||
except Exception as exc:
|
||||
log.warning("Musik-Worker-Status nicht verfügbar: %s", exc)
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _music_worker_health() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"music_health", "unknown"))
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _separator_worker_state() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"separator_worker", "missing"))
|
||||
except Exception as exc:
|
||||
log.warning("Stem-Separator-Status nicht verfügbar: %s", exc)
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _separator_worker_health() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"separator_health", "unknown"))
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _voice_worker_state() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"voice_worker", "missing"))
|
||||
except Exception as exc:
|
||||
log.warning("Voice-Worker-Status nicht verfügbar: %s", exc)
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _voice_worker_health() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"voice_health", "unknown"))
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _voice_change_worker_state() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"voice_change_worker", "missing"))
|
||||
except Exception as exc:
|
||||
log.warning("Voice-Change-Worker-Status nicht verfügbar: %s", exc)
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _voice_change_worker_health() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"voice_change_health", "unknown"))
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _worker_field(field: str) -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(field, "missing"))
|
||||
except Exception as exc:
|
||||
log.warning("Spezial-Worker-Status %s nicht verfügbar: %s", field, exc)
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _wait_music_ready() -> None:
|
||||
deadline = time.monotonic() + MUSIC_START_TIMEOUT
|
||||
while time.monotonic() < deadline:
|
||||
status = _profile_controller_request("GET", "/status")
|
||||
if (status.get("music_worker") == "running"
|
||||
and status.get("music_health") == "healthy"):
|
||||
return
|
||||
if status.get("music_health") == "unhealthy":
|
||||
raise RuntimeError("ACE-Step-Container ist unhealthy")
|
||||
time.sleep(POLL_INTERVAL)
|
||||
raise RuntimeError(
|
||||
f"ACE-Step nach {MUSIC_START_TIMEOUT:.0f} s nicht bereit")
|
||||
|
||||
|
||||
def _wait_yue2_ready() -> None:
|
||||
_wait_aux_voice_ready("yue2_worker", "yue2_health",
|
||||
"YuE2", YUE2_START_TIMEOUT)
|
||||
|
||||
|
||||
def _wait_separator_ready() -> None:
|
||||
deadline = time.monotonic() + SEPARATOR_START_TIMEOUT
|
||||
while time.monotonic() < deadline:
|
||||
status = _profile_controller_request("GET", "/status")
|
||||
if (status.get("separator_worker") == "running"
|
||||
and status.get("separator_health") == "healthy"):
|
||||
return
|
||||
if status.get("separator_health") == "unhealthy":
|
||||
raise RuntimeError("BS-RoFormer-Container ist unhealthy")
|
||||
time.sleep(POLL_INTERVAL)
|
||||
raise RuntimeError(
|
||||
f"BS-RoFormer nach {SEPARATOR_START_TIMEOUT:.0f} s nicht bereit")
|
||||
|
||||
|
||||
def _wait_voice_ready() -> None:
|
||||
deadline = time.monotonic() + VOICE_START_TIMEOUT
|
||||
while time.monotonic() < deadline:
|
||||
status = _profile_controller_request("GET", "/status")
|
||||
if (status.get("voice_worker") == "running"
|
||||
and status.get("voice_health") == "healthy"):
|
||||
return
|
||||
if status.get("voice_health") == "unhealthy":
|
||||
raise RuntimeError("OmniVoice-Container ist unhealthy")
|
||||
time.sleep(POLL_INTERVAL)
|
||||
raise RuntimeError(
|
||||
f"OmniVoice nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
|
||||
|
||||
|
||||
def _wait_voice_change_ready() -> None:
|
||||
deadline = time.monotonic() + VOICE_CHANGE_START_TIMEOUT
|
||||
while time.monotonic() < deadline:
|
||||
status = _profile_controller_request("GET", "/status")
|
||||
if (status.get("voice_change_worker") == "running"
|
||||
and status.get("voice_change_health") == "healthy"):
|
||||
return
|
||||
if status.get("voice_change_health") == "unhealthy":
|
||||
raise RuntimeError("X-VC-Container ist unhealthy")
|
||||
time.sleep(POLL_INTERVAL)
|
||||
raise RuntimeError(
|
||||
f"X-VC nach {VOICE_CHANGE_START_TIMEOUT:.0f} s nicht bereit")
|
||||
|
||||
|
||||
def _wait_aux_voice_ready(worker_field: str, health_field: str,
|
||||
label: str, timeout: float) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
status = _profile_controller_request("GET", "/status")
|
||||
if (status.get(worker_field) == "running"
|
||||
and status.get(health_field) == "healthy"):
|
||||
return
|
||||
if status.get(health_field) == "unhealthy":
|
||||
raise RuntimeError(f"{label}-Container ist unhealthy")
|
||||
time.sleep(POLL_INTERVAL)
|
||||
raise RuntimeError(f"{label} nach {timeout:.0f} s nicht bereit")
|
||||
|
||||
|
||||
def _wait_applio_ready() -> None:
|
||||
_wait_aux_voice_ready("applio_worker", "applio_health",
|
||||
"Applio", APPLIO_START_TIMEOUT)
|
||||
|
||||
|
||||
def _wait_trellis_ready() -> None:
|
||||
_wait_aux_voice_ready("trellis_worker", "trellis_health",
|
||||
"TRELLIS.2", TRELLIS_START_TIMEOUT)
|
||||
|
||||
|
||||
def _special_worker(mode: str) -> tuple[str, str, callable]:
|
||||
if mode == "music":
|
||||
return "/workers/music/start", _music_worker_state(), _wait_music_ready
|
||||
if mode == "yue2":
|
||||
return ("/workers/yue2/start", _worker_field("yue2_worker"),
|
||||
_wait_yue2_ready)
|
||||
if mode == "separation":
|
||||
return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready
|
||||
if mode == "voice":
|
||||
return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready
|
||||
if mode == "voicechange":
|
||||
return ("/workers/voice-change/start", _voice_change_worker_state(),
|
||||
_wait_voice_change_ready)
|
||||
if mode == "applio":
|
||||
return ("/workers/applio/start", _worker_field("applio_worker"),
|
||||
_wait_applio_ready)
|
||||
if mode == "trellis":
|
||||
return ("/workers/trellis/start", _worker_field("trellis_worker"),
|
||||
_wait_trellis_ready)
|
||||
raise ValueError(f"unbekannter Spezialmodus: {mode}")
|
||||
|
||||
|
||||
def set_operating_mode(mode: str) -> dict:
|
||||
"""Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen."""
|
||||
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
|
||||
raise RuntimeError("Musikmodus ist nicht konfiguriert")
|
||||
special_modes = {"music", "yue2", "separation", "voice", "voicechange",
|
||||
"applio", "trellis"}
|
||||
if mode not in {"llm", *special_modes}:
|
||||
raise ValueError("unbekannter Betriebsmodus")
|
||||
with STATE.lock:
|
||||
STATE.mode_error = None
|
||||
if mode in special_modes:
|
||||
path, worker_state, wait_ready = _special_worker(mode)
|
||||
if STATE.mode == mode and worker_state == "running":
|
||||
return {"status": "ok", "mode": mode, "changed": False}
|
||||
profile = current_profile()
|
||||
previous = RUNTIME.load()
|
||||
saved = previous.get("return_profile") or previous.get("last_profile")
|
||||
return_profile = profile if profile in PROFILES else saved
|
||||
if return_profile not in PROFILES:
|
||||
return_profile = next(iter(PROFILES))
|
||||
STATE.mode_phase = f"starting-{mode}"
|
||||
_set_qwen_unavailable(True)
|
||||
try:
|
||||
# Persist intent before stopping anything so a router restart
|
||||
# during ACE-Step loading can resume the same transition.
|
||||
RUNTIME.save(mode=mode, return_profile=return_profile,
|
||||
last_profile=return_profile,
|
||||
phase=f"starting-{mode}")
|
||||
_wait_chats_drained()
|
||||
_profile_controller_request(
|
||||
"POST", path,
|
||||
timeout=120)
|
||||
wait_ready()
|
||||
STATE.mode = mode
|
||||
STATE.mode_phase = "ready"
|
||||
RUNTIME.save(mode=mode, return_profile=return_profile,
|
||||
last_profile=return_profile, phase=mode)
|
||||
return {"status": "ok", "mode": mode, "changed": True,
|
||||
"return_profile": return_profile}
|
||||
except Exception as exc:
|
||||
STATE.mode_error = str(exc)
|
||||
STATE.mode_phase = "error"
|
||||
raise
|
||||
|
||||
previous = RUNTIME.load()
|
||||
profile = previous.get("return_profile") or previous.get("last_profile")
|
||||
if profile not in PROFILES:
|
||||
profile = next(iter(PROFILES))
|
||||
STATE.mode_phase = "restoring-llm"
|
||||
_set_qwen_unavailable(True)
|
||||
try:
|
||||
_profile_controller_request("POST", "/workers/music/stop")
|
||||
_profile_controller_request("POST", "/workers/yue2/stop")
|
||||
_profile_controller_request("POST", "/workers/separator/stop")
|
||||
_profile_controller_request("POST", "/workers/voice/stop")
|
||||
_profile_controller_request("POST", "/workers/voice-change/stop")
|
||||
_profile_controller_request("POST", "/workers/applio/stop")
|
||||
_profile_controller_request("POST", "/workers/trellis/stop")
|
||||
_restore_qwen(profile)
|
||||
STATE.mode = "llm"
|
||||
STATE.mode_phase = "ready"
|
||||
_set_qwen_unavailable(False)
|
||||
RUNTIME.save(mode="llm", return_profile=None,
|
||||
last_profile=profile, phase="idle")
|
||||
return {"status": "ok", "mode": "llm", "changed": True,
|
||||
"profile": profile}
|
||||
except Exception as exc:
|
||||
STATE.mode_error = str(exc)
|
||||
STATE.mode_phase = "error"
|
||||
raise
|
||||
|
||||
|
||||
def schedule_operating_mode(mode: str) -> tuple[bool, str]:
|
||||
"""Start a transition in the background so chat/UI acknowledgement is instant."""
|
||||
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
|
||||
raise RuntimeError("Musikmodus ist nicht konfiguriert")
|
||||
if mode not in {"llm", "music", "yue2", "separation", "voice",
|
||||
"voicechange", "applio", "trellis"}:
|
||||
raise ValueError("unbekannter Betriebsmodus")
|
||||
with STATE.lock:
|
||||
if STATE.mode_phase not in {"ready", "error"}:
|
||||
return False, STATE.mode_phase
|
||||
if STATE.mode == mode and STATE.mode_phase == "ready":
|
||||
return False, "ready"
|
||||
STATE.mode_phase = f"starting-{mode}" if mode != "llm" else "restoring-llm"
|
||||
|
||||
def transition() -> None:
|
||||
try:
|
||||
set_operating_mode(mode)
|
||||
log.info("Betriebsmodus ist jetzt %s", mode)
|
||||
except Exception:
|
||||
log.exception("Betriebsmodus-Wechsel zu %s fehlgeschlagen", mode)
|
||||
|
||||
threading.Thread(target=transition, name=f"mode-{mode}", daemon=True).start()
|
||||
return True, STATE.mode_phase
|
||||
|
||||
|
||||
def _control_command(data: dict, path: str) -> str | None:
|
||||
"""Recognise exact local commands without invoking an LLM."""
|
||||
text: object = None
|
||||
if path == "/v1/chat/completions":
|
||||
messages = data.get("messages")
|
||||
if isinstance(messages, list):
|
||||
for item in reversed(messages):
|
||||
if isinstance(item, dict) and item.get("role") == "user":
|
||||
text = item.get("content")
|
||||
break
|
||||
elif path == "/v1/responses":
|
||||
text = data.get("input")
|
||||
if not isinstance(text, str):
|
||||
return None
|
||||
command = text.strip().casefold()
|
||||
return command if command in {"/athena music", "/athena yue2",
|
||||
"/athena stems",
|
||||
"/athena separation", "/athena llm",
|
||||
"/athena voice",
|
||||
"/athena voicechange", "/athena changer",
|
||||
"/athena applio",
|
||||
"/athena 3d", "/athena trellis",
|
||||
"/athena status"} else None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Upstream (llama.cpp)
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -588,7 +925,8 @@ def _read(path: str) -> str:
|
||||
return f.read().strip()
|
||||
|
||||
|
||||
def _profile_controller_request(method: str, path: str) -> dict:
|
||||
def _profile_controller_request(method: str, path: str,
|
||||
timeout: float = 120) -> dict:
|
||||
token = os.environ.get("PROFILE_CONTROL_TOKEN", "").strip()
|
||||
if not token:
|
||||
token = _read(PROFILE_CONTROL_TOKEN_FILE)
|
||||
@@ -600,7 +938,7 @@ def _profile_controller_request(method: str, path: str) -> dict:
|
||||
headers={"Authorization": f"Bearer {token}"},
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=120) as response:
|
||||
with urllib.request.urlopen(request, timeout=timeout) as response:
|
||||
return json.load(response)
|
||||
except urllib.error.HTTPError as exc:
|
||||
body = exc.read(500).decode(errors="replace")
|
||||
@@ -750,7 +1088,10 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
|
||||
log.info("Warte, bis llama.cpp das Profil geladen hat ...")
|
||||
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
|
||||
RUNTIME.save(last_profile=profile, phase="idle")
|
||||
STATE.mode = "llm"
|
||||
STATE.mode_phase = "ready"
|
||||
RUNTIME.save(last_profile=profile, mode="llm",
|
||||
return_profile=None, phase="idle")
|
||||
finally:
|
||||
# Nach einem fehlgeschlagenen Skript/Timeout darf der Router
|
||||
# Qwen nicht blind freigeben. Nur ein semantisch verifiziertes
|
||||
@@ -765,7 +1106,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Bildgenerierung und Editing (FLUX.2-klein-4B)
|
||||
# Bildgenerierung und Editing (FLUX.2 Klein 9B FP8)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class _Worker:
|
||||
@@ -861,7 +1202,10 @@ def _worker() -> _Worker:
|
||||
if not img.worker or not img.worker.alive():
|
||||
if img.worker:
|
||||
img.worker.stop()
|
||||
img.worker = _RemoteWorker() if IMAGE_WORKER_URL else _Worker()
|
||||
img.worker = (_RemoteWorker(kind="image", url=IMAGE_WORKER_URL,
|
||||
token=IMAGE_WORKER_TOKEN,
|
||||
endpoint="/generate")
|
||||
if IMAGE_WORKER_URL else _Worker())
|
||||
img.worker.start()
|
||||
return img.worker
|
||||
|
||||
@@ -871,7 +1215,12 @@ class _RemoteWorker:
|
||||
|
||||
model_loaded = False
|
||||
|
||||
def __init__(self) -> None:
|
||||
def __init__(self, *, kind: str, url: str, token: str,
|
||||
endpoint: str) -> None:
|
||||
self.kind = kind
|
||||
self.url = url
|
||||
self.token = token
|
||||
self.endpoint = endpoint
|
||||
self.running = False
|
||||
|
||||
def alive(self) -> bool:
|
||||
@@ -880,10 +1229,10 @@ class _RemoteWorker:
|
||||
def _request(self, method: str, path: str, payload: dict | None = None,
|
||||
timeout: float = 120) -> dict:
|
||||
body = None if payload is None else json.dumps(payload).encode()
|
||||
headers = {"Authorization": f"Bearer {IMAGE_WORKER_TOKEN}"}
|
||||
headers = {"Authorization": f"Bearer {self.token}"}
|
||||
if body is not None:
|
||||
headers["Content-Type"] = "application/json"
|
||||
req = urllib.request.Request(IMAGE_WORKER_URL + path, data=body,
|
||||
req = urllib.request.Request(self.url + path, data=body,
|
||||
method=method, headers=headers)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as response:
|
||||
@@ -898,9 +1247,9 @@ class _RemoteWorker:
|
||||
raise RuntimeError(f"Bild-Worker nicht erreichbar: {exc}") from exc
|
||||
|
||||
def start(self) -> None:
|
||||
if not IMAGE_WORKER_TOKEN or len(IMAGE_WORKER_TOKEN) < 32:
|
||||
if not self.token or len(self.token) < 32:
|
||||
raise RuntimeError("Bild-Worker-Token fehlt oder ist zu kurz")
|
||||
_profile_controller_request("POST", "/workers/image/start")
|
||||
_profile_controller_request("POST", f"/workers/{self.kind}/start")
|
||||
deadline = time.monotonic() + IMAGE_START_TIMEOUT
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
@@ -920,11 +1269,11 @@ class _RemoteWorker:
|
||||
clean.pop("cmd", None)
|
||||
output = clean.pop("output", "")
|
||||
clean["filename"] = os.path.basename(output)
|
||||
return self._request("POST", "/generate", clean, timeout)
|
||||
return self._request("POST", self.endpoint, clean, timeout)
|
||||
|
||||
def stop(self) -> None:
|
||||
try:
|
||||
_profile_controller_request("POST", "/workers/image/stop")
|
||||
_profile_controller_request("POST", f"/workers/{self.kind}/stop")
|
||||
finally:
|
||||
self.running = False
|
||||
self.model_loaded = False
|
||||
@@ -1011,6 +1360,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
|
||||
guidance: float, seed: int | None, n: int,
|
||||
quality: str = "standard",
|
||||
source_files: list[str] | None = None,
|
||||
model: str = IMAGE_MODEL_NAME,
|
||||
) -> tuple[list[str], str | None]:
|
||||
"""Orchestriert die Bildgenerierung inkl. Qwen-Hotswap.
|
||||
|
||||
@@ -1030,6 +1380,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
|
||||
results: list[str] = []
|
||||
warning: str | None = None
|
||||
img.last_error = None
|
||||
img.current_model = model
|
||||
# Qwen wird gestoppt → für Chats nicht verfügbar (die warten).
|
||||
_set_qwen_unavailable(True)
|
||||
try:
|
||||
@@ -1062,7 +1413,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
|
||||
filename = time.strftime("%Y%m%d-%H%M%S") + \
|
||||
f"-{os.urandom(2).hex()}.png"
|
||||
output = os.path.join(IMAGE_DIR, filename)
|
||||
resp = worker.request({
|
||||
worker_payload = {
|
||||
"cmd": "generate",
|
||||
"prompt": prompt,
|
||||
"width": width,
|
||||
@@ -1072,7 +1423,8 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
|
||||
"seed": seed,
|
||||
"output": output,
|
||||
"source_files": source_files or [],
|
||||
}, timeout=IMAGE_GEN_TIMEOUT)
|
||||
}
|
||||
resp = worker.request(worker_payload, timeout=IMAGE_GEN_TIMEOUT)
|
||||
if resp.get("status") != "ok":
|
||||
raise RuntimeError(
|
||||
resp.get("message", "Bildgenerierung fehlgeschlagen"))
|
||||
@@ -1090,10 +1442,10 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
|
||||
"steps": steps,
|
||||
"guidance": guidance,
|
||||
"quality": quality,
|
||||
"mode": "image-edit" if source_files else "text-to-image",
|
||||
"mode": ("image-edit" if source_files else "text-to-image"),
|
||||
"reference_images": len(source_files or []),
|
||||
"seconds": resp.get("seconds"),
|
||||
"model": IMAGE_MODEL_NAME,
|
||||
"model": model,
|
||||
"created": time.strftime("%Y-%m-%dT%H:%M:%S"),
|
||||
}
|
||||
meta_path = os.path.join(IMAGE_DIR, filename[:-4] + ".json")
|
||||
@@ -1139,6 +1491,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
|
||||
log.error(warning)
|
||||
# Qwen ist down → qwen_unavailable bleibt True.
|
||||
img.phase = "idle"
|
||||
img.current_model = None
|
||||
return results, warning
|
||||
|
||||
|
||||
@@ -1497,6 +1850,10 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self._send_json(200, self._models_payload())
|
||||
elif path == "/status" and self.command == "GET":
|
||||
self._send_json(200, self._status_payload())
|
||||
elif path == "/mode" and self.command == "GET":
|
||||
self._send_json(200, self._mode_payload())
|
||||
elif path == "/mode" and self.command == "POST":
|
||||
self._mode_change()
|
||||
elif path == "/v1/audio/models" and self.command == "GET":
|
||||
self._send_json(200, self._audio_models_payload())
|
||||
elif path == "/v1/audio/voices" and self.command == "GET":
|
||||
@@ -1519,6 +1876,13 @@ class Handler(BaseHTTPRequestHandler):
|
||||
else:
|
||||
self._send_error(503, "Sprachausgabe ist nicht installiert",
|
||||
"server_error", "feature_disabled")
|
||||
elif (path == "/v1/audio/speech/pcm-stream"
|
||||
and self.command == "POST"):
|
||||
if ENABLE_TTS:
|
||||
self._speech_pcm_stream()
|
||||
else:
|
||||
self._send_error(503, "Sprachausgabe ist nicht installiert",
|
||||
"server_error", "feature_disabled")
|
||||
elif path == "/v1/audio/transcriptions" and self.command == "POST":
|
||||
if ENABLE_STT:
|
||||
self._transcribe()
|
||||
@@ -1713,6 +2077,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"current_profile": current_profile(),
|
||||
"switching": STATE.switching,
|
||||
"profiles": PROFILES,
|
||||
"mode": self._mode_payload(),
|
||||
"upstream": {
|
||||
"url": UPSTREAM_URL,
|
||||
"reachable": up["reachable"],
|
||||
@@ -1734,7 +2099,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"phase": img.phase,
|
||||
"worker": "running" if (img.worker and img.worker.alive())
|
||||
else "stopped",
|
||||
"model": IMAGE_MODEL_NAME if img.phase != "idle" else None,
|
||||
"model": img.current_model if img.phase != "idle" else None,
|
||||
"model_loaded": bool(img.worker and img.worker.model_loaded),
|
||||
"last_image": img.last_image,
|
||||
"last_seconds": img.last_seconds,
|
||||
@@ -1744,6 +2109,49 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"stt": stt_status(),
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _mode_payload() -> dict:
|
||||
state = RUNTIME.load()
|
||||
return {
|
||||
"active": STATE.mode,
|
||||
"phase": STATE.mode_phase,
|
||||
"music_worker": _music_worker_state(),
|
||||
"music_health": _music_worker_health(),
|
||||
"yue2_worker": _worker_field("yue2_worker"),
|
||||
"yue2_health": _worker_field("yue2_health"),
|
||||
"separator_worker": _separator_worker_state(),
|
||||
"separator_health": _separator_worker_health(),
|
||||
"voice_worker": _voice_worker_state(),
|
||||
"voice_health": _voice_worker_health(),
|
||||
"voice_change_worker": _voice_change_worker_state(),
|
||||
"voice_change_health": _voice_change_worker_health(),
|
||||
"applio_worker": _worker_field("applio_worker"),
|
||||
"applio_health": _worker_field("applio_health"),
|
||||
"trellis_worker": _worker_field("trellis_worker"),
|
||||
"trellis_health": _worker_field("trellis_health"),
|
||||
"return_profile": state.get("return_profile"),
|
||||
"last_error": STATE.mode_error,
|
||||
"enabled": ENABLE_MUSIC_MODE,
|
||||
}
|
||||
|
||||
def _mode_change(self) -> None:
|
||||
try:
|
||||
data = json.loads(self._read_body() or b"{}")
|
||||
mode = data.get("mode") if isinstance(data, dict) else None
|
||||
if mode not in {"llm", "music", "yue2", "separation", "voice",
|
||||
"voicechange", "applio", "trellis"}:
|
||||
raise ValueError("Feld 'mode' enthält einen unbekannten Betriebsmodus")
|
||||
started, phase = schedule_operating_mode(mode)
|
||||
self._send_json(202 if started else 200, {
|
||||
"status": "accepted" if started else "ok",
|
||||
"requested_mode": mode,
|
||||
"phase": phase,
|
||||
})
|
||||
except ValueError as exc:
|
||||
self._send_error(400, str(exc), "invalid_request_error", "invalid_mode")
|
||||
except RuntimeError as exc:
|
||||
self._send_error(503, str(exc), "server_error", "mode_unavailable")
|
||||
|
||||
# ---------- Bildgenerierung ----------
|
||||
|
||||
def _image_generate(self) -> None:
|
||||
@@ -1841,6 +2249,13 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"invalid_request_error", "prompt_too_long")
|
||||
return
|
||||
|
||||
model = data.get("model", IMAGE_MODEL_NAME)
|
||||
if model != IMAGE_MODEL_NAME:
|
||||
self._send_error(
|
||||
400, f"unbekanntes Bildmodell: {model!r}",
|
||||
"invalid_request_error", "invalid_model")
|
||||
return
|
||||
|
||||
# Größe
|
||||
size = data.get("size", "1024x1024")
|
||||
if size not in IMAGE_SIZES:
|
||||
@@ -1857,7 +2272,6 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self._send_error(400, f"'n' muss eine Ganzzahl 1..{IMAGE_MAX_N} sein",
|
||||
"invalid_request_error", "invalid_n")
|
||||
return
|
||||
|
||||
# Qualität / Schritte / Guidance
|
||||
quality = data.get("quality", IMAGE_DEFAULT_QUALITY)
|
||||
if quality not in IMAGE_QUALITY:
|
||||
@@ -1866,8 +2280,9 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"invalid_request_error", "invalid_quality")
|
||||
return
|
||||
steps = data.get("steps", IMAGE_QUALITY[quality])
|
||||
if not isinstance(steps, int) or isinstance(steps, bool) or steps != 4:
|
||||
self._send_error(400, "FLUX.2-klein-4B erfordert 'steps'=4",
|
||||
if (not isinstance(steps, int) or isinstance(steps, bool)
|
||||
or steps != 4):
|
||||
self._send_error(400, f"{IMAGE_MODEL_NAME} erfordert 'steps'=4",
|
||||
"invalid_request_error", "invalid_steps")
|
||||
return
|
||||
guidance = data.get("guidance", 1.0)
|
||||
@@ -1878,7 +2293,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"invalid_request_error", "invalid_guidance")
|
||||
return
|
||||
if guidance != 1.0:
|
||||
self._send_error(400, "FLUX.2-klein-4B erfordert 'guidance'=1.0",
|
||||
self._send_error(400, f"{IMAGE_MODEL_NAME} erfordert 'guidance'=1.0",
|
||||
"invalid_request_error", "invalid_guidance")
|
||||
return
|
||||
|
||||
@@ -1906,7 +2321,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
try:
|
||||
results, warning = generate_image(
|
||||
prompt.strip(), width, height, steps, guidance, seed, n,
|
||||
quality, source_files)
|
||||
quality, source_files, model)
|
||||
except (ValueError, RuntimeError) as e:
|
||||
self._send_error(503, str(e), "server_error", "image_generation_failed")
|
||||
return
|
||||
@@ -1979,7 +2394,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self.end_headers()
|
||||
self.wfile.write(data)
|
||||
|
||||
# ---------- Sprachausgabe (XTTS-v2) ----------
|
||||
# ---------- Sprachausgabe (Qwen3-TTS) ----------
|
||||
|
||||
def _speech(self) -> None:
|
||||
try:
|
||||
@@ -2038,7 +2453,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"invalid_request_error", "invalid_speed")
|
||||
return
|
||||
|
||||
# Modell-Name optional; falls angegeben, muss es xtts-v2 sein.
|
||||
# Modell-Name optional; falls angegeben, muss es Qwen3-TTS sein.
|
||||
model = data.get("model")
|
||||
if model is not None and model != TTS_MODEL:
|
||||
self._send_error(400, f"unbekanntes Modell: {model!r} "
|
||||
@@ -2062,6 +2477,74 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self.end_headers()
|
||||
self.wfile.write(audio)
|
||||
|
||||
def _speech_pcm_stream(self) -> None:
|
||||
"""Pass through Qwen's native 24 kHz PCM stream without buffering."""
|
||||
try:
|
||||
body = self._read_body()
|
||||
data = json.loads(body)
|
||||
except ValueError as exc:
|
||||
self._send_error(400, str(exc) or "ungültiges JSON",
|
||||
"invalid_request_error", "invalid_body")
|
||||
return
|
||||
if not isinstance(data, dict):
|
||||
self._send_error(400, "Request muss ein JSON-Objekt sein",
|
||||
"invalid_request_error", "invalid_request")
|
||||
return
|
||||
text = data.get("input", data.get("text"))
|
||||
if not isinstance(text, str) or not text.strip() or len(text) > 8000:
|
||||
self._send_error(400, "'input' fehlt, ist leer oder zu lang",
|
||||
"invalid_request_error", "invalid_input")
|
||||
return
|
||||
voice = data.get("voice", TTS_DEFAULT_VOICE)
|
||||
if voice not in TTS_VOICES:
|
||||
self._send_error(400, f"ungültige Stimme: {voice!r}",
|
||||
"invalid_request_error", "invalid_voice")
|
||||
return
|
||||
|
||||
hostport = TTS_WORKER_URL.split("://", 1)[-1]
|
||||
host, _, port = hostport.partition(":")
|
||||
connection = None
|
||||
headers_sent = False
|
||||
try:
|
||||
connection = http.client.HTTPConnection(
|
||||
host, int(port) if port else 80, timeout=TTS_CONNECT_TIMEOUT)
|
||||
connection.request(
|
||||
"POST", "/tts/pcm-stream", body=body,
|
||||
headers={"Content-Type": "application/json",
|
||||
"Accept": "application/octet-stream"})
|
||||
connection.sock.settimeout(TTS_TIMEOUT)
|
||||
response = connection.getresponse()
|
||||
if response.status != 200:
|
||||
message = response.read(512).decode(errors="replace")
|
||||
self._send_error(503,
|
||||
f"TTS-Stream fehlgeschlagen: {message}",
|
||||
"server_error", "tts_failed")
|
||||
return
|
||||
self._last_code = 200
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "application/octet-stream")
|
||||
self.send_header("Cache-Control", "no-store")
|
||||
self.send_header("Connection", "close")
|
||||
self.end_headers()
|
||||
headers_sent = True
|
||||
while True:
|
||||
chunk = response.read1(16384)
|
||||
if not chunk:
|
||||
break
|
||||
self.wfile.write(chunk)
|
||||
self.wfile.flush()
|
||||
except (OSError, http.client.HTTPException) as exc:
|
||||
if not headers_sent:
|
||||
try:
|
||||
self._send_error(502, f"TTS-Worker nicht erreichbar: {exc}",
|
||||
"server_error", "tts_unavailable")
|
||||
except (OSError, BrokenPipeError):
|
||||
pass
|
||||
finally:
|
||||
if connection is not None:
|
||||
connection.close()
|
||||
self.close_connection = True
|
||||
|
||||
# ---------- Audio-Discovery ----------
|
||||
|
||||
def _audio_models_payload(self) -> dict:
|
||||
@@ -2080,7 +2563,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
models.append({
|
||||
"id": TTS_MODEL,
|
||||
"object": "model",
|
||||
"owned_by": "coqui-xtts",
|
||||
"owned_by": "qwen",
|
||||
"type": "speech",
|
||||
})
|
||||
return {"object": "list", "data": models}
|
||||
@@ -2269,6 +2752,12 @@ class Handler(BaseHTTPRequestHandler):
|
||||
data = json.loads(body)
|
||||
except ValueError:
|
||||
data = None
|
||||
if (isinstance(data, dict)
|
||||
and path in {"/v1/chat/completions", "/v1/responses"}):
|
||||
command = _control_command(data, path)
|
||||
if command is not None:
|
||||
self._control_response(command, data, path)
|
||||
return
|
||||
model = data.get("model") if isinstance(data, dict) else None
|
||||
if (isinstance(model, str) and REVIEW_UPSTREAM_URL
|
||||
and model == REVIEW_MODEL_NAME):
|
||||
@@ -2306,6 +2795,97 @@ class Handler(BaseHTTPRequestHandler):
|
||||
# An llama.cpp weiterleiten (mit Chat-Waiting, Streaming bleibt erhalten).
|
||||
self._proxy_with_wait(body)
|
||||
|
||||
def _control_response(self, command: str, data: dict, path: str) -> None:
|
||||
"""Return OpenAI-compatible local replies for Athena control commands."""
|
||||
if command == "/athena status":
|
||||
mode = self._mode_payload()
|
||||
profile = current_profile()
|
||||
text = (f"Athena läuft im {mode['active'].upper()}-Modus. "
|
||||
f"Phase: {mode['phase']}. Musik-Worker: "
|
||||
f"{mode['music_worker']}. YuE2: "
|
||||
f"{mode['yue2_worker']}. Stem-Separator: "
|
||||
f"{mode['separator_worker']}. Voice Studio: "
|
||||
f"{mode['voice_worker']}. Voice Changer: "
|
||||
f"{mode['voice_change_worker']}. 3D Studio: "
|
||||
f"{mode['trellis_worker']}. LLM-Profil: {profile or 'entladen'}.")
|
||||
else:
|
||||
target = ("music" if command == "/athena music" else
|
||||
"yue2" if command == "/athena yue2" else
|
||||
"separation" if command in {"/athena stems", "/athena separation"}
|
||||
else "voice" if command == "/athena voice"
|
||||
else "voicechange" if command in {"/athena voicechange", "/athena changer"}
|
||||
else "applio" if command == "/athena applio"
|
||||
else "trellis" if command in {"/athena 3d", "/athena trellis"}
|
||||
else "llm")
|
||||
try:
|
||||
started, phase = schedule_operating_mode(target)
|
||||
if started:
|
||||
text = ("Musikstudio wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "music" else
|
||||
"YuE2 Studio wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "yue2" else
|
||||
"Stimmtrennung wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "separation" else
|
||||
"Voice Studio wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "voice" else
|
||||
"Voice Changer wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "voicechange" else
|
||||
"Applio wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "applio" else
|
||||
"3D Studio wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "trellis" else
|
||||
"Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.")
|
||||
else:
|
||||
text = (f"Athena ist bereits im {target.upper()}-Modus "
|
||||
f"oder wechselt gerade ({phase}).")
|
||||
except RuntimeError as exc:
|
||||
self._send_error(503, str(exc), "server_error", "mode_unavailable")
|
||||
return
|
||||
|
||||
model = str(data.get("model") or "athena-control")
|
||||
created = int(time.time())
|
||||
request_id = f"athena-mode-{uuid.uuid4().hex[:16]}"
|
||||
if path == "/v1/responses":
|
||||
self._send_json(200, {
|
||||
"id": request_id, "object": "response", "created_at": created,
|
||||
"status": "completed", "model": model,
|
||||
"output": [{"type": "message", "role": "assistant",
|
||||
"content": [{"type": "output_text", "text": text}]}],
|
||||
"output_text": text,
|
||||
"usage": {"input_tokens": 0, "output_tokens": 0,
|
||||
"total_tokens": 0},
|
||||
})
|
||||
return
|
||||
if data.get("stream") is True:
|
||||
chunks = [
|
||||
{"id": request_id, "object": "chat.completion.chunk",
|
||||
"created": created, "model": model,
|
||||
"choices": [{"index": 0, "delta": {"role": "assistant",
|
||||
"content": text}, "finish_reason": None}]},
|
||||
{"id": request_id, "object": "chat.completion.chunk",
|
||||
"created": created, "model": model,
|
||||
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]},
|
||||
]
|
||||
body = "".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks)
|
||||
body += "data: [DONE]\n\n"
|
||||
encoded = body.encode()
|
||||
self._last_code = 200
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "text/event-stream")
|
||||
self.send_header("Content-Length", str(len(encoded)))
|
||||
self.send_header("Connection", "close")
|
||||
self.end_headers()
|
||||
self.wfile.write(encoded)
|
||||
return
|
||||
self._send_json(200, {
|
||||
"id": request_id, "object": "chat.completion", "created": created,
|
||||
"model": model,
|
||||
"choices": [{"index": 0, "message": {"role": "assistant",
|
||||
"content": text}, "finish_reason": "stop"}],
|
||||
"usage": {"prompt_tokens": 0, "completion_tokens": 0,
|
||||
"total_tokens": 0},
|
||||
})
|
||||
|
||||
def _acquire_model_lease(self, profile: str | None = None) -> dict:
|
||||
"""Atomar Profil sicherstellen und einen aktiven Request registrieren."""
|
||||
with STATE.lock:
|
||||
@@ -2545,6 +3125,29 @@ def _startup_reconcile() -> None:
|
||||
if removed:
|
||||
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
|
||||
|
||||
special_mode = previous.get("mode")
|
||||
if ENABLE_MUSIC_MODE and special_mode in {"music", "yue2", "separation",
|
||||
"voice", "voicechange", "applio",
|
||||
"trellis"}:
|
||||
STATE.mode = special_mode
|
||||
STATE.mode_phase = f"starting-{special_mode}"
|
||||
_set_qwen_unavailable(True)
|
||||
try:
|
||||
path, _worker_state, wait_ready = _special_worker(special_mode)
|
||||
_profile_controller_request(
|
||||
"POST", path,
|
||||
timeout=120)
|
||||
wait_ready()
|
||||
STATE.mode_phase = "ready"
|
||||
RUNTIME.save(mode=special_mode, phase=special_mode)
|
||||
log.info("Recovery: Spezialmodus %s wiederhergestellt", special_mode)
|
||||
except Exception as exc:
|
||||
STATE.mode_error = str(exc)
|
||||
STATE.mode_phase = "error"
|
||||
log.error("Recovery: Spezialmodus %s konnte nicht gestartet werden: %s",
|
||||
special_mode, exc)
|
||||
return
|
||||
|
||||
profile = current_profile()
|
||||
if profile is None:
|
||||
saved = previous.get("last_profile")
|
||||
@@ -2568,7 +3171,9 @@ def _startup_reconcile() -> None:
|
||||
and (not EXPECTED_MODELS.get(profile)
|
||||
or up.get("model") == EXPECTED_MODELS[profile])):
|
||||
_set_qwen_unavailable(False)
|
||||
RUNTIME.save(last_profile=profile, phase="idle")
|
||||
STATE.mode = "llm"
|
||||
STATE.mode_phase = "ready"
|
||||
RUNTIME.save(last_profile=profile, mode="llm", phase="idle")
|
||||
log.info("Recovery: Profil %s ist bereits bereit", profile)
|
||||
return
|
||||
_set_qwen_unavailable(True)
|
||||
|
||||
Reference in new issue
Block a user