From a78aed8e9e2e7ce1f2b462e78b75cdf0e4934d4a Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Sun, 13 Sep 2026 21:33:57 +0200 Subject: [PATCH] Add exclusive LTX-2 video studio profile --- compose.yaml | 3 + docs/OPERATING_MODES.md | 19 +++++- .../profile-controller/profile_controller.py | 64 +++++++++++++++++++ .../docker/wireguard-gateway/entrypoint.sh | 1 + .../references/architecture-and-modes.md | 5 +- platform/llama-dashboard/app.py | 12 ++-- platform/ltx2-studio/Dockerfile | 34 ++++++++++ platform/ltx2-studio/README.md | 16 +++++ platform/ltx2-studio/compose.yaml | 35 ++++++++++ platform/ltx2-studio/entrypoint.sh | 30 +++++++++ platform/recovery/rebuild-specialized.sh | 2 + router/ai_profile_router.py | 27 ++++++-- 12 files changed, 233 insertions(+), 15 deletions(-) create mode 100644 platform/ltx2-studio/Dockerfile create mode 100644 platform/ltx2-studio/README.md create mode 100644 platform/ltx2-studio/compose.yaml create mode 100644 platform/ltx2-studio/entrypoint.sh diff --git a/compose.yaml b/compose.yaml index 538478a..273243b 100644 --- a/compose.yaml +++ b/compose.yaml @@ -498,6 +498,7 @@ services: VOICE_CHANGE_WORKER: xvc APPLIO_WORKER: applio TRELLIS_WORKER: trellis2-q8 + VIDEO_WORKER: ltx2 networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: @@ -535,6 +536,7 @@ services: REQUEST_TIMEOUT: "600" YUE2_START_TIMEOUT: "600" TRELLIS_START_TIMEOUT: "900" + VIDEO_START_TIMEOUT: "900" # Last-resort guard for every OpenAI-compatible client. Without a # request limit llama.cpp uses n_predict=-1 and a reasoning loop can # consume the complete context before yielding visible output. @@ -759,6 +761,7 @@ services: MIKES_APPLIO_UI_URL: "${MIKES_APPLIO_UI_URL:-http://192.168.1.212:8012/}" TRELLIS_UI_URL: "${TRELLIS_UI_URL:-http://192.168.1.212:8013/}" YUE2_UI_URL: "${YUE2_UI_URL:-http://192.168.1.212:8014/}" + LTX2_UI_URL: "${LTX2_UI_URL:-http://192.168.1.212:8015/}" HOST_PROC: /host/proc HOST_DATA: /host/data HOST_MODELS: /host/models diff --git a/docs/OPERATING_MODES.md b/docs/OPERATING_MODES.md index 8f5d629..c99f8a2 100644 --- a/docs/OPERATING_MODES.md +++ b/docs/OPERATING_MODES.md @@ -1,6 +1,6 @@ # Athena-Betriebsmodi -Athena besitzt sechs gegenseitig exklusive Betriebsmodi: +Athena besitzt gegenseitig exklusive Betriebsmodi: - `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt. - `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt. @@ -13,6 +13,8 @@ Athena besitzt sechs gegenseitig exklusive Betriebsmodi: GPU-Dienste sind gestoppt. - `applio`: Applio stellt RVC-Inferenz, Modellverwaltung und Training bereit. Alle anderen GPU-Dienste sind gestoppt. +- `video`: LTX Desktop erzeugt mit LTX-2 kurze Videos. Beim Start werden alle + LLM-, Bild-, TTS-, Musik-, Sprach-, RVC- und 3D-GPU-Dienste gestoppt. Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus @@ -21,7 +23,7 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen ## Bedienung Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**, -**Voice Studio**, **X-VC** und **Applio / RVC** bereit. Im Musikmodus werden zwei Oberflächen angeboten: +**Voice Studio**, **X-VC**, **Applio / RVC**, **3D Studio** und **LTX-2 Video** bereit. Im Musikmodus werden zwei Oberflächen angeboten: - **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der @@ -55,6 +57,14 @@ nicht eine reine Synthesizer-Spur. Sprache nutzt das 48-kHz-Modell `audio-separator` 0.47.0. Die ältere API-Auswahl kompletter 2-/4-/6-Stem-Sätze bleibt rückwärtskompatibel. +Das LTX-2 Studio ist ausschließlich unter `http://192.168.1.212:8015` +erreichbar. Es verwendet das offizielle LTX Desktop 1.2.7 und speichert Modelle, +Einstellungen und Ergebnisse unter `/data/video/ltx-desktop`. Die RTX 5080 liegt +mit 16 GiB am offiziellen Minimum. Daher ist LTX Fast mit höchstens etwa zehn +Sekunden und 720p oder kleiner der Startpunkt; längere oder größere Läufe sind +nicht zugesichert. Der erste Modelldownload kann eine Hugging-Face-Anmeldung und +die Annahme der Lightricks-Modelllizenz verlangen. + Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter `http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter `/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem @@ -89,6 +99,7 @@ Router lokal beantwortet, auch wenn gerade kein LLM geladen ist: /athena voice /athena voicechange /athena applio +/athena ltx2 /athena llm /athena status ``` @@ -102,6 +113,7 @@ POST /mode {"mode":"separation"} POST /mode {"mode":"voice"} POST /mode {"mode":"voicechange"} POST /mode {"mode":"applio"} +POST /mode {"mode":"video"} POST /mode {"mode":"llm"} ``` @@ -111,7 +123,8 @@ Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in `com.mike-ai.stem-separator=bs-roformer` oder `com.mike-ai.voice-worker=vevo2` beziehungsweise `com.mike-ai.voice-change-worker=xvc` oder -`com.mike-ai.applio-worker=applio` markierten Container; freie +`com.mike-ai.applio-worker=applio` beziehungsweise +`com.mike-ai.video-worker=ltx2` markierten Container; freie Container- oder Docker-Befehle werden nicht entgegengenommen. ## Wiederanlauf diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py index 4fb3660..f0f1a3b 100644 --- a/platform/docker/profile-controller/profile_controller.py +++ b/platform/docker/profile-controller/profile_controller.py @@ -42,6 +42,8 @@ APPLIO_LABEL_KEY = "com.mike-ai.applio-worker" APPLIO_WORKER = os.environ.get("APPLIO_WORKER", "").strip() TRELLIS_LABEL_KEY = "com.mike-ai.trellis-worker" TRELLIS_WORKER = os.environ.get("TRELLIS_WORKER", "").strip() +VIDEO_LABEL_KEY = "com.mike-ai.video-worker" +VIDEO_WORKER = os.environ.get("VIDEO_WORKER", "").strip() LOCK = threading.Lock() log = logging.getLogger("profile-controller") @@ -187,6 +189,22 @@ def trellis_container() -> dict: return matches[0] +def video_container() -> dict: + if not VIDEO_WORKER: + raise RuntimeError("video worker is not configured") + matches = [item for item in labelled_containers(VIDEO_LABEL_KEY) + if item.get("Labels", {}).get(VIDEO_LABEL_KEY) == VIDEO_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one video worker {VIDEO_WORKER!r}, found {len(matches)}") + return matches[0] + + +def stop_video_if_configured() -> None: + if VIDEO_WORKER: + stop_container(video_container(), timeout=30) + + def stop_music_if_configured() -> None: if MUSIC_WORKER: stop_container(music_container(), timeout=30) @@ -290,6 +308,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict: stop_separator_if_configured() stop_voice_tools() stop_trellis_if_configured() + stop_video_if_configured() for other in image_containers(): if other["Id"] != item["Id"]: stop_container(other, timeout=20) @@ -319,6 +338,7 @@ def set_music_worker(running: bool) -> dict: stop_voice_tools() stop_yue2_if_configured() stop_trellis_if_configured() + stop_video_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -340,6 +360,7 @@ def set_yue2_worker(running: bool) -> dict: stop_separator_if_configured() stop_voice_tools() stop_trellis_if_configured() + stop_video_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -361,6 +382,7 @@ def set_separator_worker(running: bool) -> dict: stop_yue2_if_configured() stop_voice_tools() stop_trellis_if_configured() + stop_video_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -383,6 +405,7 @@ def set_voice_worker(running: bool) -> dict: stop_separator_if_configured() stop_voice_tools("voice") stop_trellis_if_configured() + stop_video_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -405,6 +428,7 @@ def set_voice_change_worker(running: bool) -> dict: stop_separator_if_configured() stop_voice_tools("voicechange") stop_trellis_if_configured() + stop_video_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -427,6 +451,7 @@ def set_applio_worker(running: bool) -> dict: stop_separator_if_configured() stop_voice_tools("applio") stop_trellis_if_configured() + stop_video_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -455,6 +480,28 @@ def set_trellis_worker(running: bool) -> dict: "state": "running" if running else "stopped"} +def set_video_worker(running: bool) -> dict: + """Start LTX-2 exclusively, or stop it before another mode is loaded.""" + with LOCK: + item = video_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_yue2_if_configured() + stop_separator_if_configured() + stop_voice_tools() + stop_trellis_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"video_worker": VIDEO_WORKER, + "state": "running" if running else "stopped"} + + def active_profile(items: dict[str, dict] | None = None) -> str | None: items = items or containers() active = [name for name, item in items.items() if item.get("State") == "running"] @@ -475,6 +522,7 @@ def activate(profile: str) -> dict: stop_separator_if_configured() stop_voice_tools() stop_trellis_if_configured() + stop_video_if_configured() start_container(tts_container()) items = containers() missing = [name for name in ALLOWED if name not in items] @@ -588,6 +636,13 @@ class Handler(BaseHTTPRequestHandler): "unhealthy" if "(unhealthy)" in trellis_status else "starting" if trellis.get("State") == "running" else "stopped") + video = video_container() if VIDEO_WORKER else {} + video_status = video.get("Status", "") + video_health = ("disabled" if not VIDEO_WORKER else + "healthy" if "(healthy)" in video_status else + "unhealthy" if "(unhealthy)" in video_status else + "starting" if video.get("State") == "running" else + "stopped") self.reply(200, {"active_profile": active_profile(items), "music_worker": music.get("State", "disabled"), "music_health": music_health, @@ -603,6 +658,8 @@ class Handler(BaseHTTPRequestHandler): "applio_health": applio_health, "trellis_worker": trellis.get("State", "disabled"), "trellis_health": trellis_health, + "video_worker": video.get("State", "disabled"), + "video_health": video_health, "profiles": {name: items.get(name, {}).get( "State", "missing") for name in ALLOWED}}) except Exception as exc: @@ -669,6 +726,13 @@ class Handler(BaseHTTPRequestHandler): log.exception("TRELLIS worker transition failed") self.reply(503, {"error": str(exc)}) return + if self.path in {"/workers/video/start", "/workers/video/stop"}: + try: + self.reply(200, set_video_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("video worker transition failed") + self.reply(503, {"error": str(exc)}) + return worker_paths = { "/workers/image/start": (IMAGE_WORKER, True), "/workers/image/stop": (IMAGE_WORKER, False), diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh index 34c2043..51d1682 100644 --- a/platform/docker/wireguard-gateway/entrypoint.sh +++ b/platform/docker/wireguard-gateway/entrypoint.sh @@ -86,6 +86,7 @@ start_proxy 8011 applio-studio:6969 start_proxy 8012 mikes-applio-ui:8012 start_proxy 8013 trellis-studio:8080 start_proxy 8014 yue2-studio:8014 +start_proxy 8015 ltx2-studio:8015 start_proxy 8202 mcp-athena-operator:8000 start_proxy 9443 portainer:9443 diff --git a/platform/hermes/skills/athena-operator/references/architecture-and-modes.md b/platform/hermes/skills/athena-operator/references/architecture-and-modes.md index 1f5f714..8095193 100644 --- a/platform/hermes/skills/athena-operator/references/architecture-and-modes.md +++ b/platform/hermes/skills/athena-operator/references/architecture-and-modes.md @@ -18,7 +18,7 @@ correct the durable source when the user requested maintenance. ## Exclusive states -Athena has five mutually exclusive persistent modes: `llm`, `music`, +Athena has mutually exclusive persistent modes: `llm`, `music`, `separation`, `voice`, and `voicechange`. Image generation is a transactional request: it temporarily pauses the active text profile and Qwen3-TTS, runs the image worker, then restores the previous LLM state. @@ -45,3 +45,6 @@ The GPU workers use `restart: "no"` and are created once, then started on demand. A stopped `mike-ai-llama-*`, image, music, separator, OmniVoice or X-VC container is expected. A candidate is stale only after checking Compose, labels, mounts, router/controller references, model paths and test history. + + +`video` starts the allowlisted LTX Desktop worker exclusively. It is controlled with `/athena ltx2` and its private UI is exposed at `http://192.168.1.212:8015`. diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py index 2f5a925..6f8f37d 100644 --- a/platform/llama-dashboard/app.py +++ b/platform/llama-dashboard/app.py @@ -36,6 +36,7 @@ MIKES_APPLIO_UI_URL = os.getenv( ) TRELLIS_UI_URL = os.getenv("TRELLIS_UI_URL", "http://192.168.1.212:8013/") YUE2_UI_URL = os.getenv("YUE2_UI_URL", "http://192.168.1.212:8014/") +LTX2_UI_URL = os.getenv("LTX2_UI_URL", "http://192.168.1.212:8015/") HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc")) HOST_DATA = os.getenv("HOST_DATA", "/host/data") HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models")) @@ -314,7 +315,7 @@ def router_status() -> tuple[dict[str, Any], str | None]: def change_mode(mode: str) -> tuple[int, dict[str, Any]]: if mode not in {"llm", "music", "yue2", "separation", "voice", - "voicechange", "applio", "trellis"}: + "voicechange", "applio", "trellis", "video"}: return 400, {"error": "invalid mode"} headers = {"Accept": "application/json", "Content-Type": "application/json"} if ROUTER_API_KEY: @@ -673,7 +674,7 @@ HTML = r'''