Update Athena runtimes and sync live extensions
This commit is contained in:
@@ -1,10 +1,11 @@
|
|||||||
# Athena AI
|
# Athena AI
|
||||||
|
|
||||||
Stand: **12. September 2026**, auf Athena geprüft. Produktiv läuft
|
Stand: **15. September 2026**, auf Athena geprüft. Produktiv läuft
|
||||||
**llama.cpp b10930** (`56381e407c0ccfb3a6f71e668a27a901001d22ce`).
|
**llama.cpp 0.4.1** (`b29c606`).
|
||||||
Der [geprüfte Live-Stand](docs/LIVE_STATE.md) beschreibt Profile, GPUs und die
|
Der [geprüfte Live-Stand](docs/LIVE_STATE.md) beschreibt Profile, GPUs und die
|
||||||
Abweichung zwischen dem bereitgestellten Stack und dem Git-Checkout.
|
Abweichung zwischen dem bereitgestellten Stack und dem Git-Checkout.
|
||||||
Der [Updatebericht](docs/LLAMA_B10930_UPDATE_20260912.md) enthält Tests und Rückfallstand.
|
Der [Update- und Aufräumbericht vom 15. September](docs/UPDATE_AUDIT_20260915.md)
|
||||||
|
enthält Versionsvergleich, Tests und Rückfallstand.
|
||||||
|
|
||||||
|
|
||||||
Athena ist die lokale Inferenzmaschine. Der Docker-Stack stellt
|
Athena ist die lokale Inferenzmaschine. Der Docker-Stack stellt
|
||||||
@@ -49,10 +50,10 @@ Schätzungen stehen in [Auflösungstest vom 12. September](docs/FLUX_RESOLUTION_
|
|||||||
|
|
||||||
## Installation des Repository-Stands
|
## Installation des Repository-Stands
|
||||||
|
|
||||||
Der Live-Stack enthält zusätzliche, noch nicht vollständig in Git übernommene
|
Der produktive Quellstand ist mit diesem Repository abgeglichen. Ein frischer
|
||||||
Anpassungen. Ein frischer Clone bildet daher nicht automatisch den kompletten
|
Clone enthält den Athena-Kern, die Spezialprofile sowie die LTX-Erweiterungen.
|
||||||
Live-Stand einschließlich Spezialdiensten ab. Vor Wiederherstellung den
|
Modelle, Laufzeitdaten und geheime Konfiguration bleiben außerhalb von Git und
|
||||||
gesicherten bereitgestellten Quellstand und [LIVE_STATE.md](docs/LIVE_STATE.md) berücksichtigen.
|
werden über die dokumentierten Sicherungen wiederhergestellt.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cp config/install.env.example /root/mike-ai-install.env
|
cp config/install.env.example /root/mike-ai-install.env
|
||||||
@@ -150,8 +151,7 @@ Unraid-DockerMan-Templates. Details stehen in
|
|||||||
## Wiederherstellung
|
## Wiederherstellung
|
||||||
|
|
||||||
Der Sicherungsumfang und die Grenzen eines Neuaufbaus aus dem Repository stehen
|
Der Sicherungsumfang und die Grenzen eines Neuaufbaus aus dem Repository stehen
|
||||||
in [docs/RECOVERY.md](docs/RECOVERY.md). Der gesicherte Live-Quellstand enthält
|
in [docs/RECOVERY.md](docs/RECOVERY.md). Der vor dem Update gesicherte Live-Quellstand dient als Rückfallstand.
|
||||||
zusätzliche Anpassungen und muss erhalten bleiben.
|
|
||||||
|
|
||||||
## Sicherheitsregeln
|
## Sicherheitsregeln
|
||||||
|
|
||||||
|
|||||||
+4
-4
@@ -530,8 +530,6 @@ services:
|
|||||||
ROUTER_STATE_FILE: /var/lib/mike-ai-profile-router/state.json
|
ROUTER_STATE_FILE: /var/lib/mike-ai-profile-router/state.json
|
||||||
ROUTER_MAX_CONCURRENT_REQUESTS: "16"
|
ROUTER_MAX_CONCURRENT_REQUESTS: "16"
|
||||||
UPSTREAM_URL: http://llama-upstream:8080
|
UPSTREAM_URL: http://llama-upstream:8080
|
||||||
PROFILE_CONTROL_URL: http://profile-controller:8090
|
|
||||||
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
|
||||||
SWITCH_TIMEOUT: "600"
|
SWITCH_TIMEOUT: "600"
|
||||||
REQUEST_TIMEOUT: "600"
|
REQUEST_TIMEOUT: "600"
|
||||||
YUE2_START_TIMEOUT: "600"
|
YUE2_START_TIMEOUT: "600"
|
||||||
@@ -736,7 +734,7 @@ services:
|
|||||||
image: mike-ai/llama-dashboard:local
|
image: mike-ai/llama-dashboard:local
|
||||||
container_name: mike-ai-llama-dashboard
|
container_name: mike-ai-llama-dashboard
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
networks: [frontend]
|
networks: [frontend, control]
|
||||||
gpus: all
|
gpus: all
|
||||||
read_only: true
|
read_only: true
|
||||||
tmpfs:
|
tmpfs:
|
||||||
@@ -752,6 +750,8 @@ services:
|
|||||||
DASHBOARD_PORT: "8099"
|
DASHBOARD_PORT: "8099"
|
||||||
ROUTER_URL: http://router:8081
|
ROUTER_URL: http://router:8081
|
||||||
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
||||||
|
PROFILE_CONTROL_URL: http://profile-controller:8090
|
||||||
|
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||||
MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}"
|
MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}"
|
||||||
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
|
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
|
||||||
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
|
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
|
||||||
@@ -800,7 +800,7 @@ services:
|
|||||||
security_opt: ["no-new-privileges:true"]
|
security_opt: ["no-new-privileges:true"]
|
||||||
|
|
||||||
backup:
|
backup:
|
||||||
image: ${BACKUP_IMAGE:-offen/docker-volume-backup@sha256:19102d8e59eb1d598cf8c647c2b21100abaadc5a1c808ac643fa612e323c3013}
|
image: ${BACKUP_IMAGE:-offen/docker-volume-backup@sha256:ca882e494b409297885a8429af2c311e627d69dc8897857159a84ef5efc1a05b}
|
||||||
container_name: mike-ai-backup
|
container_name: mike-ai-backup
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
environment:
|
environment:
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
# Container-Inventar auf Athena
|
# Container-Inventar auf Athena
|
||||||
|
|
||||||
Stand: 12. September 2026
|
Stand: 15. September 2026
|
||||||
|
|
||||||
Athena besteht beim lesenden Abgleich aus 25 Docker-Containern. Nicht jeder Container enthält
|
Athena besteht beim lesenden Abgleich aus 25 Docker-Containern. Nicht jeder Container enthält
|
||||||
ein KI-Modell: Router, Oberflächen, Netzwerk, Steuerung und Sicherung sind
|
ein KI-Modell: Router, Oberflächen, Netzwerk, Steuerung und Sicherung sind
|
||||||
@@ -10,7 +10,7 @@ nicht automatisch ein ungenutzter Rest.
|
|||||||
|
|
||||||
| Container | Modell oder wesentliche Komponente | Aufgabe |
|
| Container | Modell oder wesentliche Komponente | Aufgabe |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
| `mike-ai-backup` | kein Modell; Offen Docker Volume Backup | Sichert `/data`, `/etc/mike-ai`, den Stack und die persistenten Docker-Volumes im Fünf-Stunden-Takt. |
|
| `mike-ai-backup` | kein Modell; Offen Docker Volume Backup `v2` (Digest vom 15. September 2026) | Sichert `/data`, `/etc/mike-ai`, den Stack und die persistenten Docker-Volumes im Fünf-Stunden-Takt. |
|
||||||
| `mike-ai-applio-studio` | Applio/RVC; Stimmenmodelle werden nutzerseitig ergänzt | Vollständige RVC-Oberfläche für Inferenz, Modellverwaltung und Training auf der RTX 5080. Für eine Konvertierung ist ein importiertes oder trainiertes `.pth`-Modell nötig; eine Referenzaufnahme allein reicht nicht. |
|
| `mike-ai-applio-studio` | Applio/RVC; Stimmenmodelle werden nutzerseitig ergänzt | Vollständige RVC-Oberfläche für Inferenz, Modellverwaltung und Training auf der RTX 5080. Für eine Konvertierung ist ein importiertes oder trainiertes `.pth`-Modell nötig; eine Referenzaufnahme allein reicht nicht. |
|
||||||
| `mike-ai-image-worker` | FLUX.2 Klein 9B FP8, Qwen3-8B NF4 Textencoder und VAE | Erzeugt und bearbeitet Bilder transaktional; nutzt während eines Auftrags RTX 5080 und RTX 3060. |
|
| `mike-ai-image-worker` | FLUX.2 Klein 9B FP8, Qwen3-8B NF4 Textencoder und VAE | Erzeugt und bearbeitet Bilder transaktional; nutzt während eines Auftrags RTX 5080 und RTX 3060. |
|
||||||
| `mike-ai-llama-dashboard` | kein Modell | Zeigt Telemetrie, Profile, GPU-Nutzung und Betriebsarten an und bietet die Modusumschaltung. |
|
| `mike-ai-llama-dashboard` | kein Modell | Zeigt Telemetrie, Profile, GPU-Nutzung und Betriebsarten an und bietet die Modusumschaltung. |
|
||||||
@@ -29,14 +29,14 @@ nicht automatisch ein ungenutzter Rest.
|
|||||||
| `mike-ai-stem-separator` | BS-RoFormer Viperx 1297, `htdemucs_ft`, `htdemucs_6s`, `MossFormer2_SE_48K` | Trennt Gesang, Instrumente oder Sprache/Hintergrundgeräusche im exklusiven Separationsmodus. |
|
| `mike-ai-stem-separator` | BS-RoFormer Viperx 1297, `htdemucs_ft`, `htdemucs_6s`, `MossFormer2_SE_48K` | Trennt Gesang, Instrumente oder Sprache/Hintergrundgeräusche im exklusiven Separationsmodus. |
|
||||||
| `mike-ai-tts-gateway` | kein eigenes Modell | Normalisiert Text, konvertiert Ausgabeformate und stellt Qwen3-TTS sowie natives PCM-Streaming über eine stabile interne API bereit. |
|
| `mike-ai-tts-gateway` | kein eigenes Modell | Normalisiert Text, konvertiert Ausgabeformate und stellt Qwen3-TTS sowie natives PCM-Streaming über eine stabile interne API bereit. |
|
||||||
| `mike-ai-voice-studio` | `k2-fsa/OmniVoice` 0.2.1 mit Whisper-ASR | Erzeugt Text-to-Speech mit einer Referenzstimme; kein Audio-to-Audio-Voice-Changer. |
|
| `mike-ai-voice-studio` | `k2-fsa/OmniVoice` 0.2.1 mit Whisper-ASR | Erzeugt Text-to-Speech mit einer Referenzstimme; kein Audio-to-Audio-Voice-Changer. |
|
||||||
| `mike-ai-whisper` | Whisper.cpp `ggml-small` | Lokale deutsche Spracherkennung auf der CPU über `/v1/audio/transcriptions`. |
|
| `mike-ai-whisper` | Whisper.cpp 1.9.4, `ggml-small` | Lokale deutsche Spracherkennung auf der CPU über `/v1/audio/transcriptions`. |
|
||||||
| `mike-ai-wireguard-gateway` | kein Modell | Veröffentlicht Dashboard und Fachoberflächen ausschließlich über den privaten WireGuard-Pfad. |
|
| `mike-ai-wireguard-gateway` | kein Modell | Veröffentlicht Dashboard und Fachoberflächen ausschließlich über den privaten WireGuard-Pfad. |
|
||||||
| `mike-ai-xvc-studio` | `chenxie95/X-VC`, GLM-4-Voice-Tokenizer und optional Resemble Enhance | Wandelt eine vorhandene Sprachaufnahme anhand einer Referenzstimme in Audio zu Audio um; gibt das native 16-kHz-Ergebnis und optional eine neural restaurierte 44,1-kHz-Fassung aus. |
|
| `mike-ai-xvc-studio` | `chenxie95/X-VC`, GLM-4-Voice-Tokenizer und optional Resemble Enhance | Wandelt eine vorhandene Sprachaufnahme anhand einer Referenzstimme in Audio zu Audio um; gibt das native 16-kHz-Ergebnis und optional eine neural restaurierte 44,1-kHz-Fassung aus. |
|
||||||
| `mike-ai-yue2-playground` | Image `mike-ai/yue2:3b-0.1.6` | Vorhandener, gestoppter Playground; in diesem Abgleich nicht funktional getestet. |
|
| `mike-ai-yue2-playground` | Image `mike-ai/yue2:3b-0.1.6` | Vorhandener, gestoppter Playground; in diesem Abgleich nicht funktional getestet. |
|
||||||
| `mike-ai-trellis-studio` | Image `mike-ai/trellis-studio:0.6.0-q8` | Vorhandener, gestoppter 3D-Worker; in diesem Abgleich nicht funktional getestet. |
|
| `mike-ai-trellis-studio` | Image `mike-ai/trellis-studio:0.6.0-q8` | Vorhandener, gestoppter 3D-Worker; in diesem Abgleich nicht funktional getestet. |
|
||||||
| `mike-ai-mikes-applio-ui` | Image `mike-ai/mikes-applio-ui:latest` | Laufende zusätzliche Applio-Oberfläche. |
|
| `mike-ai-mikes-applio-ui` | Image `mike-ai/mikes-applio-ui:latest` | Laufende zusätzliche Applio-Oberfläche. |
|
||||||
|
|
||||||
Alle fünf llama-Container verwenden b10930. Beim Abgleich lief Medium;
|
Alle fünf llama-Container verwenden llama.cpp 0.4.1 (`b29c606`). Beim Abgleich lief Medium;
|
||||||
die anderen vier waren gestoppt. Der Image-Worker stand auf Created.
|
die anderen vier waren gestoppt. Der Image-Worker stand auf Created.
|
||||||
Die übrigen GPU-Spezialworker waren gestoppt, die Infrastruktur und die
|
Die übrigen GPU-Spezialworker waren gestoppt, die Infrastruktur und die
|
||||||
Musik-/Applio-Oberflächen liefen. Diese Zustände ändern sich mit der Nutzung.
|
Musik-/Applio-Oberflächen liefen. Diese Zustände ändern sich mit der Nutzung.
|
||||||
|
|||||||
@@ -1,19 +1,23 @@
|
|||||||
# Aktuelle Laufzeitnotizen
|
# Aktuelle Laufzeitnotizen
|
||||||
|
|
||||||
Stand: 12. September 2026.
|
Stand: 15. September 2026.
|
||||||
|
|
||||||
Der verbindliche Abgleich steht im [geprüften Live-Stand](LIVE_STATE.md).
|
Der verbindliche Abgleich steht im [geprüften Live-Stand](LIVE_STATE.md).
|
||||||
Produktiv läuft **llama.cpp b10930**, zuvor b10872. Die frühere Angabe b10781
|
Produktiv läuft **llama.cpp 0.4.1** auf Commit `b29c606`, zuvor b10930.
|
||||||
war veraltet. Alle fünf installierten Profile wurden aktualisiert; die
|
Alle fünf installierten Profile verwenden dasselbe CUDA-Image. Die Startoption
|
||||||
Startoption wurde auf `--load-mode none` migriert.
|
bleibt `--load-mode none`.
|
||||||
|
|
||||||
[B10930-Updatebericht](LLAMA_B10930_UPDATE_20260912.md): Version, Image-ID,
|
[B10930-Updatebericht](LLAMA_B10930_UPDATE_20260912.md): Version, Image-ID,
|
||||||
Funktionsproben, Vergleichsmessungen und gesicherter Rückfallstand.
|
Funktionsproben, Vergleichsmessungen und gesicherter Rückfallstand.
|
||||||
|
|
||||||
|
[Update-Audit vom 15. September](UPDATE_AUDIT_20260915.md): aktualisierte
|
||||||
|
Komponenten, unveränderte aktuelle Komponenten, Aufräumarbeiten und Rollback.
|
||||||
|
|
||||||
FLUX verwendet **9B FP8 Beta** mit Textencoder auf der RTX 3060.
|
FLUX verwendet **9B FP8 Beta** mit Textencoder auf der RTX 3060.
|
||||||
[Auflösungstest](FLUX_RESOLUTION_TEST_20260912.md): 1024 erfolgreich,
|
[Auflösungstest](FLUX_RESOLUTION_TEST_20260912.md): 1024 erfolgreich,
|
||||||
1280 CUDA-OOM; produktive Begrenzung weiterhin 1024.
|
1280 CUDA-OOM; produktive Begrenzung weiterhin 1024.
|
||||||
Sprachausgabe: Qwen3-TTS 1.7B ohne Piper. Spracherkennung: Whisper.cpp small auf CPU.
|
Sprachausgabe: Qwen3-TTS 1.7B ohne Piper. Spracherkennung: Whisper.cpp 1.9.4
|
||||||
|
mit dem Modell `small` auf der CPU. Applio basiert auf dem stabilen Release 3.6.4.
|
||||||
|
|
||||||
Datierte ältere Testberichte beschreiben ihre damaligen Versuchsbedingungen,
|
Datierte ältere Testberichte beschreiben ihre damaligen Versuchsbedingungen,
|
||||||
keine automatische Freigabe für den heutigen Betrieb. Die Repository-Matrix
|
keine automatische Freigabe für den heutigen Betrieb. Die Repository-Matrix
|
||||||
|
|||||||
@@ -0,0 +1,48 @@
|
|||||||
|
# Update-Audit und Aufräumarbeiten vom 15. September 2026
|
||||||
|
|
||||||
|
Der Abgleich wurde zunächst ohne Zustandsänderung gegen die offiziellen Git-Repositories
|
||||||
|
und Container-Registries durchgeführt. Danach wurden ausschließlich bestätigte veraltete
|
||||||
|
Komponenten aktualisiert. Athena wurde weder neu gestartet noch heruntergefahren.
|
||||||
|
|
||||||
|
## Aktualisiert
|
||||||
|
|
||||||
|
| Komponente | Vorher | Nachher |
|
||||||
|
|---|---|---|
|
||||||
|
| llama.cpp | b10930 / `56381e4` | Release 0.4.1 / `b29c606` |
|
||||||
|
| whisper.cpp | 1.9.1 | 1.9.4 |
|
||||||
|
| Applio | Commit `7fa68ec` | stabiles Release 3.6.4 / `45f2dcb` |
|
||||||
|
| Offen Docker Volume Backup | Image-Digest `19102d…` | aktueller `v2`-Digest `ca882e…` |
|
||||||
|
|
||||||
|
Das llama.cpp-Update enthält unter anderem Korrekturen für Qwen-Modelle,
|
||||||
|
Chat- und Tool-Parsing, multimodale Eingaben sowie den Serverbetrieb.
|
||||||
|
|
||||||
|
## Bereits aktuell
|
||||||
|
|
||||||
|
LTX Desktop 1.2.7, ACE-Step 1.5 0.1.8, Portainer CE 2.45.0 LTS,
|
||||||
|
Qwen3-TTS-Server, TRELLIS.cpp 0.6.0, X-VC und YuE2 0.1.6 waren beim
|
||||||
|
Abgleich bereits auf dem aktuellen stabilen beziehungsweise festgelegten Stand.
|
||||||
|
|
||||||
|
## Konsolidierter Quellstand
|
||||||
|
|
||||||
|
Die produktiven Erweiterungen für LTX, den lokalen Prompt Enhancer, das rollierende
|
||||||
|
Video-Extend sowie das Schlafen und Aufwecken der VNC-Oberfläche wurden aus dem
|
||||||
|
Live-Stack in Git übernommen. AppleDouble-Dateien (`._*`), Python-Caches,
|
||||||
|
temporäre Sicherungskopien und überholte Zweit-Checkouts gehören nicht zum
|
||||||
|
Produktionsquellstand.
|
||||||
|
|
||||||
|
Vor der Bereinigung wurde auf Athena ein vollständiges Archiv der Konfiguration und
|
||||||
|
Arbeitskopien unter `/root/athena-pre-update-20260915-073740.tar.zst` abgelegt.
|
||||||
|
SHA-256: `c15aa831908bed36ba9de2c7f80ae4aa7b3a9bb49d0d4ec579737e506f0aa4cc`.
|
||||||
|
|
||||||
|
## Update-Regel
|
||||||
|
|
||||||
|
Ein Update-Audit vergleicht getrennt:
|
||||||
|
|
||||||
|
1. den in Dockerfile oder Compose festgelegten Upstream-Stand,
|
||||||
|
2. den Git-Stand des Repositories,
|
||||||
|
3. das gebaute lokale Image und
|
||||||
|
4. das Image des tatsächlich vorhandenen Containers.
|
||||||
|
|
||||||
|
Floating Tags allein gelten nicht als Nachweis eines Updates. Produktionsstände
|
||||||
|
werden auf Commit, Release oder Registry-Digest festgelegt und erst nach Build,
|
||||||
|
Healthcheck und Funktionsprobe dokumentiert.
|
||||||
@@ -1,7 +1,7 @@
|
|||||||
# syntax=docker/dockerfile:1
|
# syntax=docker/dockerfile:1
|
||||||
FROM python:3.12-trixie
|
FROM python:3.12-trixie
|
||||||
|
|
||||||
ARG APPLIO_COMMIT=7fa68ec2166ab1331c539704159fa14901e94e5a
|
ARG APPLIO_COMMIT=45f2dcbb9c6dfaa188c5428f4d56297e1e30a63c
|
||||||
ENV PATH=/app/.venv/bin:$PATH \
|
ENV PATH=/app/.venv/bin:$PATH \
|
||||||
HF_HOME=/models/huggingface \
|
HF_HOME=/models/huggingface \
|
||||||
PIP_DISABLE_PIP_VERSION_CHECK=1
|
PIP_DISABLE_PIP_VERSION_CHECK=1
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
services:
|
services:
|
||||||
applio-studio:
|
applio-studio:
|
||||||
build: .
|
build: .
|
||||||
image: mike-ai/applio-studio:7fa68ec
|
image: mike-ai/applio-studio:3.6.4
|
||||||
container_name: mike-ai-applio-studio
|
container_name: mike-ai-applio-studio
|
||||||
restart: "no"
|
restart: "no"
|
||||||
# PyTorch DataLoader workers exchange training batches through /dev/shm.
|
# PyTorch DataLoader workers exchange training batches through /dev/shm.
|
||||||
|
|||||||
@@ -54,10 +54,12 @@ class UnixConnection(http.client.HTTPConnection):
|
|||||||
self.sock.connect(SOCKET_PATH)
|
self.sock.connect(SOCKET_PATH)
|
||||||
|
|
||||||
|
|
||||||
def docker_request(method: str, path: str) -> tuple[int, bytes]:
|
def docker_request(method: str, path: str, payload: dict | None = None) -> tuple[int, bytes]:
|
||||||
conn = UnixConnection("localhost", timeout=30)
|
conn = UnixConnection("localhost", timeout=30)
|
||||||
try:
|
try:
|
||||||
conn.request(method, path, headers={"Content-Type": "application/json"})
|
body = json.dumps(payload).encode() if payload is not None else None
|
||||||
|
conn.request(method, path, body=body,
|
||||||
|
headers={"Content-Type": "application/json"})
|
||||||
response = conn.getresponse()
|
response = conn.getresponse()
|
||||||
return response.status, response.read()
|
return response.status, response.read()
|
||||||
finally:
|
finally:
|
||||||
@@ -200,6 +202,69 @@ def video_container() -> dict:
|
|||||||
return matches[0]
|
return matches[0]
|
||||||
|
|
||||||
|
|
||||||
|
def docker_exec(item: dict, command: str) -> str:
|
||||||
|
status, body = docker_request("POST", f"/containers/{item['Id']}/exec", {
|
||||||
|
"AttachStdout": True, "AttachStderr": True, "Tty": True,
|
||||||
|
"Cmd": ["/bin/sh", "-c", command],
|
||||||
|
})
|
||||||
|
if status != 201:
|
||||||
|
raise RuntimeError(f"failed to create container exec: HTTP {status}")
|
||||||
|
exec_id = json.loads(body)["Id"]
|
||||||
|
status, body = docker_request("POST", f"/exec/{exec_id}/start",
|
||||||
|
{"Detach": False, "Tty": True})
|
||||||
|
if status != 200:
|
||||||
|
raise RuntimeError(f"failed to start container exec: HTTP {status}")
|
||||||
|
return body.decode(errors="replace").strip()
|
||||||
|
|
||||||
|
|
||||||
|
VIDEO_UI_PROCESSES = r'''for p in /proc/[0-9]*; do
|
||||||
|
[ -r "$p/cmdline" ] || continue
|
||||||
|
exe=$(readlink "$p/exe" 2>/dev/null) || continue
|
||||||
|
c=$(tr '\000' ' ' < "$p/cmdline")
|
||||||
|
case "${exe##*/}:$c" in
|
||||||
|
ltx-desktop:*--type=*)
|
||||||
|
printf '%s\n' "${p##*/}";;
|
||||||
|
esac
|
||||||
|
done'''
|
||||||
|
|
||||||
|
VIDEO_UI_STATE: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def video_ui_state(item: dict | None = None) -> str:
|
||||||
|
item = item or video_container()
|
||||||
|
if item.get("State") != "running":
|
||||||
|
return "unavailable"
|
||||||
|
output = docker_exec(item, VIDEO_UI_PROCESSES + r''' | while read p; do
|
||||||
|
state=$(sed -n 's/^State:[[:space:]]*\([A-Z]\).*/\1/p' "/proc/$p/status")
|
||||||
|
printf '%s\n' "$state"
|
||||||
|
done''')
|
||||||
|
states = [line.strip() for line in output.splitlines() if line.strip()]
|
||||||
|
if not states:
|
||||||
|
return "unavailable"
|
||||||
|
return "sleeping" if all(state == "T" for state in states) else "awake"
|
||||||
|
|
||||||
|
|
||||||
|
def cached_video_ui_state(item: dict | None = None) -> str:
|
||||||
|
global VIDEO_UI_STATE
|
||||||
|
if VIDEO_UI_STATE is None:
|
||||||
|
VIDEO_UI_STATE = video_ui_state(item)
|
||||||
|
return VIDEO_UI_STATE
|
||||||
|
|
||||||
|
|
||||||
|
def set_video_ui(awake: bool) -> dict:
|
||||||
|
global VIDEO_UI_STATE
|
||||||
|
with LOCK:
|
||||||
|
item = video_container()
|
||||||
|
if item.get("State") != "running":
|
||||||
|
raise RuntimeError("LTX-2 is not running")
|
||||||
|
signal = "CONT" if awake else "STOP"
|
||||||
|
docker_exec(item, VIDEO_UI_PROCESSES +
|
||||||
|
f''' | while read p; do kill -{signal} "$p"; done''')
|
||||||
|
time.sleep(0.25)
|
||||||
|
VIDEO_UI_STATE = video_ui_state(item)
|
||||||
|
return {"video_ui": VIDEO_UI_STATE}
|
||||||
|
|
||||||
|
|
||||||
def stop_video_if_configured() -> None:
|
def stop_video_if_configured() -> None:
|
||||||
if VIDEO_WORKER:
|
if VIDEO_WORKER:
|
||||||
stop_container(video_container(), timeout=30)
|
stop_container(video_container(), timeout=30)
|
||||||
@@ -482,6 +547,7 @@ def set_trellis_worker(running: bool) -> dict:
|
|||||||
|
|
||||||
def set_video_worker(running: bool) -> dict:
|
def set_video_worker(running: bool) -> dict:
|
||||||
"""Start LTX-2 exclusively, or stop it before another mode is loaded."""
|
"""Start LTX-2 exclusively, or stop it before another mode is loaded."""
|
||||||
|
global VIDEO_UI_STATE
|
||||||
with LOCK:
|
with LOCK:
|
||||||
item = video_container()
|
item = video_container()
|
||||||
if running:
|
if running:
|
||||||
@@ -496,8 +562,10 @@ def set_video_worker(running: bool) -> dict:
|
|||||||
stop_voice_tools()
|
stop_voice_tools()
|
||||||
stop_trellis_if_configured()
|
stop_trellis_if_configured()
|
||||||
start_container(item)
|
start_container(item)
|
||||||
|
VIDEO_UI_STATE = "awake"
|
||||||
else:
|
else:
|
||||||
stop_container(item, timeout=30)
|
stop_container(item, timeout=30)
|
||||||
|
VIDEO_UI_STATE = "unavailable"
|
||||||
return {"video_worker": VIDEO_WORKER,
|
return {"video_worker": VIDEO_WORKER,
|
||||||
"state": "running" if running else "stopped"}
|
"state": "running" if running else "stopped"}
|
||||||
|
|
||||||
@@ -660,6 +728,7 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
"trellis_health": trellis_health,
|
"trellis_health": trellis_health,
|
||||||
"video_worker": video.get("State", "disabled"),
|
"video_worker": video.get("State", "disabled"),
|
||||||
"video_health": video_health,
|
"video_health": video_health,
|
||||||
|
"video_ui": cached_video_ui_state(video) if video else "unavailable",
|
||||||
"profiles": {name: items.get(name, {}).get(
|
"profiles": {name: items.get(name, {}).get(
|
||||||
"State", "missing") for name in ALLOWED}})
|
"State", "missing") for name in ALLOWED}})
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -733,6 +802,13 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
log.exception("video worker transition failed")
|
log.exception("video worker transition failed")
|
||||||
self.reply(503, {"error": str(exc)})
|
self.reply(503, {"error": str(exc)})
|
||||||
return
|
return
|
||||||
|
if self.path in {"/workers/video/ui/sleep", "/workers/video/ui/wake"}:
|
||||||
|
try:
|
||||||
|
self.reply(200, set_video_ui(self.path.endswith("/wake")))
|
||||||
|
except Exception as exc:
|
||||||
|
log.exception("video UI transition failed")
|
||||||
|
self.reply(503, {"error": str(exc)})
|
||||||
|
return
|
||||||
worker_paths = {
|
worker_paths = {
|
||||||
"/workers/image/start": (IMAGE_WORKER, True),
|
"/workers/image/start": (IMAGE_WORKER, True),
|
||||||
"/workers/image/stop": (IMAGE_WORKER, False),
|
"/workers/image/stop": (IMAGE_WORKER, False),
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
FROM python:3.13.7-slim-bookworm
|
FROM python:3.13.7-slim-bookworm
|
||||||
|
|
||||||
ARG WHISPER_CPP_VERSION=v1.9.1
|
ARG WHISPER_CPP_VERSION=v1.9.4
|
||||||
|
|
||||||
RUN apt-get update \
|
RUN apt-get update \
|
||||||
&& apt-get install -y --no-install-recommends \
|
&& apt-get install -y --no-install-recommends \
|
||||||
|
|||||||
File diff suppressed because one or more lines are too long
@@ -1 +1 @@
|
|||||||
56381e407c0ccfb3a6f71e668a27a901001d22ce
|
b29c606
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ ARG LTX_DESKTOP_VERSION=1.2.7
|
|||||||
ARG LTX_DESKTOP_SHA512=d1d59027988a48490492feb42156665bbed511f187739a664ad326492fd8fc0ce43429537150eb6aea9a75a522f6353a40839f9b3ed0449fb720b7c13b091706
|
ARG LTX_DESKTOP_SHA512=d1d59027988a48490492feb42156665bbed511f187739a664ad326492fd8fc0ce43429537150eb6aea9a75a522f6353a40839f9b3ed0449fb720b7c13b091706
|
||||||
|
|
||||||
RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
|
RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
|
||||||
ca-certificates curl dbus-x11 ffmpeg libasound2t64 libatk-bridge2.0-0 \
|
build-essential ca-certificates curl dbus-x11 ffmpeg libasound2t64 libatk-bridge2.0-0 \
|
||||||
libatk1.0-0 libcups2 libdrm2 libgbm1 libgtk-3-0 libnss3 libx11-xcb1 \
|
libatk1.0-0 libcups2 libdrm2 libgbm1 libgtk-3-0 libnss3 libx11-xcb1 \
|
||||||
libxcomposite1 libxdamage1 libxfixes3 libxkbcommon0 libxrandr2 \
|
libxcomposite1 libxdamage1 libxfixes3 libxkbcommon0 libxrandr2 \
|
||||||
novnc openbox procps python3-websockify x11vnc xvfb \
|
novnc openbox procps python3-websockify x11vnc xvfb \
|
||||||
@@ -26,7 +26,8 @@ COPY entrypoint.sh /usr/local/bin/ltx-desktop-entrypoint
|
|||||||
COPY tar-no-owner.sh /usr/local/bin/tar
|
COPY tar-no-owner.sh /usr/local/bin/tar
|
||||||
RUN chmod 0755 /usr/local/bin/ltx-desktop-entrypoint /usr/local/bin/tar
|
RUN chmod 0755 /usr/local/bin/ltx-desktop-entrypoint /usr/local/bin/tar
|
||||||
|
|
||||||
ENV DISPLAY=:0 \
|
ENV CC=/usr/bin/gcc \
|
||||||
|
DISPLAY=:0 \
|
||||||
HOME=/data/home \
|
HOME=/data/home \
|
||||||
XDG_DATA_HOME=/data \
|
XDG_DATA_HOME=/data \
|
||||||
XDG_CONFIG_HOME=/data/config \
|
XDG_CONFIG_HOME=/data/config \
|
||||||
|
|||||||
@@ -0,0 +1,354 @@
|
|||||||
|
"""Local, catalog-aware prompt enhancement handler."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import random
|
||||||
|
import uuid
|
||||||
|
from threading import RLock
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from _routes._errors import HTTPError
|
||||||
|
from api_types import EnhancePromptRequest, EnhancePromptResponse, IcLoraCatalogItem, LoraCatalogItem
|
||||||
|
from handlers.base import StateHandlerBase
|
||||||
|
from handlers.generation_handler import GenerationHandler
|
||||||
|
from handlers.pipelines_handler import PipelinesHandler
|
||||||
|
from handlers.text_handler import TextHandler
|
||||||
|
from server_utils.media_validation import normalize_optional_path, validate_image_file
|
||||||
|
from services.gemini_text_client import resolve_gemini_model
|
||||||
|
from services.interfaces import PromptEnhancerPipeline
|
||||||
|
from services.lora_catalog import LoraCatalogProvider
|
||||||
|
from services.prompt_enhancement import (
|
||||||
|
build_audio_visual_caption_system_prompt,
|
||||||
|
build_conditioning_system_prompt,
|
||||||
|
build_ic_lora_enhancement_system_prompt,
|
||||||
|
build_image_edit_system_prompt,
|
||||||
|
build_image_generation_system_prompt,
|
||||||
|
build_keyframe_enhancement_system_prompt,
|
||||||
|
build_lora_enhancement_system_prompt,
|
||||||
|
build_template_fill_system_prompt,
|
||||||
|
enforce_trigger_placements,
|
||||||
|
fill_prompt_template,
|
||||||
|
parse_template_fill_response,
|
||||||
|
)
|
||||||
|
from services.prompt_enhancement.i2v_frames import KeyframeStill
|
||||||
|
from services.prompt_enhancer_pipeline.gemini_prompt_enhancer_pipeline import GeminiPromptEnhancerPipeline
|
||||||
|
from services.services_utils import get_device_type
|
||||||
|
from state.app_state_types import AppState
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from runtime_config.runtime_config import RuntimeConfig
|
||||||
|
|
||||||
|
# Deliberately independent of StateHandlerBase._resolve_seed(): that helper honors the app's
|
||||||
|
# reproducibility seed lock (and a fixed constant in dev mode), which would make every enhance
|
||||||
|
# call — including a redo — produce the exact same output. Enhancement is a quick, exploratory
|
||||||
|
# action where a fresh draw each call is the whole point.
|
||||||
|
_MAX_ENHANCE_SEED = 2147483647
|
||||||
|
|
||||||
|
|
||||||
|
class PromptEnhancementHandler(StateHandlerBase):
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
state: AppState,
|
||||||
|
lock: RLock,
|
||||||
|
generation_handler: GenerationHandler,
|
||||||
|
pipelines_handler: PipelinesHandler,
|
||||||
|
text_handler: TextHandler,
|
||||||
|
lora_catalog_provider: LoraCatalogProvider,
|
||||||
|
prompt_enhancer_pipeline_class: type[PromptEnhancerPipeline],
|
||||||
|
gemini_pipeline: GeminiPromptEnhancerPipeline,
|
||||||
|
config: RuntimeConfig,
|
||||||
|
) -> None:
|
||||||
|
super().__init__(state, lock, config)
|
||||||
|
self._generation = generation_handler
|
||||||
|
self._pipelines = pipelines_handler
|
||||||
|
self._text_handler = text_handler
|
||||||
|
self._lora_catalog_provider = lora_catalog_provider
|
||||||
|
self._prompt_enhancer_pipeline_class = prompt_enhancer_pipeline_class
|
||||||
|
self._gemini_pipeline = gemini_pipeline
|
||||||
|
|
||||||
|
def _random_seed(self) -> int:
|
||||||
|
return random.randint(0, _MAX_ENHANCE_SEED)
|
||||||
|
|
||||||
|
def enhance(self, req: EnhancePromptRequest) -> EnhancePromptResponse:
|
||||||
|
# Enhance never occupies the GPU slot (see PipelinesHandler.
|
||||||
|
# evict_gpu_pipeline_for_prompt_enhancement) but still needs to mutually exclude with
|
||||||
|
# generation and with itself — an abandoned/orphaned enhance call (e.g. the tab reloaded
|
||||||
|
# mid-request) must not race a Generate click, a second Enhance click, or a generation
|
||||||
|
# that's still loading its pipeline (reserved_generation_start covers that window; a bare
|
||||||
|
# is_generation_running() check does not — see its own docstring). The "api" generation
|
||||||
|
# slot gives us the mutual exclusion for free: it's the same bookkeeping every other
|
||||||
|
# handler already does, and it doesn't require gpu_slot to be set.
|
||||||
|
with self._generation.reserved_generation_start():
|
||||||
|
gemma_root: str | None = None
|
||||||
|
if req.provider == "local":
|
||||||
|
gemma_root = self._text_handler.resolve_prompt_enhancer_root_if_downloaded()
|
||||||
|
if gemma_root is None:
|
||||||
|
raise HTTPError(409, "LOCAL_TEXT_ENCODER_NOT_AVAILABLE")
|
||||||
|
elif not self.state.app_settings.gemini_api_key:
|
||||||
|
raise HTTPError(400, "GEMINI_API_KEY_MISSING")
|
||||||
|
|
||||||
|
generation_id = uuid.uuid4().hex[:8]
|
||||||
|
self._generation.start_api_generation(generation_id)
|
||||||
|
try:
|
||||||
|
enhanced = self._resolve_and_enhance(req, gemma_root)
|
||||||
|
except HTTPError as e:
|
||||||
|
self._generation.fail_generation(e.detail)
|
||||||
|
raise
|
||||||
|
except Exception as e:
|
||||||
|
self._generation.fail_generation(str(e))
|
||||||
|
raise HTTPError(500, str(e)) from e
|
||||||
|
|
||||||
|
self._generation.complete_generation(enhanced)
|
||||||
|
return EnhancePromptResponse(enhancedPrompt=enhanced)
|
||||||
|
|
||||||
|
def _resolve_and_enhance(self, req: EnhancePromptRequest, gemma_root: str | None) -> str:
|
||||||
|
if req.mediaType == "image":
|
||||||
|
# No catalog LoRA concept for images (validated at the request level) — always an
|
||||||
|
# explicit, image-domain system prompt, never the video-oriented generic fallback.
|
||||||
|
system_prompt = (
|
||||||
|
build_image_edit_system_prompt() if req.imagePath is not None
|
||||||
|
else build_image_generation_system_prompt()
|
||||||
|
)
|
||||||
|
return self._run_free_rewrite(req, system_prompt, gemma_root)
|
||||||
|
|
||||||
|
if req.icLoraId is not None:
|
||||||
|
ic_lora = self._lora_catalog_provider.get_ic_lora(req.icLoraId)
|
||||||
|
if ic_lora is None:
|
||||||
|
raise HTTPError(404, "LORA_CATALOG_ID_NOT_FOUND")
|
||||||
|
return self._enhance_ic_lora(ic_lora, req, gemma_root)
|
||||||
|
|
||||||
|
if req.loraCatalogIds:
|
||||||
|
loras: list[LoraCatalogItem] = []
|
||||||
|
for catalog_id in req.loraCatalogIds:
|
||||||
|
lora = self._lora_catalog_provider.get_lora(catalog_id)
|
||||||
|
if lora is None:
|
||||||
|
raise HTTPError(404, "LORA_CATALOG_ID_NOT_FOUND")
|
||||||
|
loras.append(lora)
|
||||||
|
return self._enhance_loras(loras, req, gemma_root)
|
||||||
|
|
||||||
|
if req.conditioningType is not None:
|
||||||
|
system_prompt = build_conditioning_system_prompt(req.conditioningType)
|
||||||
|
return self._run_free_rewrite(req, system_prompt, gemma_root)
|
||||||
|
|
||||||
|
return self._run_free_rewrite(req, self._default_video_system_prompt(req), gemma_root)
|
||||||
|
|
||||||
|
def _default_video_system_prompt(self, req: EnhancePromptRequest) -> str | None:
|
||||||
|
if req.keyframes:
|
||||||
|
return self._keyframe_system_prompt()
|
||||||
|
return self._video_system_prompt(t2v=req.imagePath is None)
|
||||||
|
|
||||||
|
def _keyframe_system_prompt(self) -> str:
|
||||||
|
spec = self._text_handler.active_ltx_model_spec()
|
||||||
|
audio_visual = spec is not None and spec.wants_audio_visual_captions
|
||||||
|
return build_keyframe_enhancement_system_prompt(audio_visual=audio_visual)
|
||||||
|
|
||||||
|
def _video_system_prompt(self, *, t2v: bool) -> str | None:
|
||||||
|
"""The active model's own caption style, or None to keep each provider's default.
|
||||||
|
|
||||||
|
Only the audio-visual generations (2.5) need this: their captions cover the soundscape,
|
||||||
|
which neither the generic Gemini fallback nor a 2.3-era prompt asks for.
|
||||||
|
"""
|
||||||
|
spec = self._text_handler.active_ltx_model_spec()
|
||||||
|
if spec is None or not spec.wants_audio_visual_captions:
|
||||||
|
return None
|
||||||
|
return build_audio_visual_caption_system_prompt(t2v=t2v)
|
||||||
|
|
||||||
|
def enhance_for_generation(
|
||||||
|
self,
|
||||||
|
prompt: str,
|
||||||
|
*,
|
||||||
|
image_path: str | None,
|
||||||
|
last_image_path: str | None = None,
|
||||||
|
keyframes: list[KeyframeStill] | None = None,
|
||||||
|
duration: int | None = None,
|
||||||
|
fps: int | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""Rewrite ``prompt`` on the local enhancer for a generation that's already started.
|
||||||
|
|
||||||
|
Only for the local text-encoding path: API encoding enhances server-side inside the
|
||||||
|
same call, so it never gets here. Deliberately not `enhance()` — the caller already
|
||||||
|
holds the generation slot, and there's no provider choice to make, only "is the
|
||||||
|
enhancer on disk".
|
||||||
|
|
||||||
|
Never raises: enhancement is a quality step, so a missing checkpoint or a failed
|
||||||
|
rewrite degrades to the prompt as typed rather than failing the generation.
|
||||||
|
"""
|
||||||
|
if not prompt.strip():
|
||||||
|
return prompt
|
||||||
|
gemma_root = self._text_handler.resolve_prompt_enhancer_root_if_downloaded()
|
||||||
|
if gemma_root is None:
|
||||||
|
logger.info("Skipping automatic enhancement: no local prompt enhancer downloaded")
|
||||||
|
return prompt
|
||||||
|
|
||||||
|
try:
|
||||||
|
pipeline = self._load_prompt_enhancer_pipeline(gemma_root)
|
||||||
|
system_prompt = (
|
||||||
|
self._keyframe_system_prompt()
|
||||||
|
if keyframes
|
||||||
|
else self._video_system_prompt(t2v=image_path is None)
|
||||||
|
)
|
||||||
|
seed = self._random_seed()
|
||||||
|
if image_path is not None or keyframes:
|
||||||
|
first_path = image_path or (keyframes[0][0] if keyframes else None)
|
||||||
|
assert first_path is not None
|
||||||
|
enhanced = pipeline.enhance_i2v(
|
||||||
|
prompt,
|
||||||
|
first_path,
|
||||||
|
system_prompt=system_prompt,
|
||||||
|
seed=seed,
|
||||||
|
last_image_path=None if keyframes else last_image_path,
|
||||||
|
keyframes=keyframes,
|
||||||
|
duration=duration,
|
||||||
|
fps=fps,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
enhanced = pipeline.enhance_t2v(prompt, system_prompt=system_prompt, seed=seed)
|
||||||
|
except Exception:
|
||||||
|
logger.warning("Automatic local enhancement failed; using the prompt as typed", exc_info=True)
|
||||||
|
return prompt
|
||||||
|
|
||||||
|
if not enhanced.strip():
|
||||||
|
return prompt
|
||||||
|
logger.info(
|
||||||
|
"Enhanced prompt locally for generation (%d -> %d chars): %s",
|
||||||
|
len(prompt),
|
||||||
|
len(enhanced),
|
||||||
|
enhanced,
|
||||||
|
)
|
||||||
|
return enhanced
|
||||||
|
|
||||||
|
def _enhance_loras(
|
||||||
|
self, loras: list[LoraCatalogItem], req: EnhancePromptRequest, gemma_root: str | None
|
||||||
|
) -> str:
|
||||||
|
# None of the plain LoRAs in the catalog have a prompt_template today (only IC-LoRAs
|
||||||
|
# do) — the multi-select path is always a free rewrite.
|
||||||
|
system_prompt = build_lora_enhancement_system_prompt(loras)
|
||||||
|
enhanced = self._run_free_rewrite(req, system_prompt, gemma_root)
|
||||||
|
return enforce_trigger_placements(enhanced, loras)
|
||||||
|
|
||||||
|
def _enhance_ic_lora(
|
||||||
|
self, ic_lora: IcLoraCatalogItem, req: EnhancePromptRequest, gemma_root: str | None
|
||||||
|
) -> str:
|
||||||
|
if ic_lora.prompt_template is not None:
|
||||||
|
return self._run_template_fill(ic_lora, req, gemma_root)
|
||||||
|
system_prompt = build_ic_lora_enhancement_system_prompt(ic_lora)
|
||||||
|
enhanced = self._run_free_rewrite(req, system_prompt, gemma_root)
|
||||||
|
return enforce_trigger_placements(enhanced, [ic_lora])
|
||||||
|
|
||||||
|
def _run_free_rewrite(
|
||||||
|
self, req: EnhancePromptRequest, system_prompt: str | None, gemma_root: str | None
|
||||||
|
) -> str:
|
||||||
|
# Reject an invalid/unreadable/oversized path before it reaches either provider — the
|
||||||
|
# API path in particular would otherwise base64-encode and ship arbitrary file bytes to
|
||||||
|
# a third-party API with no gate at all.
|
||||||
|
keyframes = self._validated_keyframes(req)
|
||||||
|
image_path = (
|
||||||
|
None if keyframes else normalize_optional_path(req.imagePath)
|
||||||
|
)
|
||||||
|
last_image_path = (
|
||||||
|
None
|
||||||
|
if req.mediaType == "image" or keyframes
|
||||||
|
else normalize_optional_path(req.lastImagePath)
|
||||||
|
)
|
||||||
|
if image_path is not None:
|
||||||
|
validate_image_file(image_path)
|
||||||
|
if last_image_path is not None:
|
||||||
|
validate_image_file(last_image_path)
|
||||||
|
|
||||||
|
seed = self._random_seed()
|
||||||
|
first_path = image_path or (keyframes[0][0] if keyframes else None)
|
||||||
|
if req.provider == "api":
|
||||||
|
resolved_model = resolve_gemini_model(self.state.app_settings.gemini_model)
|
||||||
|
logger.info("Enhancing prompt via Gemini API (%s)", resolved_model)
|
||||||
|
api_key = self.state.app_settings.gemini_api_key
|
||||||
|
if first_path is not None:
|
||||||
|
return self._gemini_pipeline.enhance_i2v(
|
||||||
|
req.prompt,
|
||||||
|
first_path,
|
||||||
|
system_prompt=system_prompt,
|
||||||
|
seed=seed,
|
||||||
|
api_key=api_key,
|
||||||
|
model=resolved_model,
|
||||||
|
last_image_path=last_image_path,
|
||||||
|
keyframes=keyframes,
|
||||||
|
duration=req.duration,
|
||||||
|
fps=req.fps,
|
||||||
|
)
|
||||||
|
return self._gemini_pipeline.enhance_t2v(
|
||||||
|
req.prompt,
|
||||||
|
system_prompt=system_prompt,
|
||||||
|
seed=seed,
|
||||||
|
api_key=api_key,
|
||||||
|
model=resolved_model,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info("Enhancing prompt via local Gemma")
|
||||||
|
assert gemma_root is not None
|
||||||
|
pipeline = self._load_prompt_enhancer_pipeline(gemma_root)
|
||||||
|
if first_path is not None:
|
||||||
|
return pipeline.enhance_i2v(
|
||||||
|
req.prompt,
|
||||||
|
first_path,
|
||||||
|
system_prompt=system_prompt,
|
||||||
|
seed=seed,
|
||||||
|
last_image_path=last_image_path,
|
||||||
|
keyframes=keyframes,
|
||||||
|
duration=req.duration,
|
||||||
|
fps=req.fps,
|
||||||
|
)
|
||||||
|
return pipeline.enhance_t2v(req.prompt, system_prompt=system_prompt, seed=seed)
|
||||||
|
|
||||||
|
def _validated_keyframes(self, req: EnhancePromptRequest) -> list[KeyframeStill] | None:
|
||||||
|
if req.mediaType == "image" or not req.keyframes:
|
||||||
|
return None
|
||||||
|
frames: list[KeyframeStill] = []
|
||||||
|
for keyframe in req.keyframes:
|
||||||
|
path = normalize_optional_path(keyframe.imagePath)
|
||||||
|
if path is None:
|
||||||
|
raise HTTPError(400, "Each keyframe requires an image path")
|
||||||
|
validate_image_file(path)
|
||||||
|
frames.append((path, keyframe.frameIndex, keyframe.strength))
|
||||||
|
frames.sort(key=lambda item: item[1])
|
||||||
|
return frames
|
||||||
|
|
||||||
|
def _run_template_fill(
|
||||||
|
self, ic_lora: IcLoraCatalogItem, req: EnhancePromptRequest, gemma_root: str | None
|
||||||
|
) -> str:
|
||||||
|
# req.imagePath is intentionally unused here — template fill is always a text-only
|
||||||
|
# enhance_t2v call (the fixed template scaffold carries no reference-image slot), unlike
|
||||||
|
# the free-rewrite IC-LoRA path below it, which does route an image through enhance_i2v.
|
||||||
|
assert ic_lora.prompt_template is not None
|
||||||
|
system_prompt = build_template_fill_system_prompt(ic_lora)
|
||||||
|
seed = self._random_seed()
|
||||||
|
try:
|
||||||
|
if req.provider == "api":
|
||||||
|
resolved_model = resolve_gemini_model(self.state.app_settings.gemini_model)
|
||||||
|
logger.info("Enhancing prompt via Gemini API (%s)", resolved_model)
|
||||||
|
raw = self._gemini_pipeline.enhance_t2v(
|
||||||
|
req.prompt,
|
||||||
|
system_prompt=system_prompt,
|
||||||
|
seed=seed,
|
||||||
|
api_key=self.state.app_settings.gemini_api_key,
|
||||||
|
model=resolved_model,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
logger.info("Enhancing prompt via local Gemma")
|
||||||
|
assert gemma_root is not None
|
||||||
|
pipeline = self._load_prompt_enhancer_pipeline(gemma_root)
|
||||||
|
raw = pipeline.enhance_t2v(req.prompt, system_prompt=system_prompt, seed=seed)
|
||||||
|
values = parse_template_fill_response(raw, set(ic_lora.prompt_template.placeholders))
|
||||||
|
return fill_prompt_template(ic_lora.prompt_template, values)
|
||||||
|
except ValueError as e:
|
||||||
|
raise HTTPError(500, f"PROMPT_TEMPLATE_FILL_FAILED: {e}") from e
|
||||||
|
|
||||||
|
def _load_prompt_enhancer_pipeline(self, gemma_root: str) -> PromptEnhancerPipeline:
|
||||||
|
self._pipelines.evict_gpu_pipeline_for_prompt_enhancement()
|
||||||
|
device = os.getenv("LTX_PROMPT_ENHANCER_DEVICE", "").strip()
|
||||||
|
if not device:
|
||||||
|
device = get_device_type(self.config.device)
|
||||||
|
logger.info("Loading local prompt enhancer on %s", device)
|
||||||
|
return self._prompt_enhancer_pipeline_class.create(gemma_root, device)
|
||||||
@@ -0,0 +1,179 @@
|
|||||||
|
"""FastAPI app factory decoupled from runtime bootstrap side effects."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import base64
|
||||||
|
import hmac
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
from collections.abc import Awaitable, Callable
|
||||||
|
from typing import TYPE_CHECKING, Any
|
||||||
|
|
||||||
|
from fastapi import FastAPI, Request
|
||||||
|
from fastapi.exceptions import RequestValidationError
|
||||||
|
from fastapi.middleware.cors import CORSMiddleware
|
||||||
|
from fastapi.responses import JSONResponse
|
||||||
|
from starlette.exceptions import HTTPException as StarletteHTTPException
|
||||||
|
from starlette.responses import Response as StarletteResponse
|
||||||
|
|
||||||
|
from _routes._errors import HTTPError, build_http_error_response
|
||||||
|
from _routes.generation import router as generation_router
|
||||||
|
from _routes.hf_auth import router as hf_auth_router
|
||||||
|
from _routes.health import router as health_router
|
||||||
|
from _routes.ic_lora import router as ic_lora_router
|
||||||
|
from _routes.lora_catalog import router as lora_catalog_router
|
||||||
|
from _routes.image_gen import router as image_gen_router
|
||||||
|
from _routes.prompt_enhancement import router as prompt_enhancement_router
|
||||||
|
from _routes.models import router as models_router
|
||||||
|
from _routes.suggest_gap_prompt import router as suggest_gap_prompt_router
|
||||||
|
from _routes.retake import router as retake_router
|
||||||
|
from _routes.extend import router as extend_router
|
||||||
|
from _routes.runtime_policy import router as runtime_policy_router
|
||||||
|
from _routes.settings import router as settings_router
|
||||||
|
from api_types import HTTPErrorResponse
|
||||||
|
from logging_policy import log_http_error, log_unhandled_exception
|
||||||
|
from state import init_state_service
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from app_handler import AppHandler
|
||||||
|
|
||||||
|
DEFAULT_ALLOWED_ORIGINS: list[str] = [
|
||||||
|
"http://localhost:5173",
|
||||||
|
"http://127.0.0.1:5173",
|
||||||
|
]
|
||||||
|
|
||||||
|
DEFAULT_ERROR_RESPONSES: dict[int | str, dict[str, Any]] = {
|
||||||
|
"4XX": {
|
||||||
|
"model": HTTPErrorResponse,
|
||||||
|
"description": "Client Error",
|
||||||
|
},
|
||||||
|
"5XX": {
|
||||||
|
"model": HTTPErrorResponse,
|
||||||
|
"description": "Server Error",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def create_app(
|
||||||
|
*,
|
||||||
|
handler: "AppHandler",
|
||||||
|
allowed_origins: list[str] | None = None,
|
||||||
|
title: str = "LTX-2 Video Generation Server",
|
||||||
|
auth_token: str = "",
|
||||||
|
admin_token: str = "",
|
||||||
|
) -> FastAPI:
|
||||||
|
"""Create a configured FastAPI app bound to the provided handler."""
|
||||||
|
init_state_service(handler)
|
||||||
|
|
||||||
|
app = FastAPI(title=title, responses=DEFAULT_ERROR_RESPONSES)
|
||||||
|
remote_auth_token = os.environ.get("LTX_REMOTE_AUTH_TOKEN", "").strip()
|
||||||
|
remote_token_file = os.environ.get("LTX_REMOTE_AUTH_TOKEN_FILE", "").strip()
|
||||||
|
if remote_token_file:
|
||||||
|
remote_auth_token = Path(remote_token_file).read_text(encoding="utf-8").strip()
|
||||||
|
app.state.admin_token = admin_token # type: ignore[attr-defined]
|
||||||
|
app.add_middleware(
|
||||||
|
CORSMiddleware,
|
||||||
|
allow_origins=allowed_origins or DEFAULT_ALLOWED_ORIGINS,
|
||||||
|
allow_methods=["*"],
|
||||||
|
allow_headers=["*"],
|
||||||
|
)
|
||||||
|
|
||||||
|
@app.middleware("http")
|
||||||
|
async def _auth_middleware( # pyright: ignore[reportUnusedFunction]
|
||||||
|
request: Request,
|
||||||
|
call_next: Callable[[Request], Awaitable[StarletteResponse]],
|
||||||
|
) -> StarletteResponse:
|
||||||
|
if not auth_token:
|
||||||
|
return await call_next(request)
|
||||||
|
if request.method == "OPTIONS":
|
||||||
|
return await call_next(request)
|
||||||
|
if request.url.path == "/api/auth/huggingface/callback":
|
||||||
|
return await call_next(request)
|
||||||
|
def _token_matches(candidate: str) -> bool:
|
||||||
|
return hmac.compare_digest(candidate, auth_token) or (
|
||||||
|
bool(remote_auth_token) and hmac.compare_digest(candidate, remote_auth_token)
|
||||||
|
)
|
||||||
|
|
||||||
|
# WebSocket: check query param
|
||||||
|
if request.headers.get("upgrade", "").lower() == "websocket":
|
||||||
|
if _token_matches(request.query_params.get("token", "")):
|
||||||
|
return await call_next(request)
|
||||||
|
return JSONResponse(
|
||||||
|
status_code=401,
|
||||||
|
content=build_http_error_response(401, "Unauthorized").model_dump(),
|
||||||
|
)
|
||||||
|
# HTTP: Bearer or Basic auth
|
||||||
|
auth_header = request.headers.get("authorization", "")
|
||||||
|
if auth_header.startswith("Bearer ") and _token_matches(auth_header[7:]):
|
||||||
|
return await call_next(request)
|
||||||
|
if auth_header.startswith("Basic "):
|
||||||
|
try:
|
||||||
|
decoded = base64.b64decode(auth_header[6:]).decode()
|
||||||
|
_, _, password = decoded.partition(":")
|
||||||
|
if _token_matches(password):
|
||||||
|
return await call_next(request)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return JSONResponse(
|
||||||
|
status_code=401,
|
||||||
|
content=build_http_error_response(401, "Unauthorized").model_dump(),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def _route_http_error_handler(request: Request, exc: Exception) -> JSONResponse:
|
||||||
|
if isinstance(exc, HTTPError):
|
||||||
|
log_http_error(request, exc)
|
||||||
|
return JSONResponse(status_code=exc.status_code, content=exc.response.model_dump())
|
||||||
|
return JSONResponse(
|
||||||
|
status_code=500,
|
||||||
|
content=build_http_error_response(500, str(exc)).model_dump(),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def _starlette_http_error_handler(request: Request, exc: Exception) -> JSONResponse:
|
||||||
|
if isinstance(exc, StarletteHTTPException):
|
||||||
|
return JSONResponse(
|
||||||
|
status_code=exc.status_code,
|
||||||
|
content=build_http_error_response(exc.status_code, exc.detail).model_dump(),
|
||||||
|
)
|
||||||
|
return JSONResponse(
|
||||||
|
status_code=500,
|
||||||
|
content=build_http_error_response(500, str(exc)).model_dump(),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def _validation_error_handler(request: Request, exc: Exception) -> JSONResponse:
|
||||||
|
if isinstance(exc, RequestValidationError):
|
||||||
|
return JSONResponse(
|
||||||
|
status_code=422,
|
||||||
|
content=build_http_error_response(422, str(exc)).model_dump(),
|
||||||
|
)
|
||||||
|
return JSONResponse(
|
||||||
|
status_code=422,
|
||||||
|
content=build_http_error_response(422, str(exc)).model_dump(),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def _route_generic_error_handler(request: Request, exc: Exception) -> JSONResponse:
|
||||||
|
log_unhandled_exception(request, exc)
|
||||||
|
return JSONResponse(
|
||||||
|
status_code=500,
|
||||||
|
content=build_http_error_response(500, str(exc)).model_dump(),
|
||||||
|
)
|
||||||
|
|
||||||
|
app.add_exception_handler(RequestValidationError, _validation_error_handler)
|
||||||
|
app.add_exception_handler(HTTPError, _route_http_error_handler)
|
||||||
|
app.add_exception_handler(StarletteHTTPException, _starlette_http_error_handler)
|
||||||
|
app.add_exception_handler(Exception, _route_generic_error_handler)
|
||||||
|
|
||||||
|
app.include_router(health_router)
|
||||||
|
app.include_router(generation_router)
|
||||||
|
app.include_router(models_router)
|
||||||
|
app.include_router(settings_router)
|
||||||
|
app.include_router(image_gen_router)
|
||||||
|
app.include_router(suggest_gap_prompt_router)
|
||||||
|
app.include_router(retake_router)
|
||||||
|
app.include_router(extend_router)
|
||||||
|
app.include_router(ic_lora_router)
|
||||||
|
app.include_router(lora_catalog_router)
|
||||||
|
app.include_router(prompt_enhancement_router)
|
||||||
|
app.include_router(runtime_policy_router)
|
||||||
|
app.include_router(hf_auth_router)
|
||||||
|
|
||||||
|
return app
|
||||||
Reference in New Issue
Block a user