Add explicit HYPIR restoration profile

This commit is contained in:
Mikei386
2026-09-07 22:52:59 +02:00
parent 2ae61baec7
commit 118e32005e
11 changed files with 593 additions and 32 deletions
@@ -25,6 +25,7 @@ ALLOWED = tuple(x.strip() for x in os.environ.get(
LABEL_KEY = "com.mike-ai.llama-profile"
IMAGE_LABEL_KEY = "com.mike-ai.image-worker"
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore")
TTS_LABEL_KEY = "com.mike-ai.tts-worker"
TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3")
LOCK = threading.Lock()
@@ -70,15 +71,22 @@ def labelled_containers(label: str) -> list[dict]:
return json.loads(body)
def image_container() -> dict:
def image_container(kind: str = IMAGE_WORKER) -> dict:
matches = [item for item in labelled_containers(IMAGE_LABEL_KEY)
if item.get("Labels", {}).get(IMAGE_LABEL_KEY) == IMAGE_WORKER]
if item.get("Labels", {}).get(IMAGE_LABEL_KEY) == kind]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one image worker {IMAGE_WORKER!r}, found {len(matches)}")
f"expected exactly one image worker {kind!r}, found {len(matches)}")
return matches[0]
def image_containers() -> list[dict]:
"""All allowlisted GPU workers that must never overlap an LLM."""
allowed = {IMAGE_WORKER, RESTORE_WORKER}
return [item for item in labelled_containers(IMAGE_LABEL_KEY)
if item.get("Labels", {}).get(IMAGE_LABEL_KEY) in allowed]
def tts_container() -> dict:
matches = [item for item in labelled_containers(TTS_LABEL_KEY)
if item.get("Labels", {}).get(TTS_LABEL_KEY) == TTS_WORKER]
@@ -113,9 +121,11 @@ def stop_inference() -> dict:
return {"active_profile": None, "previous_profile": previous}
def set_image_worker(running: bool) -> dict:
def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
if kind not in {IMAGE_WORKER, RESTORE_WORKER}:
raise ValueError("worker is not allowlisted")
with LOCK:
item = image_container()
item = image_container(kind)
if running:
# The image worker may never overlap a llama profile on the 5080.
for profile_item in containers().values():
@@ -123,6 +133,9 @@ def set_image_worker(running: bool) -> dict:
# The 9B beta text encoder temporarily borrows the RTX 3060 from
# Qwen3-TTS. The gateway retains Piper as a fallback meanwhile.
stop_container(tts_container(), timeout=30)
for other in image_containers():
if other["Id"] != item["Id"]:
stop_container(other, timeout=20)
start_container(item)
else:
# CUDA/PyTorch may not react promptly to SIGTERM after an OOM.
@@ -131,7 +144,8 @@ def set_image_worker(running: bool) -> dict:
# TTS is restored by the following profile activation. Keeping it
# stopped here lets the router verify that both GPUs really
# released the image model before Qwen and TTS are reloaded.
return {"image_worker": "running" if running else "stopped"}
return {"image_worker": kind,
"state": "running" if running else "stopped"}
def active_profile(items: dict[str, dict] | None = None) -> str | None:
@@ -147,7 +161,8 @@ def activate(profile: str) -> dict:
raise ValueError("profile is not allowlisted")
with LOCK:
# Defensive mutual exclusion even if a caller bypasses the router.
stop_container(image_container())
for worker in image_containers():
stop_container(worker)
start_container(tts_container())
items = containers()
missing = [name for name in ALLOWED if name not in items]
@@ -230,9 +245,16 @@ class Handler(BaseHTTPRequestHandler):
log.exception("stopping inference failed")
self.reply(503, {"error": str(exc)})
return
if self.path in ("/workers/image/start", "/workers/image/stop"):
worker_paths = {
"/workers/image/start": (IMAGE_WORKER, True),
"/workers/image/stop": (IMAGE_WORKER, False),
"/workers/restore/start": (RESTORE_WORKER, True),
"/workers/restore/stop": (RESTORE_WORKER, False),
}
if self.path in worker_paths:
try:
self.reply(200, set_image_worker(self.path.endswith("/start")))
kind, running = worker_paths[self.path]
self.reply(200, set_image_worker(running, kind))
except Exception as exc:
log.exception("image worker transition failed")
self.reply(503, {"error": str(exc)})