Release FLUX CUDA context immediately
This commit is contained in:
@@ -6,6 +6,7 @@ from __future__ import annotations
|
||||
import gc
|
||||
import json
|
||||
import os
|
||||
import signal
|
||||
import time
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
@@ -21,6 +22,11 @@ LOAD_SECONDS = 0.0
|
||||
if len(TOKEN) < 32:
|
||||
raise RuntimeError("WORKER_TOKEN is missing or too short")
|
||||
|
||||
# The container is intentionally disposable. After the router has received a
|
||||
# completed response and saved image, an immediate process exit releases the
|
||||
# CUDA context much faster than Python/PyTorch interpreter teardown.
|
||||
signal.signal(signal.SIGTERM, lambda *_: os._exit(0))
|
||||
|
||||
|
||||
def load_pipeline() -> None:
|
||||
global PIPE, LOAD_SECONDS
|
||||
|
||||
Reference in New Issue
Block a user