Release FLUX CUDA context immediately
This commit is contained in:
@@ -6,6 +6,7 @@ from __future__ import annotations
|
|||||||
import gc
|
import gc
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
|
import signal
|
||||||
import time
|
import time
|
||||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -21,6 +22,11 @@ LOAD_SECONDS = 0.0
|
|||||||
if len(TOKEN) < 32:
|
if len(TOKEN) < 32:
|
||||||
raise RuntimeError("WORKER_TOKEN is missing or too short")
|
raise RuntimeError("WORKER_TOKEN is missing or too short")
|
||||||
|
|
||||||
|
# The container is intentionally disposable. After the router has received a
|
||||||
|
# completed response and saved image, an immediate process exit releases the
|
||||||
|
# CUDA context much faster than Python/PyTorch interpreter teardown.
|
||||||
|
signal.signal(signal.SIGTERM, lambda *_: os._exit(0))
|
||||||
|
|
||||||
|
|
||||||
def load_pipeline() -> None:
|
def load_pipeline() -> None:
|
||||||
global PIPE, LOAD_SECONDS
|
global PIPE, LOAD_SECONDS
|
||||||
|
|||||||
Reference in New Issue
Block a user