Add configurable TTS and STT residency policies
This commit is contained in:
1 parent
90a97567f3
commit
b2256647e9
9 files changed
+209
-40
No files matched your search
+41
-16
@@ -6,9 +6,9 @@ from image_test import probe,cgroup_headroom
|
||||
WORKER=Path(__file__).with_name('tts_worker.py')
|
||||
class TTSTests:
|
||||
def __init__(self,root,profiles,runtime):
|
||||
self.root=Path(root);self.profiles=profiles;self.runtime=runtime;self.lock=threading.RLock();self.job=None;self.process=None;self.gpu_uuid=None;self.cancel=threading.Event();self.acquire=lambda wait=False:lambda:None
|
||||
self.root=Path(root);self.profiles=profiles;self.runtime=runtime;self.lock=threading.RLock();self.job=None;self.process=None;self.gpu_uuid=None;self.cancel=threading.Event();self.acquire=lambda wait=False:lambda:None;self.policy='auto';self.warming=False
|
||||
def status(self):
|
||||
with self.lock:return dict(job=dict(self.job) if self.job else None,runtime=self.runtime.status(),loaded=bool(self.process and self.process.poll() is None),gpu_uuid=self.gpu_uuid)
|
||||
with self.lock:return dict(job=dict(self.job) if self.job else None,runtime=self.runtime.status(),loaded=bool(self.process and self.process.poll() is None),gpu_uuid=self.gpu_uuid,warming=self.warming,policy=self.policy)
|
||||
def blockers(self,p):
|
||||
if p['model']['repo']!=REPO or p['model']['file']!='model.safetensors' or p['model'].get('revision')!=REVISION:return ['Für diese TTS-Variante fehlt die Anbindung in Deck. Unter TTS → Einrichten kannst du ausdrücklich auf die unterstützte Variante mit eingebauten Stimmen wechseln. Die ursprüngliche Datei bleibt erhalten.']
|
||||
return [] if self.runtime.status()['installed'] else ['TTS-Laufzeit unter TTS → Einrichten installieren.']
|
||||
@@ -25,6 +25,7 @@ class TTSTests:
|
||||
release=self.acquire(wait)
|
||||
try:
|
||||
with self.lock:
|
||||
if self.warming:raise ValueError('TTS wird gerade vorgeladen. Gleich erneut versuchen.')
|
||||
if self.job and self.job['state']=='running':raise ValueError('TTS-Auftrag läuft bereits.')
|
||||
p=next((p for p in self.profiles.status()['profiles'] if p['id']==profile_id and p['kind']=='audio'),None)
|
||||
if not p or not p['runnable']:raise ValueError('TTS-Profil noch nicht eingerichtet.')
|
||||
@@ -40,18 +41,7 @@ class TTSTests:
|
||||
def _run(self,ident,text,speaker,language,speed,gpu,release):
|
||||
directory=self.root/ident;process=None;state='failed';phase='TTS-Ausführung fehlgeschlagen'
|
||||
try:
|
||||
directory.mkdir(mode=0o700);python,model=self.runtime.paths()
|
||||
with self.lock:
|
||||
if self.cancel.is_set():raise InterruptedError()
|
||||
process=self.process if self.process and self.process.poll() is None else None
|
||||
new_process=process is None
|
||||
if process is None:
|
||||
env={k:v for k,v in os.environ.items() if not k.startswith(('HF_','LLAMA_'))};env.update(CUDA_VISIBLE_DEVICES=gpu['uuid'],HF_HUB_OFFLINE='1',TRANSFORMERS_OFFLINE='1',OMP_NUM_THREADS='2',HOME=str(self.root))
|
||||
process=subprocess.Popen([str(python),str(WORKER),str(model)],stdin=subprocess.PIPE,stdout=subprocess.PIPE,stderr=subprocess.DEVNULL,start_new_session=True,env=env,text=True,bufsize=1);self.process=process;self.gpu_uuid=gpu['uuid']
|
||||
if new_process:
|
||||
if not select.select([process.stdout],[],[],180)[0]:raise ValueError('Zeitlimit beim Laden des TTS-Modells.')
|
||||
ready=json.loads(process.stdout.readline())
|
||||
if not ready.get('ready'):raise ValueError(ready.get('error','TTS-Modell konnte nicht geladen werden.'))
|
||||
directory.mkdir(mode=0o700);process=self._ensure(gpu)
|
||||
with self.lock:self.job['phase']='Sprache wird auf der RTX 3060 erzeugt'
|
||||
process.stdin.write(json.dumps(dict(output=str(directory),text=text,speaker=speaker,language=language,speed=speed))+'\n');process.stdin.flush()
|
||||
end=time.monotonic()+600
|
||||
@@ -66,14 +56,49 @@ class TTSTests:
|
||||
if not result['ok']:raise ValueError(result['error'])
|
||||
wav=directory/'result.wav'
|
||||
if not wav.exists() or wav.stat().st_size>32*1024**2:raise ValueError('Ungültige Audioausgabe.')
|
||||
state='complete';phase='Sprache fertig · Modell bleibt geladen'
|
||||
state='complete';phase='Sprache fertig · '+('Modell entladen' if self.policy=='per_request' else 'Modell bleibt geladen')
|
||||
except InterruptedError:state='cancelled';phase='Sprachgenerierung abgebrochen'
|
||||
except Exception as exc:phase=str(exc) if isinstance(exc,ValueError) else 'Sprachgenerierung fehlgeschlagen. Laufzeit und Ressourcen prüfen.'
|
||||
finally:
|
||||
if state!='complete':self.unload_idle()
|
||||
if state!='complete' or self.policy=='per_request':self.unload_idle()
|
||||
if state=='complete':(directory/'complete').touch()
|
||||
with self.lock:self.job.update(state=state,phase=phase,finished_at=time.time())
|
||||
release()
|
||||
def _ensure(self,gpu):
|
||||
python,model=self.runtime.paths()
|
||||
with self.lock:
|
||||
if self.cancel.is_set():raise InterruptedError()
|
||||
process=self.process if self.process and self.process.poll() is None else None
|
||||
if process:return process
|
||||
self.root.mkdir(parents=True,exist_ok=True,mode=0o700)
|
||||
env={k:v for k,v in os.environ.items() if not k.startswith(('HF_','LLAMA_'))};env.update(CUDA_VISIBLE_DEVICES=gpu['uuid'],HF_HUB_OFFLINE='1',TRANSFORMERS_OFFLINE='1',OMP_NUM_THREADS='2',HOME=str(self.root))
|
||||
process=subprocess.Popen([str(python),str(WORKER),str(model)],stdin=subprocess.PIPE,stdout=subprocess.PIPE,stderr=subprocess.DEVNULL,start_new_session=True,env=env,text=True,bufsize=1);self.process=process;self.gpu_uuid=gpu['uuid']
|
||||
if not select.select([process.stdout],[],[],180)[0]:raise ValueError('Zeitlimit beim Laden des TTS-Modells.')
|
||||
if self.cancel.is_set():raise InterruptedError()
|
||||
ready=json.loads(process.stdout.readline())
|
||||
if not ready.get('ready'):raise ValueError(ready.get('error','TTS-Modell konnte nicht geladen werden.'))
|
||||
return process
|
||||
def warm(self,profile_id):
|
||||
release=self.acquire(True)
|
||||
try:
|
||||
with self.lock:
|
||||
if self.job and self.job.get('state')=='running':return False
|
||||
if self.process and self.process.poll() is None:return True
|
||||
if self.warming:return False
|
||||
p=next((p for p in self.profiles.status()['profiles'] if p['id']==profile_id and p['kind']=='audio' and p['runnable']),None)
|
||||
if not p:raise ValueError('Das ausgewählte TTS-Profil ist nicht ausführbar.')
|
||||
self.warming=True;self.cancel.clear()
|
||||
try:
|
||||
gpu=next((g for g in probe() if 'RTX 3060' in g['name'] and not g['processes'] and g['free_mib']>7000),None)
|
||||
if not gpu:raise ValueError('Für dauerhaftes TTS sind mindestens 7 GiB freie RTX-3060-VRAM nötig.')
|
||||
headroom=cgroup_headroom()
|
||||
if headroom is not None and headroom<8*1024**3:raise ValueError('Zu wenig freier Deck-RAM für TTS.')
|
||||
self._ensure(gpu);return True
|
||||
except Exception:
|
||||
self.unload_idle();raise
|
||||
finally:
|
||||
with self.lock:self.warming=False
|
||||
finally:release()
|
||||
def unload_idle(self):
|
||||
with self.lock:
|
||||
process=self.process;self.process=None;self.gpu_uuid=None
|
||||
|
||||
Reference in new issue
Block a user