Files
Athena-Deck/voxcpm_test.py
T

92 lines
6.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""VoxCPM2 speech and reference-voice generation, serialized by Deck GPU scheduler."""
import io,json,os,shutil,signal,subprocess,threading,time,uuid,wave
from pathlib import Path
from tts_test import TTSTests
from image_test import probe,cgroup_headroom
from profiles import voice_recipe
WORKER=Path(__file__).with_name('voxcpm_worker.py')
def validate_reference(audio):
if not isinstance(audio,bytes) or not 0<len(audio)<=8*1024**2:raise ValueError('Referenz als WAV bis 8 MiB hochladen.')
try:
with wave.open(io.BytesIO(audio)) as w:
if w.getnchannels() not in (1,2) or w.getsampwidth()!=2 or not 8000<=w.getframerate()<=96000 or not 1<=w.getnframes()/w.getframerate()<=30:raise ValueError('Referenz: PCM-16-WAV, mono/stereo, 1–30 Sekunden erforderlich.')
except (wave.Error,EOFError):raise ValueError('Ungültige WAV-Datei. PCM-16-WAV verwenden.') from None
class VoxCPMTests(TTSTests):
def blockers(self,p):
if not voice_recipe(p['model']):return ['Diese Stimm-Modellfamilie wird noch nicht unterstützt.']
errors=[]
if not self.runtime.status()['installed']:errors.append('VoxCPM-Laufzeit unter Laufzeiten → VoxCPM installieren.')
for role,info in voice_recipe(p['model']).items():
try:self.profiles._component(p['model'],role,p.get('components',{}).get(role))
except ValueError:errors.append(info['label']+' fehlt oder ist noch nicht zugeordnet. Unter Einrichtung & Komponenten ergänzen.')
return errors
def stage(self,p,root):
model=root/'model';model.mkdir();entries=[p['model']]+[self.profiles._component(p['model'],r,p.get('components',{}).get(r)) for r in voice_recipe(p['model'])]
for entry in entries:
src=self.profiles.catalog.root/entry['id']/('model'+Path(entry['file']).suffix)
(model/Path(entry['file']).name).symlink_to(src.resolve())
def start(self,profile_id,text,device='5080',cfg=2.0,steps=10,seed=42,reference=None,wait=False):
if not isinstance(text,str) or not 1<=len(text.strip())<=1000:raise ValueError('Text mit 1–1000 Zeichen erforderlich.')
if device not in ('5080','3060'):raise ValueError('RTX 5080 oder RTX 3060 wählen.')
if type(cfg) not in (int,float) or not .5<=cfg<=5 or type(steps)!=int or not 1<=steps<=50 or type(seed)!=int or not 0<=seed<=2147483647:raise ValueError('Ungültige Guidance, Schritte oder Seed.')
if reference is not None:validate_reference(reference)
p=next((p for p in self.profiles.status()['profiles'] if p['id']==profile_id),None)
if not p or not p['model'] or not voice_recipe(p['model']):raise ValueError('VoxCPM2-Profil nicht gefunden.')
errors=self.blockers(p)
if errors:raise ValueError(' '.join(errors))
# Claim local job before scheduler admission to prevent races between two callers.
with self.lock:
if self.job and self.job['state']=='running':raise ValueError('Stimmgenerierung läuft bereits.')
self.cancel.clear();ident=uuid.uuid4().hex;self.job=dict(id=ident,state='running',phase='GPU wird reserviert',profile_id=profile_id,started_at=time.time())
release=None
try:
release=self.acquire(wait)
gpu=next((g for g in probe() if 'RTX '+device in g['name'] and not g['processes'] and g['free_mib']>=6500),None)
if not gpu:raise ValueError('RTX '+device+' benötigt mindestens 6,5 GiB freien VRAM und darf nicht fremd belegt sein.')
headroom=cgroup_headroom()
if headroom is not None and headroom<8*1024**3:raise ValueError('Mindestens 8 GiB freier Deck-Systemspeicher erforderlich.')
self.root.mkdir(parents=True,exist_ok=True,mode=0o700)
self.job.update(phase='VoxCPM2 wird geladen',gpu=gpu['name']);self.gpu_uuid=gpu['uuid']
threading.Thread(target=self._run_voice,args=(p,ident,dict(text=text,cfg=cfg,steps=steps,seed=seed),reference,gpu,release),daemon=True).start();return dict(self.job)
except Exception as exc:
if release:release()
with self.lock:self.job.update(state='failed',phase=str(exc),finished_at=time.time())
raise
def _run_voice(self,p,ident,request,reference,gpu,release):
root=self.root/ident;state='failed';phase='VoxCPM2-Ausführung fehlgeschlagen'
try:
root.mkdir(mode=0o700);self.stage(p,root)
if reference is not None:(root/'reference.wav').write_bytes(reference)
env={k:v for k,v in os.environ.items() if not k.startswith(('HF_','LLAMA_','DECK_'))};env.update(CUDA_VISIBLE_DEVICES=gpu['uuid'],HF_HUB_OFFLINE='1',TRANSFORMERS_OFFLINE='1',OMP_NUM_THREADS='4',HOME=str(root))
with self.lock:
if self.cancel.is_set():raise InterruptedError()
self.process=subprocess.Popen([str(self.runtime.paths()[0]),str(WORKER),str(root)],stdin=subprocess.PIPE,stdout=subprocess.DEVNULL,stderr=subprocess.DEVNULL,env=env,start_new_session=True,text=True)
process=self.process
process.stdin.write(json.dumps(request));process.stdin.close();process.stdin=None;deadline=time.monotonic()+600
while process.poll() is None:
if self.cancel.wait(.25):raise InterruptedError()
if time.monotonic()>deadline:raise ValueError('Zeitlimit für VoxCPM2 erreicht.')
if (root/'phase').exists():
with self.lock:self.job['phase']=(root/'phase').read_text()
if self.cancel.is_set():raise InterruptedError()
if not (root/'result.json').exists():raise ValueError('VoxCPM2-Worker beendet ohne Ergebnis. Speicher oder Laufzeit prüfen.')
result=json.loads((root/'result.json').read_text())
if not result['ok']:raise ValueError(result['error'])
output=root/'result.wav'
with wave.open(str(output)) as w:
if w.getnframes()==0 or output.stat().st_size>32*1024**2:raise ValueError('Ungültige Audioausgabe.')
state='complete';phase='Stimme fertig · Modell entladen'
with self.lock:self.job.update(duration=result['duration'],sample_rate=result['sample_rate'])
except InterruptedError:state='cancelled';phase='Stimmgenerierung abgebrochen · Modell entladen'
except Exception as exc:
if self.cancel.is_set():state='cancelled';phase='Stimmgenerierung abgebrochen · Modell entladen'
else:phase=str(exc) if isinstance(exc,ValueError) else 'VoxCPM2-Ausführung fehlgeschlagen ('+type(exc).__name__+').'
finally:
self.unload_idle();shutil.rmtree(root/'model',ignore_errors=True)
for name in ('reference.wav','phase'):(root/name).unlink(missing_ok=True)
if state=='complete':(root/'complete').touch()
with self.lock:self.job.update(state=state,phase=phase,finished_at=time.time())
release()