Show native prefill and generation timing in chat tests

This commit is contained in:
Mikei386
2026-09-28 20:47:32 +02:00
parent 7ada7f6f20
commit ec62c859f7
4 changed files with 24 additions and 3 deletions
+8 -2
View File
@@ -1,5 +1,6 @@
"""Ephemeral admin chat tests using the same worker and leases as the API."""
import json
import math
import socket
import threading
import time
@@ -25,7 +26,7 @@ class ChatTests:
if not profile or not profile['runnable']:raise ValueError('Kein ausführbares Chatprofil. Modell und CUDA-Build prüfen.')
with self.lock:
if self.job and self.job['state']=='running':raise ValueError('Ein Chat-Test läuft bereits.')
self.cancel.clear();self.job=dict(id=uuid.uuid4().hex,state='running',phase='Wartet auf freie Modellreservierung',profile_id=profile['id'],profile_name=profile['name'],started_at=time.time(),answer='',reasoning='',error=None)
self.cancel.clear();self.job=dict(id=uuid.uuid4().hex,state='running',phase='Wartet auf freie Modellreservierung',profile_id=profile['id'],profile_name=profile['name'],started_at=time.time(),answer='',reasoning='',error=None,timings={},usage={})
threading.Thread(target=self._run,args=(profile,messages,limit),daemon=True).start()
return dict(self.job)
def phase(self,text):
@@ -51,7 +52,7 @@ class ChatTests:
with self.scheduler.lease(key,profile['parameters']['slots'],prepare,allowed=allowed):
if self.cancel.is_set():raise InterruptedError()
self.phase('Antwort wird erzeugt')
body=dict(model=profile['name'],messages=messages,max_tokens=limit,stream=True,**{k:profile['parameters'][k] for k in ('temperature','top_p','top_k')})
body=dict(model=profile['name'],messages=messages,max_tokens=limit,stream=True,stream_options={"include_usage":True},**{k:profile['parameters'][k] for k in ('temperature','top_p','top_k')})
conn,token=self.worker.connect()
conn.request('POST','/v1/chat/completions',json.dumps(body).encode(),headers={'Content-Type':'application/json','Authorization':'Bearer '+token})
with self.lock:self.socket=conn.sock
@@ -73,6 +74,11 @@ class ChatTests:
if payload==b'[DONE]':done=True;break
event=json.loads(payload)
if 'error' in event:raise InferenceError('Der Modellworker meldet einen Fehler während der Antwort.')
with self.lock:
for section,fields in [('timings',('prompt_n','prompt_ms','prompt_per_second','predicted_n','predicted_ms','predicted_per_second')),('usage',('prompt_tokens','completion_tokens','total_tokens'))]:
values=event.get(section)
if isinstance(values,dict):
self.job[section]={k:v for k,v in values.items() if k in fields and type(v) in (int,float) and math.isfinite(v) and v>=0}
for choice in event.get('choices',[]):
delta=choice.get('delta',{})
with self.lock: