Add editable LLM penalties and preserve sampling through backup restore

This commit is contained in:
Mikei386 committed 2026-10-01 21:40:44 +02:00
1 parent 3779c846db
commit 37b5394df1
14 files changed
+218 -14

No files matched your search

+2 -1
View File
@@ -12,6 +12,7 @@ import threading
import time
from http.server import BaseHTTPRequestHandler,ThreadingHTTPServer
from api_compat import normalize_chat,CompatibilityError
from profiles import generation_parameters
from inference import InferenceError
from stt import read_upload
@@ -325,7 +326,7 @@ class APIHandler(BaseHTTPRequestHandler):
def allowed():return ep.allowed() and any(p['id']==profile['id'] and p['revision']==profile['revision'] and p['enabled'] for p in ep.rows())
with ep.scheduler.lease(key,profile['parameters']['slots'],lambda:ep.worker.ensure(profile),allowed=allowed):
body=dict(data)
for field in ('temperature','top_p','top_k'):body.setdefault(field,profile['parameters'][field])
for field,value in generation_parameters(profile['parameters']).items():body.setdefault(field,value)
conn,key=ep.worker.connect()
try:
conn.request('POST','/v1/chat/completions',body=json.dumps(body).encode(),headers={'Content-Type':'application/json','Authorization':'Bearer '+key})