Add editable LLM penalties and preserve sampling through backup restore

This commit is contained in:
Mikei386 committed 2026-10-01 21:40:44 +02:00
1 parent 3779c846db
commit 37b5394df1
14 files changed
+218 -14

No files matched your search

+10 -2
View File
@@ -3,7 +3,9 @@
Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
positive levels to the model template; it handles 'none' as thinking disabled.
"""
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'})
import math
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','repeat_penalty','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'})
LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'})
# The installed llama.cpp server rejects minimal and max. Use nearest supported hints.
EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'}
@@ -14,7 +16,13 @@ def normalize_chat(data):
"""Return a new request and non-sensitive translation metadata; never mutate input."""
unknown=set(data)-CHAT_FIELDS
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
body=dict(data);requested=body.get('reasoning_effort')
body=dict(data)
for key,lo,hi in [('repeat_penalty',0,2),('presence_penalty',-2,2),('frequency_penalty',-2,2)]:
if key in body:
value=body[key]
if isinstance(value,bool) or not isinstance(value,(int,float)) or not math.isfinite(value) or not lo<=value<=hi:
raise CompatibilityError('Ungültiger '+key+'-Wert.')
requested=body.get('reasoning_effort')
if requested is None:
body.pop('reasoning_effort',None)
return body,{}