Expose profile context metadata for llama.cpp model discovery

This commit is contained in:
Mikei386 committed 2026-10-02 20:49:21 +02:00
1 parent a7b30b62d2
commit 1f8ffe2e72
3 files changed
+61 -2

No files matched your search

+7
View File
@@ -0,0 +1,7 @@
# Model discovery
The authenticated `/v1/models` chat catalog includes `context_window`, `parallel_slots`, and `context_scope: shared`. These values come from current enabled, runnable profiles rather than the model training limit. `/models` exposes the same catalog for native llama.cpp clients, including during exclusive GPU modes. In video mode `/v1/models` retains the original video API forwarding behavior.
`GET /props?model=<profile name>&autoload=false` exposes the configured context as `default_generation_settings.n_ctx` and `n_ctx`, slot count, vision-projector presence and supported known template capabilities. Queries never start, stop or load model workers. Unknown templates do not advertise tool or reasoning capabilities. Every discovery route requires the existing API bearer token.
OpenClaw's llama-cpp provider reads `/health`, `/models` and each model's `/props`. Explicit model rows override discovered rows; remove those only after verifying discovery and capability parity. Refresh then reads profile context changes from Deck. Shared context is not a guarantee that simultaneous slots can each use the full advertised budget.
+34 -1
View File
@@ -10,6 +10,7 @@ import re
import socket
import threading
import time
from urllib.parse import urlsplit,parse_qs
from http.server import BaseHTTPRequestHandler,ThreadingHTTPServer
from api_compat import normalize_chat,CompatibilityError
from profiles import generation_parameters
@@ -151,7 +152,29 @@ class Endpoint:
if not row['runnable']:raise APIError('Profil derzeit nicht ausführbar: '+' '.join(row['blockers']),503,'model_unavailable')
return row
def model_list(self,kind="chat"):
return dict(object='list',data=[dict(id='athena-image' if kind=='image' else p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck') for p in self.rows() if p['enabled'] and p['runnable'] and p['kind']==kind])
data=[]
for p in self.rows():
if not (p['enabled'] and p['runnable'] and p['kind']==kind):continue
row=dict(id='athena-image' if kind=='image' else p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck')
if kind=='chat':
params=p.get('parameters',{})
row.update(context_window=int(params.get('context',8192)),parallel_slots=int(params.get('slots',1)),context_scope='shared')
data.append(row)
return dict(object='list',data=data)
def model_properties(self,name):
# Discovery is configuration-only: never start or probe a model worker.
p=self.find_profile(name,'chat')
params=p.get('parameters',{})
model=p.get('model') or {}
identity=' '.join(str(model.get(k,'')) for k in ('repo','file','filename','name')).lower()
known_template=bool(re.search(r'qwen|gemma',identity))
return dict(model=p['name'],n_ctx=int(params.get('context',8192)),
default_generation_settings=dict(n_ctx=int(params.get('context',8192)),params={}),
total_slots=int(params.get('slots',1)),context_scope='shared',
modalities=dict(vision=bool(params.get('vision_projector'))),
chat_template_caps=dict(supports_tool_calls=known_template,
supports_typed_content=True,
supports_reasoning_effort='qwen' in identity))
class APIHTTPServer(ThreadingHTTPServer):
daemon_threads=True
@@ -206,6 +229,16 @@ class APIHandler(BaseHTTPRequestHandler):
with ep.lock:
if not ep.allowed():raise APIError('Endpunkt wird gestoppt.',503,'endpoint_stopping')
ep.inflight+=1;admitted=True
parsed=urlsplit(self.path)
if self.command=='GET' and (parsed.path=='/models' or (parsed.path=='/v1/models' and ep.scheduler.gpu_mode!='video')):
return self.send(ep.model_list())
if self.command=='GET' and parsed.path=='/props':
query=parse_qs(parsed.query)
name=query.get('model',[None])[0]
if not name:raise APIError('model ist für Router-Eigenschaften erforderlich.')
return self.send(ep.model_properties(name))
if self.command=='GET' and parsed.path=='/health':
return self.send({'status':'ok','service':'athena-deck-api'})
if ep.scheduler.gpu_mode=='separator' and self.path not in ('/v1/embeddings','/v1/embeddings/models'):raise APIError('Audio-Trennung aktiv. Für Chat, Bilder oder Musik die Betriebsart wechseln.',503,'separator_mode_active')
if ep.video and ep.scheduler.gpu_mode not in ('llm','music') and self.path not in ('/v1/embeddings','/v1/embeddings/models'):
if ep.scheduler.gpu_mode=='switching':raise APIError('Moduswechsel läuft.',503,'mode_switching')
+20 -1
View File
@@ -61,6 +61,25 @@ class EndpointTests(unittest.TestCase):
def request(self,path='/v1/models',data=None,token='A'*32):
conn=http.client.HTTPConnection('127.0.0.1',self.port,timeout=5);headers={'Content-Type':'application/json','Authorization':'Bearer '+token}
conn.request('POST' if data is not None else 'GET',path,body=json.dumps(data) if data is not None else None,headers=headers);r=conn.getresponse();body=r.read();conn.close();return r.status,(body if data and data.get('stream') else json.loads(body))
def test_discovery_reports_profile_context_without_loading(self):
self.rows[0]['parameters'].update(context=160000,slots=2)
self.rows[0]['model']={'repo':'Qwen/Qwen3.8-27B'}
self.enable('alpha')
self.ep.scheduler.gpu_mode='separator'
status,listing=self.request('/models')
self.assertEqual(status,200)
self.assertEqual(listing['data'][0]['context_window'],160000)
status,props=self.request('/props?model=alpha&autoload=false')
self.assertEqual(status,200)
self.assertEqual(props['default_generation_settings']['n_ctx'],160000)
self.assertEqual(props['total_slots'],2)
self.assertTrue(props['chat_template_caps']['supports_tool_calls'])
self.rows[0]['parameters']['context']=32768
self.assertEqual(self.request('/props?model=alpha&autoload=false')[1]['n_ctx'],32768)
self.assertEqual(self.request('/props?model=beta&autoload=false')[0],404)
self.assertEqual(self.request('/props?model=alpha',token='bad')[0],401)
self.assertFalse(self.worker.started)
def test_status_allows_video_callback_to_read_endpoint_from_other_thread(self):
finished=threading.Event();readers=[]
def video_status():
@@ -95,7 +114,7 @@ class EndpointTests(unittest.TestCase):
status,error=self.request('/v1/chat/completions',dict(model='alpha',messages=[dict(role='user',content='Synthetic')]))
self.assertEqual(status,503);self.assertEqual(error['error']['code'],'separator_mode_active')
self.assertEqual(self.worker.started,[]);self.ep.video.relay.assert_not_called()
self.assertEqual(self.request('/v1/models')[0],503)
self.assertEqual(self.request('/v1/models')[0],200)
def enable(self,name):self.ep.enable(dict(id=name,enabled=True))
def test_explicit_publication_auth_rotation_and_missing_audio(self):
self.assertEqual(self.request()[1]['data'],[]);self.assertEqual(self.request(token='bad')[0],401)