Expose profile context metadata for llama.cpp model discovery
This commit is contained in:
1 parent
a7b30b62d2
commit
1f8ffe2e72
3 files changed
+61
-2
No files matched your search
@@ -0,0 +1,7 @@
|
||||
# Model discovery
|
||||
|
||||
The authenticated `/v1/models` chat catalog includes `context_window`, `parallel_slots`, and `context_scope: shared`. These values come from current enabled, runnable profiles rather than the model training limit. `/models` exposes the same catalog for native llama.cpp clients, including during exclusive GPU modes. In video mode `/v1/models` retains the original video API forwarding behavior.
|
||||
|
||||
`GET /props?model=<profile name>&autoload=false` exposes the configured context as `default_generation_settings.n_ctx` and `n_ctx`, slot count, vision-projector presence and supported known template capabilities. Queries never start, stop or load model workers. Unknown templates do not advertise tool or reasoning capabilities. Every discovery route requires the existing API bearer token.
|
||||
|
||||
OpenClaw's llama-cpp provider reads `/health`, `/models` and each model's `/props`. Explicit model rows override discovered rows; remove those only after verifying discovery and capability parity. Refresh then reads profile context changes from Deck. Shared context is not a guarantee that simultaneous slots can each use the full advertised budget.
|
||||
+34
-1
@@ -10,6 +10,7 @@ import re
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
from urllib.parse import urlsplit,parse_qs
|
||||
from http.server import BaseHTTPRequestHandler,ThreadingHTTPServer
|
||||
from api_compat import normalize_chat,CompatibilityError
|
||||
from profiles import generation_parameters
|
||||
@@ -151,7 +152,29 @@ class Endpoint:
|
||||
if not row['runnable']:raise APIError('Profil derzeit nicht ausführbar: '+' '.join(row['blockers']),503,'model_unavailable')
|
||||
return row
|
||||
def model_list(self,kind="chat"):
|
||||
return dict(object='list',data=[dict(id='athena-image' if kind=='image' else p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck') for p in self.rows() if p['enabled'] and p['runnable'] and p['kind']==kind])
|
||||
data=[]
|
||||
for p in self.rows():
|
||||
if not (p['enabled'] and p['runnable'] and p['kind']==kind):continue
|
||||
row=dict(id='athena-image' if kind=='image' else p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck')
|
||||
if kind=='chat':
|
||||
params=p.get('parameters',{})
|
||||
row.update(context_window=int(params.get('context',8192)),parallel_slots=int(params.get('slots',1)),context_scope='shared')
|
||||
data.append(row)
|
||||
return dict(object='list',data=data)
|
||||
def model_properties(self,name):
|
||||
# Discovery is configuration-only: never start or probe a model worker.
|
||||
p=self.find_profile(name,'chat')
|
||||
params=p.get('parameters',{})
|
||||
model=p.get('model') or {}
|
||||
identity=' '.join(str(model.get(k,'')) for k in ('repo','file','filename','name')).lower()
|
||||
known_template=bool(re.search(r'qwen|gemma',identity))
|
||||
return dict(model=p['name'],n_ctx=int(params.get('context',8192)),
|
||||
default_generation_settings=dict(n_ctx=int(params.get('context',8192)),params={}),
|
||||
total_slots=int(params.get('slots',1)),context_scope='shared',
|
||||
modalities=dict(vision=bool(params.get('vision_projector'))),
|
||||
chat_template_caps=dict(supports_tool_calls=known_template,
|
||||
supports_typed_content=True,
|
||||
supports_reasoning_effort='qwen' in identity))
|
||||
|
||||
class APIHTTPServer(ThreadingHTTPServer):
|
||||
daemon_threads=True
|
||||
@@ -206,6 +229,16 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
with ep.lock:
|
||||
if not ep.allowed():raise APIError('Endpunkt wird gestoppt.',503,'endpoint_stopping')
|
||||
ep.inflight+=1;admitted=True
|
||||
parsed=urlsplit(self.path)
|
||||
if self.command=='GET' and (parsed.path=='/models' or (parsed.path=='/v1/models' and ep.scheduler.gpu_mode!='video')):
|
||||
return self.send(ep.model_list())
|
||||
if self.command=='GET' and parsed.path=='/props':
|
||||
query=parse_qs(parsed.query)
|
||||
name=query.get('model',[None])[0]
|
||||
if not name:raise APIError('model ist für Router-Eigenschaften erforderlich.')
|
||||
return self.send(ep.model_properties(name))
|
||||
if self.command=='GET' and parsed.path=='/health':
|
||||
return self.send({'status':'ok','service':'athena-deck-api'})
|
||||
if ep.scheduler.gpu_mode=='separator' and self.path not in ('/v1/embeddings','/v1/embeddings/models'):raise APIError('Audio-Trennung aktiv. Für Chat, Bilder oder Musik die Betriebsart wechseln.',503,'separator_mode_active')
|
||||
if ep.video and ep.scheduler.gpu_mode not in ('llm','music') and self.path not in ('/v1/embeddings','/v1/embeddings/models'):
|
||||
if ep.scheduler.gpu_mode=='switching':raise APIError('Moduswechsel läuft.',503,'mode_switching')
|
||||
|
||||
+20
-1
@@ -61,6 +61,25 @@ class EndpointTests(unittest.TestCase):
|
||||
def request(self,path='/v1/models',data=None,token='A'*32):
|
||||
conn=http.client.HTTPConnection('127.0.0.1',self.port,timeout=5);headers={'Content-Type':'application/json','Authorization':'Bearer '+token}
|
||||
conn.request('POST' if data is not None else 'GET',path,body=json.dumps(data) if data is not None else None,headers=headers);r=conn.getresponse();body=r.read();conn.close();return r.status,(body if data and data.get('stream') else json.loads(body))
|
||||
def test_discovery_reports_profile_context_without_loading(self):
|
||||
self.rows[0]['parameters'].update(context=160000,slots=2)
|
||||
self.rows[0]['model']={'repo':'Qwen/Qwen3.8-27B'}
|
||||
self.enable('alpha')
|
||||
self.ep.scheduler.gpu_mode='separator'
|
||||
status,listing=self.request('/models')
|
||||
self.assertEqual(status,200)
|
||||
self.assertEqual(listing['data'][0]['context_window'],160000)
|
||||
status,props=self.request('/props?model=alpha&autoload=false')
|
||||
self.assertEqual(status,200)
|
||||
self.assertEqual(props['default_generation_settings']['n_ctx'],160000)
|
||||
self.assertEqual(props['total_slots'],2)
|
||||
self.assertTrue(props['chat_template_caps']['supports_tool_calls'])
|
||||
self.rows[0]['parameters']['context']=32768
|
||||
self.assertEqual(self.request('/props?model=alpha&autoload=false')[1]['n_ctx'],32768)
|
||||
self.assertEqual(self.request('/props?model=beta&autoload=false')[0],404)
|
||||
self.assertEqual(self.request('/props?model=alpha',token='bad')[0],401)
|
||||
self.assertFalse(self.worker.started)
|
||||
|
||||
def test_status_allows_video_callback_to_read_endpoint_from_other_thread(self):
|
||||
finished=threading.Event();readers=[]
|
||||
def video_status():
|
||||
@@ -95,7 +114,7 @@ class EndpointTests(unittest.TestCase):
|
||||
status,error=self.request('/v1/chat/completions',dict(model='alpha',messages=[dict(role='user',content='Synthetic')]))
|
||||
self.assertEqual(status,503);self.assertEqual(error['error']['code'],'separator_mode_active')
|
||||
self.assertEqual(self.worker.started,[]);self.ep.video.relay.assert_not_called()
|
||||
self.assertEqual(self.request('/v1/models')[0],503)
|
||||
self.assertEqual(self.request('/v1/models')[0],200)
|
||||
def enable(self,name):self.ep.enable(dict(id=name,enabled=True))
|
||||
def test_explicit_publication_auth_rotation_and_missing_audio(self):
|
||||
self.assertEqual(self.request()[1]['data'],[]);self.assertEqual(self.request(token='bad')[0],401)
|
||||
|
||||
Reference in new issue
Block a user