135 lines
4.9 KiB
Python
Executable File
135 lines
4.9 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Mock-llama.cpp für lokale Tests des AI Profile Router.
|
|
|
|
Simuliert die relevanten Endpunkte von llama.cpp:
|
|
GET /health
|
|
GET /v1/models (n_ctx wird aus der Fake-override.conf gelesen)
|
|
POST /v1/chat/completions (non-streaming, streaming, Tool Calls)
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import time
|
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
|
|
PROFILE_DIR = os.environ.get("MOCK_PROFILE_DIR", "dev/fake-profile-dir")
|
|
PORT = int(os.environ.get("MOCK_PORT", "18080"))
|
|
|
|
|
|
def current_ctx() -> int:
|
|
"""Liest --ctx-size aus der Fake-override.conf (simuliert Profilwechsel)."""
|
|
try:
|
|
with open(os.path.join(PROFILE_DIR, "override.conf"), encoding="utf-8") as f:
|
|
for line in f:
|
|
if "--ctx-size" in line:
|
|
return int(line.split("--ctx-size")[1].split()[0])
|
|
except (OSError, ValueError, IndexError):
|
|
pass
|
|
return 76800
|
|
|
|
|
|
class Handler(BaseHTTPRequestHandler):
|
|
def do_GET(self):
|
|
if self.path == "/health":
|
|
self._json(200, {"status": "ok"})
|
|
elif self.path == "/v1/models":
|
|
ctx = current_ctx()
|
|
self._json(200, {
|
|
"object": "list",
|
|
"data": [{
|
|
"id": f"mock-model-{ctx}",
|
|
"object": "model",
|
|
"owned_by": "mock",
|
|
"meta": {"n_ctx": ctx},
|
|
}],
|
|
})
|
|
else:
|
|
self._json(404, {"error": {"message": "not found"}})
|
|
|
|
def do_POST(self):
|
|
length = int(self.headers.get("Content-Length") or 0)
|
|
body = json.loads(self.rfile.read(length) or b"{}")
|
|
if self.path == "/v1/chat/completions":
|
|
if body.get("stream"):
|
|
self._stream(body)
|
|
else:
|
|
self._json(200, self._completion(body))
|
|
else:
|
|
self._json(404, {"error": {"message": "not found"}})
|
|
|
|
def _completion(self, body: dict) -> dict:
|
|
delay = body.get("mock_delay", 0)
|
|
if isinstance(delay, (int, float)) and 0 < delay <= 10:
|
|
time.sleep(delay)
|
|
model = body.get("model")
|
|
if body.get("tools"):
|
|
name = body["tools"][0]["function"]["name"]
|
|
message = {
|
|
"role": "assistant",
|
|
"content": None,
|
|
"tool_calls": [{
|
|
"id": "call_mock_1",
|
|
"type": "function",
|
|
"function": {"name": name, "arguments": '{"city": "Berlin"}'},
|
|
}],
|
|
}
|
|
finish = "tool_calls"
|
|
else:
|
|
message = {"role": "assistant",
|
|
"content": f"Mock-Antwort (Modell: {model})"}
|
|
finish = "stop"
|
|
return {
|
|
"id": "chatcmpl-mock",
|
|
"object": "chat.completion",
|
|
"created": int(time.time()),
|
|
"model": model,
|
|
"choices": [{"index": 0, "message": message,
|
|
"finish_reason": finish}],
|
|
"usage": {"prompt_tokens": 10, "completion_tokens": 5,
|
|
"total_tokens": 15},
|
|
"mock_ctx": current_ctx(),
|
|
"mock_authorization": self.headers.get("Authorization"),
|
|
"mock_reasoning_effort": body.get("reasoning_effort"),
|
|
"mock_chat_template_kwargs": body.get("chat_template_kwargs"),
|
|
}
|
|
|
|
def _stream(self, body: dict) -> None:
|
|
self.send_response(200)
|
|
self.send_header("Content-Type", "text/event-stream")
|
|
self.send_header("Connection", "close")
|
|
self.end_headers()
|
|
model = body.get("model")
|
|
delay = body.get("mock_stream_delay", 0.05)
|
|
if not isinstance(delay, (int, float)) or delay < 0 or delay > 10:
|
|
delay = 0.05
|
|
for tok in ["Mock-", "Streaming", "-Antwort", f" ({model})"]:
|
|
chunk = {
|
|
"id": "chatcmpl-mock",
|
|
"object": "chat.completion.chunk",
|
|
"model": model,
|
|
"choices": [{"index": 0, "delta": {"content": tok},
|
|
"finish_reason": None}],
|
|
}
|
|
self.wfile.write(f"data: {json.dumps(chunk)}\n\n".encode())
|
|
self.wfile.flush()
|
|
time.sleep(delay)
|
|
try:
|
|
self.wfile.write(b"data: [DONE]\n\n")
|
|
self.wfile.flush()
|
|
except (BrokenPipeError, ConnectionResetError):
|
|
pass
|
|
|
|
def _json(self, code: int, payload: dict) -> None:
|
|
body = json.dumps(payload).encode()
|
|
self.send_response(code)
|
|
self.send_header("Content-Type", "application/json")
|
|
self.send_header("Content-Length", str(len(body)))
|
|
self.send_header("Connection", "close")
|
|
self.end_headers()
|
|
self.wfile.write(body)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
print(f"Mock-llama.cpp auf 127.0.0.1:{PORT} (Profile-Dir: {PROFILE_DIR})")
|
|
ThreadingHTTPServer(("127.0.0.1", PORT), Handler).serve_forever()
|