Reduce controller polling and bound chat admission waits

This commit is contained in:
Mikei386
2026-09-20 19:10:59 +02:00
parent 53320fa6c3
commit 6071b1a2cd
10 changed files with 303 additions and 134 deletions
Executable
+12
View File
@@ -0,0 +1,12 @@
#!/usr/bin/env bash
# Fast, GPU-free release checks. Integration mocks are opt-in.
set -Eeuo pipefail
cd "$(dirname "${BASH_SOURCE[0]}")/.."
python3 platform/scripts/sync-profile-matrix.py --check
python3 -m unittest discover -s dev -p 'test_*.py'
if [[ ${1:-} == --integration ]]; then
bash dev/test_local.sh
elif [[ $# -gt 0 ]]; then
echo 'Usage: dev/check.sh [--integration]' >&2
exit 2
fi
+32 -1
View File
@@ -1,4 +1,5 @@
import importlib.util
import json
import os
import pathlib
import unittest
@@ -42,6 +43,34 @@ def music_item(state="exited"):
class ProfileControllerTests(unittest.TestCase):
def test_status_keeps_llm_when_optional_container_is_missing(self):
payload = json.dumps([item("medium", "running")]).encode()
with patch.object(controller, "MUSIC_WORKER", "missing-music"), \
patch.object(controller, "docker_request", return_value=(200, payload)) as request:
result = controller.status_snapshot()
self.assertEqual(result["active_profile"], "medium")
self.assertEqual(result["music_worker"], "missing")
self.assertIn("music", result["worker_errors"])
request.assert_called_once_with("GET", "/containers/json?all=1")
def test_empty_snapshot_does_not_repeat_docker_query(self):
with patch.object(controller, "docker_request", return_value=(200, b"[]")) as request:
result = controller.status_snapshot()
self.assertIsNone(result["active_profile"])
self.assertEqual(request.call_count, 1)
def test_video_ui_failure_does_not_hide_llm(self):
video = {"Id": "video", "State": "running",
"Labels": {controller.VIDEO_LABEL_KEY: "ltx2"}}
payload = json.dumps([item("medium", "running"), video]).encode()
with patch.object(controller, "VIDEO_WORKER", "ltx2"), \
patch.object(controller, "docker_request", return_value=(200, payload)), \
patch.object(controller, "cached_video_ui_state", side_effect=RuntimeError("exec failed")):
result = controller.status_snapshot()
self.assertEqual(result["active_profile"], "medium")
self.assertEqual(result["video_worker"], "running")
self.assertIn("video_ui", result["worker_errors"])
def test_music_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["ultra"] = item("ultra", "running")
@@ -99,9 +128,11 @@ class ProfileControllerTests(unittest.TestCase):
profiles = {name: item(name) for name in controller.ALLOWED[:-1]}
with patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "image_containers", return_value=[image_item()]), \
patch.object(controller, "tts_container", return_value=tts_item()):
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request") as request:
with self.assertRaisesRegex(RuntimeError, "missing"):
controller.activate("fast")
request.assert_not_called()
def test_image_start_stops_inference_first(self):
profiles = {name: item(name) for name in controller.ALLOWED}
+8 -1
View File
@@ -1,4 +1,5 @@
import io
import re
import subprocess
import tarfile
import tempfile
@@ -94,7 +95,13 @@ class RecoveryScriptTests(unittest.TestCase):
gateway = (ROOT / "platform/docker/wireguard-gateway/entrypoint.sh").read_text(
encoding="utf-8"
)
self.assertNotIn('network_mode: "service:wireguard-gateway"', compose)
# WebRTC deliberately shares the gateway for private ICE candidates.
# Dashboard and Portainer must retain their independent namespaces.
for service in ("llama-dashboard", "portainer"):
block = re.search(
rf"(?ms)^ {service}:\n(.*?)(?=^ [a-zA-Z0-9_-]+:|\Z)", compose)
self.assertIsNotNone(block)
self.assertNotIn('network_mode: "service:wireguard-gateway"', block.group(1))
self.assertNotIn('"8099:8099"', compose)
self.assertNotIn('"9443:9443"', compose)
self.assertIn('start_proxy 8099 llama-dashboard:8099', gateway)
+102
View File
@@ -0,0 +1,102 @@
"""Regression coverage for bounded admission and inexpensive status reads."""
import json
import os
import sys
import threading
import unittest
from pathlib import Path
from unittest.mock import patch
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'router'))
os.environ.setdefault('ROUTER_PROFILES_FILE', '')
import ai_profile_router as router
class CoordinationTests(unittest.TestCase):
def test_mode_uses_one_consistent_controller_snapshot(self):
with patch.object(router, 'PROFILE_CONTROL_URL', 'http://controller'), \
patch.object(router, '_profile_controller_request', return_value={
'music_worker': 'running', 'music_health': 'healthy'}) as request, \
patch.object(router.RUNTIME, 'load', return_value={}):
result = router.Handler._mode_payload()
self.assertEqual(result['music_worker'], 'running')
self.assertEqual(result['music_health'], 'healthy')
request.assert_called_once_with('GET', '/status', timeout=3)
def test_active_profile_does_not_request_optional_workers(self):
with patch.object(router, 'PROFILE_CONTROL_URL', 'http://controller'), \
patch.object(router, '_profile_controller_request',
return_value={'active_profile': 'medium'}) as request:
self.assertEqual(router.current_profile(), 'medium')
request.assert_called_once_with('GET', '/profiles/status', timeout=3)
def test_waiting_chat_times_out_without_upstream_work(self):
handler = object.__new__(router.Handler)
errors = []
handler._send_error = lambda *args: errors.append(args)
router.STATE.lock.acquire()
thread = None
try:
with patch.object(router, 'CHAT_WAIT_TIMEOUT', 0.02), \
patch.object(router, 'upstream_status') as upstream:
thread = threading.Thread(target=handler._chat_proxy,
args=(None, {'messages': []}, None))
thread.start()
thread.join(0.5)
self.assertFalse(thread.is_alive(), 'lock wait ignored timeout')
upstream.assert_not_called()
self.assertEqual(errors[0][0], 503)
self.assertEqual(errors[0][3], 'model_wait_timeout')
finally:
router.STATE.lock.release()
if thread:
thread.join(1)
def test_ultra_catalog_is_text_only(self):
with patch.object(router, 'current_profile', return_value='medium'):
catalog = router.Handler._llamacpp_models_payload()['data']
ultra = next(x for x in catalog if x['id'] == 'qwen-ultra')
self.assertEqual(ultra['architecture']['input_modalities'], ['text'])
def test_ultra_image_rejected_before_profile_switch(self):
handler = object.__new__(router.Handler)
errors = []
handler._send_error = lambda *args: errors.append(args)
data = {'messages': [{'role': 'user', 'content': [{'type': 'image_url',
'image_url': {'url': 'data:image/png;base64,iVBORw0KGgo='}}]}]}
with patch.object(router, 'switch_profile') as switch:
handler._chat_proxy(None, data, 'ultra')
switch.assert_not_called()
self.assertEqual(errors[0][0], 400)
def test_ready_chat_reuses_profile_readiness_result_and_releases_lease(self):
handler = object.__new__(router.Handler)
before = router.STATE.active_chats
available = router.STATE.qwen_unavailable
router.STATE.qwen_unavailable = False
handler._proxy = lambda body: self.assertEqual(
json.loads(body)['model'], 'qwen-medium')
try:
with patch.object(router, 'switch_profile', return_value={
'reachable': True, 'model': 'qwen-medium', 'ctx': 160000}), \
patch.object(router, 'upstream_status') as upstream:
handler._chat_proxy(None, {'messages': []}, 'medium')
upstream.assert_not_called()
self.assertEqual(router.STATE.active_chats, before)
finally:
router.STATE.qwen_unavailable = available
def test_startup_accepts_tested_mtp_context_overhead_without_restart(self):
with patch.object(router.RUNTIME, 'load', return_value={}), \
patch.object(router.RUNTIME, 'save'), \
patch.object(router, 'enforce_artifact_retention', return_value=[]), \
patch.object(router, 'current_profile', return_value='medium'), \
patch.object(router, 'upstream_status', return_value={
'reachable': True, 'model': 'qwen-medium', 'ctx': 160128}), \
patch.object(router, '_restore_qwen') as restore:
router._startup_reconcile()
restore.assert_not_called()
if __name__ == '__main__':
unittest.main()