Reduce controller polling and bound chat admission waits
This commit is contained in:
Executable
+12
@@ -0,0 +1,12 @@
|
||||
#!/usr/bin/env bash
|
||||
# Fast, GPU-free release checks. Integration mocks are opt-in.
|
||||
set -Eeuo pipefail
|
||||
cd "$(dirname "${BASH_SOURCE[0]}")/.."
|
||||
python3 platform/scripts/sync-profile-matrix.py --check
|
||||
python3 -m unittest discover -s dev -p 'test_*.py'
|
||||
if [[ ${1:-} == --integration ]]; then
|
||||
bash dev/test_local.sh
|
||||
elif [[ $# -gt 0 ]]; then
|
||||
echo 'Usage: dev/check.sh [--integration]' >&2
|
||||
exit 2
|
||||
fi
|
||||
@@ -1,4 +1,5 @@
|
||||
import importlib.util
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import unittest
|
||||
@@ -42,6 +43,34 @@ def music_item(state="exited"):
|
||||
|
||||
|
||||
class ProfileControllerTests(unittest.TestCase):
|
||||
def test_status_keeps_llm_when_optional_container_is_missing(self):
|
||||
payload = json.dumps([item("medium", "running")]).encode()
|
||||
with patch.object(controller, "MUSIC_WORKER", "missing-music"), \
|
||||
patch.object(controller, "docker_request", return_value=(200, payload)) as request:
|
||||
result = controller.status_snapshot()
|
||||
self.assertEqual(result["active_profile"], "medium")
|
||||
self.assertEqual(result["music_worker"], "missing")
|
||||
self.assertIn("music", result["worker_errors"])
|
||||
request.assert_called_once_with("GET", "/containers/json?all=1")
|
||||
|
||||
def test_empty_snapshot_does_not_repeat_docker_query(self):
|
||||
with patch.object(controller, "docker_request", return_value=(200, b"[]")) as request:
|
||||
result = controller.status_snapshot()
|
||||
self.assertIsNone(result["active_profile"])
|
||||
self.assertEqual(request.call_count, 1)
|
||||
|
||||
def test_video_ui_failure_does_not_hide_llm(self):
|
||||
video = {"Id": "video", "State": "running",
|
||||
"Labels": {controller.VIDEO_LABEL_KEY: "ltx2"}}
|
||||
payload = json.dumps([item("medium", "running"), video]).encode()
|
||||
with patch.object(controller, "VIDEO_WORKER", "ltx2"), \
|
||||
patch.object(controller, "docker_request", return_value=(200, payload)), \
|
||||
patch.object(controller, "cached_video_ui_state", side_effect=RuntimeError("exec failed")):
|
||||
result = controller.status_snapshot()
|
||||
self.assertEqual(result["active_profile"], "medium")
|
||||
self.assertEqual(result["video_worker"], "running")
|
||||
self.assertIn("video_ui", result["worker_errors"])
|
||||
|
||||
def test_music_start_exclusively_stops_gpu_workers(self):
|
||||
profiles = {name: item(name) for name in controller.ALLOWED}
|
||||
profiles["ultra"] = item("ultra", "running")
|
||||
@@ -99,9 +128,11 @@ class ProfileControllerTests(unittest.TestCase):
|
||||
profiles = {name: item(name) for name in controller.ALLOWED[:-1]}
|
||||
with patch.object(controller, "containers", return_value=profiles), \
|
||||
patch.object(controller, "image_containers", return_value=[image_item()]), \
|
||||
patch.object(controller, "tts_container", return_value=tts_item()):
|
||||
patch.object(controller, "tts_container", return_value=tts_item()), \
|
||||
patch.object(controller, "docker_request") as request:
|
||||
with self.assertRaisesRegex(RuntimeError, "missing"):
|
||||
controller.activate("fast")
|
||||
request.assert_not_called()
|
||||
|
||||
def test_image_start_stops_inference_first(self):
|
||||
profiles = {name: item(name) for name in controller.ALLOWED}
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import io
|
||||
import re
|
||||
import subprocess
|
||||
import tarfile
|
||||
import tempfile
|
||||
@@ -94,7 +95,13 @@ class RecoveryScriptTests(unittest.TestCase):
|
||||
gateway = (ROOT / "platform/docker/wireguard-gateway/entrypoint.sh").read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
self.assertNotIn('network_mode: "service:wireguard-gateway"', compose)
|
||||
# WebRTC deliberately shares the gateway for private ICE candidates.
|
||||
# Dashboard and Portainer must retain their independent namespaces.
|
||||
for service in ("llama-dashboard", "portainer"):
|
||||
block = re.search(
|
||||
rf"(?ms)^ {service}:\n(.*?)(?=^ [a-zA-Z0-9_-]+:|\Z)", compose)
|
||||
self.assertIsNotNone(block)
|
||||
self.assertNotIn('network_mode: "service:wireguard-gateway"', block.group(1))
|
||||
self.assertNotIn('"8099:8099"', compose)
|
||||
self.assertNotIn('"9443:9443"', compose)
|
||||
self.assertIn('start_proxy 8099 llama-dashboard:8099', gateway)
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
"""Regression coverage for bounded admission and inexpensive status reads."""
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'router'))
|
||||
os.environ.setdefault('ROUTER_PROFILES_FILE', '')
|
||||
import ai_profile_router as router
|
||||
|
||||
|
||||
class CoordinationTests(unittest.TestCase):
|
||||
def test_mode_uses_one_consistent_controller_snapshot(self):
|
||||
with patch.object(router, 'PROFILE_CONTROL_URL', 'http://controller'), \
|
||||
patch.object(router, '_profile_controller_request', return_value={
|
||||
'music_worker': 'running', 'music_health': 'healthy'}) as request, \
|
||||
patch.object(router.RUNTIME, 'load', return_value={}):
|
||||
result = router.Handler._mode_payload()
|
||||
self.assertEqual(result['music_worker'], 'running')
|
||||
self.assertEqual(result['music_health'], 'healthy')
|
||||
request.assert_called_once_with('GET', '/status', timeout=3)
|
||||
|
||||
def test_active_profile_does_not_request_optional_workers(self):
|
||||
with patch.object(router, 'PROFILE_CONTROL_URL', 'http://controller'), \
|
||||
patch.object(router, '_profile_controller_request',
|
||||
return_value={'active_profile': 'medium'}) as request:
|
||||
self.assertEqual(router.current_profile(), 'medium')
|
||||
request.assert_called_once_with('GET', '/profiles/status', timeout=3)
|
||||
|
||||
def test_waiting_chat_times_out_without_upstream_work(self):
|
||||
handler = object.__new__(router.Handler)
|
||||
errors = []
|
||||
handler._send_error = lambda *args: errors.append(args)
|
||||
router.STATE.lock.acquire()
|
||||
thread = None
|
||||
try:
|
||||
with patch.object(router, 'CHAT_WAIT_TIMEOUT', 0.02), \
|
||||
patch.object(router, 'upstream_status') as upstream:
|
||||
thread = threading.Thread(target=handler._chat_proxy,
|
||||
args=(None, {'messages': []}, None))
|
||||
thread.start()
|
||||
thread.join(0.5)
|
||||
self.assertFalse(thread.is_alive(), 'lock wait ignored timeout')
|
||||
upstream.assert_not_called()
|
||||
self.assertEqual(errors[0][0], 503)
|
||||
self.assertEqual(errors[0][3], 'model_wait_timeout')
|
||||
finally:
|
||||
router.STATE.lock.release()
|
||||
if thread:
|
||||
thread.join(1)
|
||||
|
||||
def test_ultra_catalog_is_text_only(self):
|
||||
with patch.object(router, 'current_profile', return_value='medium'):
|
||||
catalog = router.Handler._llamacpp_models_payload()['data']
|
||||
ultra = next(x for x in catalog if x['id'] == 'qwen-ultra')
|
||||
self.assertEqual(ultra['architecture']['input_modalities'], ['text'])
|
||||
|
||||
def test_ultra_image_rejected_before_profile_switch(self):
|
||||
handler = object.__new__(router.Handler)
|
||||
errors = []
|
||||
handler._send_error = lambda *args: errors.append(args)
|
||||
data = {'messages': [{'role': 'user', 'content': [{'type': 'image_url',
|
||||
'image_url': {'url': 'data:image/png;base64,iVBORw0KGgo='}}]}]}
|
||||
with patch.object(router, 'switch_profile') as switch:
|
||||
handler._chat_proxy(None, data, 'ultra')
|
||||
switch.assert_not_called()
|
||||
self.assertEqual(errors[0][0], 400)
|
||||
|
||||
def test_ready_chat_reuses_profile_readiness_result_and_releases_lease(self):
|
||||
handler = object.__new__(router.Handler)
|
||||
before = router.STATE.active_chats
|
||||
available = router.STATE.qwen_unavailable
|
||||
router.STATE.qwen_unavailable = False
|
||||
handler._proxy = lambda body: self.assertEqual(
|
||||
json.loads(body)['model'], 'qwen-medium')
|
||||
try:
|
||||
with patch.object(router, 'switch_profile', return_value={
|
||||
'reachable': True, 'model': 'qwen-medium', 'ctx': 160000}), \
|
||||
patch.object(router, 'upstream_status') as upstream:
|
||||
handler._chat_proxy(None, {'messages': []}, 'medium')
|
||||
upstream.assert_not_called()
|
||||
self.assertEqual(router.STATE.active_chats, before)
|
||||
finally:
|
||||
router.STATE.qwen_unavailable = available
|
||||
|
||||
def test_startup_accepts_tested_mtp_context_overhead_without_restart(self):
|
||||
with patch.object(router.RUNTIME, 'load', return_value={}), \
|
||||
patch.object(router.RUNTIME, 'save'), \
|
||||
patch.object(router, 'enforce_artifact_retention', return_value=[]), \
|
||||
patch.object(router, 'current_profile', return_value='medium'), \
|
||||
patch.object(router, 'upstream_status', return_value={
|
||||
'reachable': True, 'model': 'qwen-medium', 'ctx': 160128}), \
|
||||
patch.object(router, '_restore_qwen') as restore:
|
||||
router._startup_reconcile()
|
||||
restore.assert_not_called()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user