Define four-profile production matrix with Medium default
This commit is contained in:
+25
-24
@@ -123,11 +123,12 @@ echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
ids={m["id"]:m for m in d["data"]}
|
||||
assert set(ids)=={"qwen-fast","qwen-medium","qwen-long"}, ids
|
||||
assert set(ids)=={"qwen-fast","qwen-medium","qwen-large","qwen-ultra"}, ids
|
||||
assert ids["qwen-fast"]["context_length"]==76800
|
||||
assert ids["qwen-medium"]["context_length"]==94208
|
||||
assert ids["qwen-long"]["context_length"]==131072
|
||||
' && ok "drei virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
|
||||
assert ids["qwen-medium"]["context_length"]==160000
|
||||
assert ids["qwen-large"]["context_length"]==192000
|
||||
assert ids["qwen-ultra"]["context_length"]==262144
|
||||
' && ok "vier virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
|
||||
|
||||
# --- 2. /status -----------------------------------------------------------------
|
||||
echo "== Test 2: /status"
|
||||
@@ -183,14 +184,14 @@ echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["profile"]=="medium" and d["context_length"]==94208, d
|
||||
assert d["profile"]=="medium" and d["context_length"]==160000, d
|
||||
' && ok "Profil medium aktiv" || bad "Profilwechsel medium"
|
||||
curl -sf "$BASE/status" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["current_profile"]=="medium", d
|
||||
assert d["upstream"]["ctx"]==94208, d
|
||||
' && ok "Status bestätigt medium (ctx 94208)" || bad "Status nach Wechsel"
|
||||
assert d["upstream"]["ctx"]==160000, d
|
||||
' && ok "Status bestätigt medium (ctx 160000)" || bad "Status nach Wechsel"
|
||||
|
||||
# --- 7. Profilwechsel medium -> fast --------------------------------------------------
|
||||
echo "== Test 7: Profilwechsel medium -> fast"
|
||||
@@ -203,21 +204,21 @@ assert d["profile"]=="fast" and d["context_length"]==76800, d
|
||||
' && ok "Profil fast wieder aktiv" || bad "Profilwechsel fast"
|
||||
|
||||
# --- 8. Virtuelles Modell triggert Profilwechsel ----------------------------------------
|
||||
echo "== Test 8: Chat mit qwen-long triggert Wechsel auf long"
|
||||
echo "== Test 8: Chat mit qwen-large triggert Wechsel auf large"
|
||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-long","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
-d '{"model":"qwen-large","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
echo "$RESP" | python3 -m json.tool
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d["model"]=="mock-model-131072", d
|
||||
' && ok "qwen-long hat Profil long aktiviert und weitergeleitet" || bad "virtuelles Modell"
|
||||
assert d["model"]=="mock-model-192000", d
|
||||
' && ok "qwen-large hat Profil large aktiviert und weitergeleitet" || bad "virtuelles Modell"
|
||||
|
||||
# --- 9. Methoden und ungültiges virtuelles Modell -----------------------------------------
|
||||
echo "== Test 9: sichere Profilmethoden + ungültiges virtuelles Modell"
|
||||
CODE=$(curl -s -o /tmp/err9.json -w "%{http_code}" "$BASE/long")
|
||||
CODE=$(curl -s -o /tmp/err9.json -w "%{http_code}" "$BASE/large")
|
||||
cat /tmp/err9.json; echo
|
||||
[ "$CODE" = "405" ] && ok "GET /long verändert kein Profil" || bad "erwartet 405, bekam $CODE"
|
||||
[ "$CODE" = "405" ] && ok "GET /large verändert kein Profil" || bad "erwartet 405, bekam $CODE"
|
||||
|
||||
CODE=$(curl -s -o /tmp/err9b.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
|
||||
-H "Content-Type: application/json" -d '{"model":"qwen-huge","messages":[]}')
|
||||
@@ -243,7 +244,7 @@ CODE=$(curl -s -o /tmp/chat9e.json -w "%{http_code}" "$BASE/v1/chat/completions"
|
||||
|
||||
# --- 10. llama.cpp down -> 502, danach Recovery ---------------------------------------------------
|
||||
echo "== Test 10: Upstream down -> 502, danach Recovery"
|
||||
# Profil auf fast setzen (aus Test 8 ist long aktiv)
|
||||
# Profil auf fast setzen (aus Test 8 ist large aktiv)
|
||||
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||
# Mock stoppen (simuliert Crash) – über Fake-systemctl
|
||||
FAKE_SYSTEMD_PIDFILE=/tmp/mock_upstream_pid FAKE_SYSTEMD_PORT="$UP_PORT" \
|
||||
@@ -423,14 +424,14 @@ sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "medium" ] && ok "Medium → Image → Medium" || bad "Profil nach Image: $PROFILE (erwartet medium)"
|
||||
|
||||
# --- 24. Long → Image → Long ------------------------------------------------------------------------
|
||||
echo "== Test 24: Long → Image → Long"
|
||||
curl -sf -X POST "$BASE/long" >/dev/null
|
||||
# --- 24. Large → Image → Large ----------------------------------------------------------------------
|
||||
echo "== Test 24: Large → Image → Large"
|
||||
curl -sf -X POST "$BASE/large" >/dev/null
|
||||
RESP=$(curl -sf "$BASE/v1/images/generations" -H "Content-Type: application/json" \
|
||||
-d '{"prompt":"long test","size":"1024x1024"}')
|
||||
-d '{"prompt":"large test","size":"1024x1024"}')
|
||||
sleep 0.5
|
||||
PROFILE=$(curl -sf "$BASE/status" | python3 -c 'import json,sys; print(json.load(sys.stdin)["current_profile"])')
|
||||
[ "$PROFILE" = "long" ] && ok "Long → Image → Long" || bad "Profil nach Image: $PROFILE (erwartet long)"
|
||||
[ "$PROFILE" = "large" ] && ok "Large → Image → Large" || bad "Profil nach Image: $PROFILE (erwartet large)"
|
||||
|
||||
# --- 25. /status während Image-Job -------------------------------------------------------------------
|
||||
echo "== Test 25: /status während Image-Job"
|
||||
@@ -702,8 +703,8 @@ wait $STT_PID43
|
||||
# --- 44. Zwei konkurrierende Profilanfragen ------------------------------------------------
|
||||
echo "== Test 44: Profil-Lease verhindert Wechsel während eines Chats"
|
||||
curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-long","mock_delay":1.0,"messages":[{"role":"user","content":"Lang"}]}' \
|
||||
>/tmp/chat44-long.json &
|
||||
-d '{"model":"qwen-large","mock_delay":1.0,"messages":[{"role":"user","content":"Groß"}]}' \
|
||||
>/tmp/chat44-large.json &
|
||||
CHAT44_PID=$!
|
||||
sleep 0.2
|
||||
curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
@@ -712,10 +713,10 @@ curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
wait "$CHAT44_PID"
|
||||
python3 -c '
|
||||
import json
|
||||
long=json.load(open("/tmp/chat44-long.json"))
|
||||
large=json.load(open("/tmp/chat44-large.json"))
|
||||
medium=json.load(open("/tmp/chat44-medium.json"))
|
||||
assert long["mock_ctx"] == 131072, long
|
||||
assert medium["mock_ctx"] == 94208, medium
|
||||
assert large["mock_ctx"] == 192000, large
|
||||
assert medium["mock_ctx"] == 160000, medium
|
||||
' && ok "konkurrierende Chats behielten jeweils ihr Profil" \
|
||||
|| bad "Profil-Lease bei konkurrierenden Chats"
|
||||
|
||||
|
||||
Reference in New Issue
Block a user