Files
AI-Profile-Router/dev/test_chunked.py
T

442 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Regressionstests für HTTP/1.1 chunked Transfer-Encoding im Router.
Testet:
1. Multipart + Content-Length (bestehender Pfad)
2. Multipart + Transfer-Encoding chunked
3. Mehrere unterschiedlich große Chunks
4. Boundary über Chunk-Grenzen verteilt
5. Chunk Extensions
6. Terminierender 0-Chunk
7. Malformed Chunk Size
8. Uploadgrößenlimit
9. Echter Open-WebUI-artiger WebM-Multipart-Request ohne Content-Length
Startet Mock-llama.cpp, Mock-TTS, Mock-STT und den Router,
führt dann HTTP-Requests gegen den Router aus.
"""
from __future__ import annotations
import json
import os
import socket
import subprocess
import sys
import time
import urllib.request
import urllib.error
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
ROUTER = os.path.join(BASE, "router", "ai_profile_router.py")
MOCK_LLAMA = os.path.join(BASE, "dev", "mock_upstream.py")
MOCK_TTS = os.path.join(BASE, "dev", "mock_tts_worker.py")
MOCK_STT = os.path.join(BASE, "dev", "mock_stt_worker.py")
FAKE_PROFILE = os.path.join(BASE, "dev", "fake-llama-profile.sh")
FAKE_PROFILE_DIR = os.path.join(BASE, "dev", "fake-profile-dir")
PORTS = {"router": 18091, "llama": 18090, "tts": 18089, "stt": 18088}
MAX_UPLOAD_SIZE = 1024 * 1024 # 1 MB für Tests
passed = 0
failed = 0
procs: list[subprocess.Popen] = []
def report(name: str, ok: bool, detail: str = "") -> None:
global passed, failed
if ok:
passed += 1
print(f" ✓ {name}")
else:
failed += 1
print(f" ✗ {name} {detail}")
def http_request(
method: str, port: int, path: str,
body: bytes | None = None, headers: dict | None = None,
) -> tuple[int, bytes]:
url = f"http://127.0.0.1:{port}{path}"
req = urllib.request.Request(url, data=body, method=method)
if headers:
for k, v in headers.items():
req.add_header(k, v)
try:
with urllib.request.urlopen(req, timeout=10) as resp:
return resp.status, resp.read()
except urllib.error.HTTPError as e:
return e.code, e.read()
def raw_http_request(
port: int, raw_request: bytes
) -> tuple[int, bytes]:
"""Sendet einen rohen HTTP-Request und liefert (status, body)."""
sock = socket.create_connection(("127.0.0.1", port), timeout=10)
sock.sendall(raw_request)
sock.shutdown(socket.SHUT_WR)
data = b""
while True:
chunk = sock.recv(65536)
if not chunk:
break
data += chunk
sock.close()
parts = data.split(b"\r\n\r\n", 1)
if len(parts) < 2:
return 0, b""
status_line = parts[0].split(b"\n")[0].decode()
status = int(status_line.split()[1])
resp_body = parts[1]
for line in parts[0].split(b"\n"):
if line.lower().startswith(b"content-length:"):
cl = int(line.split(b":")[1].strip())
resp_body = resp_body[:cl]
break
return status, resp_body
def build_chunked_body(chunks: list[bytes]) -> bytes:
body = b""
for chunk in chunks:
body += f"{len(chunk):x}\r\n".encode() + chunk + b"\r\n"
body += b"0\r\n\r\n"
return body
def build_chunked_body_with_ext(
chunks: list[tuple[bytes, str | None]]
) -> bytes:
body = b""
for chunk, ext in chunks:
if ext:
body += f"{len(chunk):x};{ext}\r\n".encode() + chunk + b"\r\n"
else:
body += f"{len(chunk):x}\r\n".encode() + chunk + b"\r\n"
body += b"0\r\n\r\n"
return body
def build_multipart(
fields: dict[str, str],
file_data: bytes | None = None,
filename: str = "test.webm",
boundary: str = "testboundary123",
) -> bytes:
body = b""
for name, value in fields.items():
body += (
f"--{boundary}\r\n"
f'Content-Disposition: form-data; name="{name}"\r\n'
f"\r\n"
f"{value}\r\n"
).encode()
if file_data is not None:
body += (
f"--{boundary}\r\n"
f'Content-Disposition: form-data; name="file"; '
f'filename="{filename}"\r\n'
f"Content-Type: audio/webm\r\n"
f"\r\n"
).encode() + file_data + b"\r\n"
body += f"--{boundary}--\r\n".encode()
return body
def start_process(cmd: list[str], env: dict) -> subprocess.Popen:
p = subprocess.Popen(
cmd, env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
procs.append(p)
return p
def wait_port(port: int, timeout: float = 10.0) -> bool:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
try:
with socket.create_connection(("127.0.0.1", port), timeout=1):
return True
except OSError:
time.sleep(0.1)
return False
def cleanup() -> None:
for p in procs:
try:
p.terminate()
p.wait(timeout=3)
except Exception:
try:
p.kill()
except Exception:
pass
def main() -> None:
global procs
try:
os.unlink("/tmp/test-chunked-router-state.json")
except FileNotFoundError:
pass
env = os.environ.copy()
env.update({
"ROUTER_HOST": "127.0.0.1",
"ROUTER_PORT": str(PORTS["router"]),
"ROUTER_AUTH_MODE": "off",
"ROUTER_PROFILES_FILE": "",
"ROUTER_STATE_FILE": "/tmp/test-chunked-router-state.json",
"UPSTREAM_URL": f"http://127.0.0.1:{PORTS['llama']}",
"PROFILE_SCRIPT": FAKE_PROFILE,
"PROFILE_DIR": FAKE_PROFILE_DIR,
"TTS_WORKER_URL": f"http://127.0.0.1:{PORTS['tts']}",
"STT_WORKER_URL": f"http://127.0.0.1:{PORTS['stt']}",
"MAX_UPLOAD_SIZE": str(MAX_UPLOAD_SIZE),
"LOG_LEVEL": "WARNING",
})
print("Starte Mock-Server und Router ...")
start_process([sys.executable, MOCK_LLAMA],
{**env, "MOCK_PORT": str(PORTS["llama"])})
start_process([sys.executable, MOCK_TTS],
{**env, "MOCK_TTS_PORT": str(PORTS["tts"])})
start_process([sys.executable, MOCK_STT],
{**env, "MOCK_STT_PORT": str(PORTS["stt"])})
start_process([sys.executable, ROUTER], env)
for name, port in PORTS.items():
if not wait_port(port, timeout=10):
print(f"FEHLER: {name} (Port {port}) nicht erreichbar")
cleanup()
sys.exit(1)
print("Alle Server erreichbar.\n")
# Fake WebM-Daten (Opus-Header + Dummy-Bytes)
fake_webm = (b"\x1a\x45\xdf\xa3" # EBML magic
b"\x00" * 100 + b"FAKE_WEBM_DATA" * 50)
# ------------------------------------------------------------------
# Test 1: Multipart + Content-Length (bestehender Pfad)
# ------------------------------------------------------------------
print("Test 1: Multipart + Content-Length")
mp = build_multipart({"model": "whisper-1", "language": "de"},
file_data=fake_webm)
status, body = http_request(
"POST", PORTS["router"], "/v1/audio/transcriptions",
body=mp,
headers={"Content-Type":
"multipart/form-data; boundary=testboundary123"})
data = json.loads(body) if body else {}
report("HTTP 200", status == 200, f"got {status}")
report("text vorhanden", "text" in data, str(data))
# ------------------------------------------------------------------
# Test 2: Multipart + Transfer-Encoding chunked (einfach)
# ------------------------------------------------------------------
print("Test 2: Multipart + chunked (einfach)")
mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm)
chunked = build_chunked_body([mp])
raw = (
b"POST /v1/audio/transcriptions HTTP/1.1\r\n"
b"Host: 127.0.0.1\r\n"
b"Content-Type: multipart/form-data; boundary=testboundary123\r\n"
b"Transfer-Encoding: chunked\r\n"
b"Connection: close\r\n"
b"\r\n"
) + chunked
status, body = raw_http_request(PORTS["router"], raw)
data = json.loads(body) if body else {}
report("HTTP 200", status == 200, f"got {status}")
report("text vorhanden", "text" in data, str(data))
# ------------------------------------------------------------------
# Test 3: Mehrere unterschiedlich große Chunks
# ------------------------------------------------------------------
print("Test 3: Mehrere unterschiedlich große Chunks")
mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm)
# In 5 Chunks aufteilen (unterschiedlich groß)
chunks = []
sizes = [10, 50, 7, 100, 33]
pos = 0
for s in sizes:
if pos < len(mp):
chunks.append(mp[pos:pos + s])
pos += s
if pos < len(mp):
chunks.append(mp[pos:])
chunked = build_chunked_body(chunks)
raw = (
b"POST /v1/audio/transcriptions HTTP/1.1\r\n"
b"Host: 127.0.0.1\r\n"
b"Content-Type: multipart/form-data; boundary=testboundary123\r\n"
b"Transfer-Encoding: chunked\r\n"
b"Connection: close\r\n"
b"\r\n"
) + chunked
status, body = raw_http_request(PORTS["router"], raw)
data = json.loads(body) if body else {}
report("HTTP 200", status == 200, f"got {status}")
report("text vorhanden", "text" in data, str(data))
# ------------------------------------------------------------------
# Test 4: Boundary über Chunk-Grenzen verteilt
# ------------------------------------------------------------------
print("Test 4: Boundary über Chunk-Grenzen verteilt")
mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm)
# Boundary-String finden und Chunk-Grenze genau dorthin setzen
boundary_str = b"--testboundary123"
idx = mp.find(boundary_str, 10) # zweite Boundary (vor file)
if idx == -1:
idx = len(mp) // 2
# Chunk 1 endet mitten in der Boundary
split_at = idx + len(boundary_str) // 2
chunks = [mp[:split_at], mp[split_at:]]
chunked = build_chunked_body(chunks)
raw = (
b"POST /v1/audio/transcriptions HTTP/1.1\r\n"
b"Host: 127.0.0.1\r\n"
b"Content-Type: multipart/form-data; boundary=testboundary123\r\n"
b"Transfer-Encoding: chunked\r\n"
b"Connection: close\r\n"
b"\r\n"
) + chunked
status, body = raw_http_request(PORTS["router"], raw)
data = json.loads(body) if body else {}
report("HTTP 200", status == 200, f"got {status}")
report("text vorhanden", "text" in data, str(data))
# ------------------------------------------------------------------
# Test 5: Chunk Extensions
# ------------------------------------------------------------------
print("Test 5: Chunk Extensions")
mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm)
chunks_ext = [
(mp[:20], "ext1=value1"),
(mp[20:60], None),
(mp[60:], "ext2=value2;ext3=value3"),
]
chunked = build_chunked_body_with_ext(chunks_ext)
raw = (
b"POST /v1/audio/transcriptions HTTP/1.1\r\n"
b"Host: 127.0.0.1\r\n"
b"Content-Type: multipart/form-data; boundary=testboundary123\r\n"
b"Transfer-Encoding: chunked\r\n"
b"Connection: close\r\n"
b"\r\n"
) + chunked
status, body = raw_http_request(PORTS["router"], raw)
data = json.loads(body) if body else {}
report("HTTP 200", status == 200, f"got {status}")
report("text vorhanden", "text" in data, str(data))
# ------------------------------------------------------------------
# Test 6: Terminierender 0-Chunk (bereits in allen Tests enthalten,
# hier explizit mit Trailer)
# ------------------------------------------------------------------
print("Test 6: 0-Chunk mit Trailer")
mp = build_multipart({"model": "whisper-1"}, file_data=fake_webm)
chunked = build_chunked_body([mp])
# Trailer hinzufügen
chunked_with_trailer = chunked.replace(
b"0\r\n\r\n", b"0\r\nTrailer-Test: value\r\n\r\n")
raw = (
b"POST /v1/audio/transcriptions HTTP/1.1\r\n"
b"Host: 127.0.0.1\r\n"
b"Content-Type: multipart/form-data; boundary=testboundary123\r\n"
b"Transfer-Encoding: chunked\r\n"
b"Connection: close\r\n"
b"\r\n"
) + chunked_with_trailer
status, body = raw_http_request(PORTS["router"], raw)
data = json.loads(body) if body else {}
report("HTTP 200", status == 200, f"got {status}")
report("text vorhanden", "text" in data, str(data))
# ------------------------------------------------------------------
# Test 7: Malformed Chunk Size
# ------------------------------------------------------------------
print("Test 7: Malformed Chunk Size")
raw = (
b"POST /v1/audio/transcriptions HTTP/1.1\r\n"
b"Host: 127.0.0.1\r\n"
b"Content-Type: multipart/form-data; boundary=testboundary123\r\n"
b"Transfer-Encoding: chunked\r\n"
b"Connection: close\r\n"
b"\r\n"
b"XYZ\r\n" # ungültige hexadezimale Größe
b"0\r\n\r\n"
)
status, body = raw_http_request(PORTS["router"], raw)
report("HTTP 400", status == 400, f"got {status}")
# ------------------------------------------------------------------
# Test 8: Uploadgrößenlimit
# ------------------------------------------------------------------
print("Test 8: Uploadgrößenlimit")
# MAX_UPLOAD_SIZE = 1 MB, also 2 MB senden
big_data = b"A" * (2 * 1024 * 1024)
mp = build_multipart({"model": "whisper-1"}, file_data=big_data)
chunked = build_chunked_body([mp[:1024 * 1024], mp[1024 * 1024:]])
raw = (
b"POST /v1/audio/transcriptions HTTP/1.1\r\n"
b"Host: 127.0.0.1\r\n"
b"Content-Type: multipart/form-data; boundary=testboundary123\r\n"
b"Transfer-Encoding: chunked\r\n"
b"Connection: close\r\n"
b"\r\n"
) + chunked
try:
status, body = raw_http_request(PORTS["router"], raw)
report("HTTP 400 (zu groß)", status == 400, f"got {status}")
except (BrokenPipeError, ConnectionResetError, OSError):
# Server schließt Verbindung bei zu großem Upload – OK
report("HTTP 400 (zu groß)", True, "Verbindung geschlossen")
# ------------------------------------------------------------------
# Test 9: Open-WebUI-artiger WebM-Multipart ohne Content-Length
# ------------------------------------------------------------------
print("Test 9: Open-WebUI-artiger WebM-Multipart (chunked)")
# Realistischer Open-WebUI-Request: WebM-Datei + model + language
webm_data = (b"\x1a\x45\xdf\xa3" + b"\x00" * 50
+ b"WEBM_OPUS_AUDIO_DATA" * 100)
boundary = "950bd961b24c4a32801e31b128c85e09"
mp = build_multipart(
{"model": "whisper-1", "language": "de"},
file_data=webm_data,
filename="recording.webm",
boundary=boundary,
)
# In mehrere Chunks aufteilen (wie Open WebUI es tut)
chunk_size = 4096
chunks = [mp[i:i + chunk_size]
for i in range(0, len(mp), chunk_size)]
chunked = build_chunked_body(chunks)
raw = (
b"POST /v1/audio/transcriptions HTTP/1.1\r\n"
b"Host: 127.0.0.1\r\n"
b"Content-Type: multipart/form-data; "
b"boundary=" + boundary.encode() + b"\r\n"
b"Transfer-Encoding: chunked\r\n"
b"Connection: close\r\n"
b"\r\n"
) + chunked
status, body = raw_http_request(PORTS["router"], raw)
data = json.loads(body) if body else {}
report("HTTP 200", status == 200, f"got {status}")
report("text vorhanden", "text" in data, str(data))
# ------------------------------------------------------------------
# Zusammenfassung
# ------------------------------------------------------------------
print(f"\n== Ergebnis: {passed} bestanden, {failed} fehlgeschlagen ==")
cleanup()
sys.exit(1 if failed else 0)
if __name__ == "__main__":
try:
main()
finally:
cleanup()