diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index 0f37d80..ae3b849 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -2091,7 +2091,11 @@ class Handler(BaseHTTPRequestHandler): self.end_headers() try: while True: - chunk = resp.read(16384) + # ``read(n)`` may wait until the complete buffer is filled. + # That defeats SSE: a client sees no token for a long time and + # may time out while llama.cpp is already generating. read1() + # returns the next currently available wire chunk instead. + chunk = resp.read1(16384) if not chunk: break self.wfile.write(chunk)