From 104c9904a52ee0c481394e957317423a743b37d0 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Wed, 26 Aug 2026 07:06:03 +0200 Subject: [PATCH] Stream llama responses without proxy buffering --- router/ai_profile_router.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index 0f37d80..ae3b849 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -2091,7 +2091,11 @@ class Handler(BaseHTTPRequestHandler): self.end_headers() try: while True: - chunk = resp.read(16384) + # ``read(n)`` may wait until the complete buffer is filled. + # That defeats SSE: a client sees no token for a long time and + # may time out while llama.cpp is already generating. read1() + # returns the next currently available wire chunk instead. + chunk = resp.read1(16384) if not chunk: break self.wfile.write(chunk)