Stream llama responses without proxy buffering

This commit is contained in:
Mikei386
2026-08-26 07:06:03 +02:00
parent d1eed44341
commit 104c9904a5
+5 -1
View File
@@ -2091,7 +2091,11 @@ class Handler(BaseHTTPRequestHandler):
self.end_headers() self.end_headers()
try: try:
while True: while True:
chunk = resp.read(16384) # ``read(n)`` may wait until the complete buffer is filled.
# That defeats SSE: a client sees no token for a long time and
# may time out while llama.cpp is already generating. read1()
# returns the next currently available wire chunk instead.
chunk = resp.read1(16384)
if not chunk: if not chunk:
break break
self.wfile.write(chunk) self.wfile.write(chunk)