Stream llama responses without proxy buffering
This commit is contained in:
@@ -2091,7 +2091,11 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self.end_headers()
|
||||
try:
|
||||
while True:
|
||||
chunk = resp.read(16384)
|
||||
# ``read(n)`` may wait until the complete buffer is filled.
|
||||
# That defeats SSE: a client sees no token for a long time and
|
||||
# may time out while llama.cpp is already generating. read1()
|
||||
# returns the next currently available wire chunk instead.
|
||||
chunk = resp.read1(16384)
|
||||
if not chunk:
|
||||
break
|
||||
self.wfile.write(chunk)
|
||||
|
||||
Reference in New Issue
Block a user