Stream llama responses without proxy buffering
This commit is contained in:
@@ -2091,7 +2091,11 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
self.end_headers()
|
self.end_headers()
|
||||||
try:
|
try:
|
||||||
while True:
|
while True:
|
||||||
chunk = resp.read(16384)
|
# ``read(n)`` may wait until the complete buffer is filled.
|
||||||
|
# That defeats SSE: a client sees no token for a long time and
|
||||||
|
# may time out while llama.cpp is already generating. read1()
|
||||||
|
# returns the next currently available wire chunk instead.
|
||||||
|
chunk = resp.read1(16384)
|
||||||
if not chunk:
|
if not chunk:
|
||||||
break
|
break
|
||||||
self.wfile.write(chunk)
|
self.wfile.write(chunk)
|
||||||
|
|||||||
Reference in New Issue
Block a user