Cap runaway chat generations in router
This commit is contained in:
@@ -567,6 +567,10 @@ services:
|
||||
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||
SWITCH_TIMEOUT: "600"
|
||||
REQUEST_TIMEOUT: "600"
|
||||
# Last-resort guard for every OpenAI-compatible client. Without a
|
||||
# request limit llama.cpp uses n_predict=-1 and a reasoning loop can
|
||||
# consume the complete context before yielding visible output.
|
||||
MAX_GENERATION_TOKENS: "8192"
|
||||
IMAGE_DIR: /data/images
|
||||
IMAGE_WORKER_URL: http://flux-worker:8086
|
||||
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||
|
||||
Reference in New Issue
Block a user