Use Qwen3-ASR as production speech recognizer

This commit is contained in:
Mikei386
2026-09-25 21:31:33 +02:00
parent 6378b50086
commit 8a323e5b9e
17 changed files with 294 additions and 66 deletions
+64 -28
View File
@@ -658,7 +658,7 @@ services:
MUSIC_START_TIMEOUT: "600"
VOICE_CHANGE_START_TIMEOUT: "600"
APPLIO_START_TIMEOUT: "900"
STT_WORKER_URL: http://whisper:8084
STT_WORKER_URL: http://qwen-asr-worker:8084
STT_TIMEOUT: "300"
networks: [frontend, control, inference]
security_opt: ["no-new-privileges:true"]
@@ -681,7 +681,7 @@ services:
condition: service_healthy
tts-gateway:
condition: service_healthy
whisper:
qwen-asr-worker:
condition: service_healthy
image-worker:
@@ -976,40 +976,77 @@ services:
retries: 12
start_period: 10s
whisper:
build:
context: .
dockerfile: platform/docker/whisper/Dockerfile
args:
WHISPER_CPP_VERSION: ${WHISPER_CPP_VERSION:-v1.9.1}
image: mike-ai/whisper:local
container_name: mike-ai-whisper
qwen-asr:
image: ${LLAMA_CPU_IMAGE:-mike-ai/llama.cpp-cpu:local}
container_name: mike-ai-qwen-asr
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:size=2g,mode=1777
- /tmp:size=256m,mode=1777
volumes:
- whisper-data:/models
environment:
WHISPER_HOST: 0.0.0.0
WHISPER_PORT: "8084"
WHISPER_CLI: /opt/whisper.cpp/build/bin/whisper-cli
WHISPER_MODEL: /models/ggml-small.bin
WHISPER_MODEL_URL: ${WHISPER_MODEL_URL:-https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin}
WHISPER_SERVER_PORT: "8085"
WHISPER_THREADS: ${WHISPER_THREADS:-8}
WHISPER_LANGUAGE: ${WHISPER_LANGUAGE:-de}
networks: [frontend, inference]
- "${QWEN_ASR_MODEL_DIR:-/data/models/qwen3-asr-0.6b-q8}:/models:ro"
command:
- --model
- /models/Qwen3-ASR-0.6B-Q8_0.gguf
- --mmproj
- /models/mmproj-Qwen3-ASR-0.6B-Q8_0.gguf
- --no-mmproj-offload
- --n-gpu-layers
- "0"
- --alias
- qwen3-asr-0.6b
- --ctx-size
- "4096"
- --threads
- "6"
- --parallel
- "1"
- --host
- 0.0.0.0
- --port
- "8080"
- --no-ui
- --fit
- "off"
cpus: 6
mem_limit: 6g
networks: [inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
# The entrypoint supervises whisper-server after dropping it to uid 10004.
cap_add: [CHOWN, SETUID, SETGID, KILL]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8084/status"]
test: [CMD, curl, -fsS, "http://127.0.0.1:8080/health"]
interval: 10s
timeout: 5s
retries: 90
start_period: 20m
retries: 12
start_period: 30s
qwen-asr-worker:
build:
context: .
dockerfile: platform/docker/qwen-asr-worker/Dockerfile
image: mike-ai/qwen-asr-worker:local
container_name: mike-ai-qwen-asr-worker
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
environment:
QWEN_ASR_HOST: 0.0.0.0
QWEN_ASR_PORT: "8084"
QWEN_ASR_LANGUAGE: de
QWEN_ASR_SERVER_URL: http://qwen-asr:8080
networks: [inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
depends_on:
qwen-asr:
condition: service_healthy
healthcheck:
test: [CMD, python, -c, "import json,urllib.request; assert json.load(urllib.request.urlopen('http://127.0.0.1:8084/status', timeout=2))['ready']"]
interval: 10s
timeout: 5s
retries: 12
start_period: 15s
llama-dashboard:
build: ./platform/llama-dashboard
@@ -1124,7 +1161,6 @@ networks:
name: mike-ai-tools-egress
volumes:
whisper-data:
router-state:
router-images:
portainer-data: