FROM pytorch/pytorch:2.7.1-cuda12.8-cudnn9-runtime@sha256:c16f4c749e2d9e96878875cdf6cc45cddda1d1a36fddd371dd6f2360f1b6e2a2 RUN apt-get update \ && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends build-essential curl ffmpeg libsndfile1 \ && rm -rf /var/lib/apt/lists/* RUN python -m pip install --no-cache-dir \ "audio-separator[gpu]==0.47.0" \ "onnxruntime-gpu==1.22.0" \ "fastapi==0.116.1" \ "python-multipart==0.0.20" \ "uvicorn[standard]==0.35.0" # audio-separator 0.47 requires NumPy 2 while ClearVoice 0.1.2 still pins # NumPy 1.x. Keep ClearVoice in a small overlay venv but share the image's # CUDA-enabled PyTorch installation instead of duplicating it. RUN python -m venv --system-site-packages /opt/clearvoice-venv \ && /opt/clearvoice-venv/bin/python -m pip install --no-cache-dir \ "clearvoice==0.1.2" \ "numpy>=1.24.3,<2.0" WORKDIR /app COPY app.py index.html speech_enhance.py ./ ENV MODEL_FILENAME=model_bs_roformer_ep_317_sdr_12.9755.ckpt \ MODEL_DIR=/models \ JOB_DIR=/data/jobs EXPOSE 8080 CMD ["sh", "-c", "mkdir -p \"$MODEL_DIR/clearvoice\" && ln -sfn \"$MODEL_DIR/clearvoice\" /app/checkpoints && for model in \"$MODEL_FILENAME\" htdemucs_ft.yaml htdemucs_6s.yaml; do audio-separator --model_filename \"$model\" --model_file_dir \"$MODEL_DIR\" --download_model_only || exit 1; done; /opt/clearvoice-venv/bin/python /app/speech_enhance.py --download-only || exit 1; exec uvicorn app:app --host 0.0.0.0 --port 8080 --workers 1"]