ARG PYTORCH_BASE_IMAGE=pytorch/pytorch:2.11.0-cuda13.0-cudnn9-runtime FROM ${PYTORCH_BASE_IMAGE} ARG CUDA_NVCC_PACKAGE=cuda-nvcc-13-0 ARG PYTORCH_CUDA_INDEX=https://download.pytorch.org/whl/cu130 ARG TORCH_VERSION=2.11.0 ARG TORCHAUDIO_VERSION=2.11.0 ARG TORCHVISION_VERSION=0.26.0 ARG VLLM_VERSION=0.20.0 ARG TORCH_CUDA_ARCH_LIST=12.0+PTX ENV DEBIAN_FRONTEND=noninteractive \ PYTHONUNBUFFERED=1 \ PYTHONDONTWRITEBYTECODE=1 \ HF_HUB_DISABLE_SYMLINKS_WARNING=1 \ HF_HUB_DISABLE_PROGRESS_BARS=1 \ PIP_BREAK_SYSTEM_PACKAGES=1 \ TORCH_CUDA_ARCH_LIST=${TORCH_CUDA_ARCH_LIST} # Add NVIDIA apt repository for CUDA packages RUN apt-get update && apt-get install -y --no-install-recommends \ ca-certificates \ gnupg \ wget \ && wget -qO - https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/x86_64/3bf863cc.pub | apt-key add - \ && echo "deb https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/x86_64 /" > /etc/apt/sources.list.d/cuda.list \ && apt-get update # Install system packages required for audio processing # Note: nvcc and build-essential are required for FlashInfer JIT compilation. RUN apt-get install -y --no-install-recommends \ ffmpeg \ sox \ libsox-dev \ libsndfile1 \ nginx \ build-essential \ ${CUDA_NVCC_PACKAGE} WORKDIR /app # Install Python dependencies directly into the image runtime Python. # The image defaults to CUDA 13.0/cu130 for Blackwell-capable GPUs. Developers # can rebuild with a different PyTorch CUDA backend by overriding: # PYTORCH_BASE_IMAGE, PYTORCH_CUDA_INDEX, CUDA_NVCC_PACKAGE, TORCH_CUDA_ARCH_LIST. COPY pyproject.toml /app/ RUN python - <<'PY' > /tmp/qwen3-asr-gpu-reqs.txt import tomllib from pathlib import Path data = tomllib.loads(Path("/app/pyproject.toml").read_text()) for dep in data["project"]["dependencies"]: if dep.startswith("torch=="): continue if dep.startswith("torchaudio=="): continue if dep.startswith("torchvision=="): continue if dep.startswith("vllm=="): continue print(dep) PY RUN pip install --no-cache-dir -r /tmp/qwen3-asr-gpu-reqs.txt && \ pip install --no-cache-dir \ --index-url "${PYTORCH_CUDA_INDEX}" \ --extra-index-url https://pypi.org/simple \ "torch==${TORCH_VERSION}" \ "torchaudio==${TORCHAUDIO_VERSION}" \ "torchvision==${TORCHVISION_VERSION}" && \ pip install --no-cache-dir "vllm==${VLLM_VERSION}" && \ rm -f /tmp/qwen3-asr-gpu-reqs.txt # Fail the image build early if the runtime dependency chain is inconsistent. RUN python - <<'PY' import torch import torchaudio import transformers import vllm from transformers import PreTrainedModel print("torch", torch.__version__) print("torchaudio", torchaudio.__version__) print("transformers", transformers.__version__) print("vllm", vllm.__version__) print("PreTrainedModel", PreTrainedModel) PY # Clean apt cache but keep build-essential and nvcc for FlashInfer JIT compilation. RUN apt-get clean && rm -rf /var/lib/apt/lists/* # Copy application code COPY . . # Create runtime directories RUN mkdir -p /app/data/temp /app/data/logs /app/data/tasks \ && chmod +x start.py /app/scripts/docker/entrypoint.sh EXPOSE 8000 ENTRYPOINT ["/app/scripts/docker/entrypoint.sh"] CMD ["python", "start.py"]