102 lines
3.3 KiB
Docker
102 lines
3.3 KiB
Docker
ARG PYTORCH_BASE_IMAGE=pytorch/pytorch:2.11.0-cuda13.0-cudnn9-runtime
|
|
FROM ${PYTORCH_BASE_IMAGE}
|
|
|
|
ARG CUDA_NVCC_PACKAGE=cuda-nvcc-13-0
|
|
ARG PYTORCH_CUDA_INDEX=https://download.pytorch.org/whl/cu130
|
|
ARG TORCH_VERSION=2.11.0
|
|
ARG TORCHAUDIO_VERSION=2.11.0
|
|
ARG TORCHVISION_VERSION=0.26.0
|
|
ARG VLLM_VERSION=0.20.0
|
|
ARG TORCH_CUDA_ARCH_LIST=12.0+PTX
|
|
|
|
ENV DEBIAN_FRONTEND=noninteractive \
|
|
PYTHONUNBUFFERED=1 \
|
|
PYTHONDONTWRITEBYTECODE=1 \
|
|
HF_HUB_DISABLE_SYMLINKS_WARNING=1 \
|
|
HF_HUB_DISABLE_PROGRESS_BARS=1 \
|
|
PIP_BREAK_SYSTEM_PACKAGES=1 \
|
|
TORCH_CUDA_ARCH_LIST=${TORCH_CUDA_ARCH_LIST}
|
|
|
|
# Add NVIDIA apt repository for CUDA packages
|
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
ca-certificates \
|
|
gnupg \
|
|
wget \
|
|
&& wget -qO - https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/x86_64/3bf863cc.pub | apt-key add - \
|
|
&& echo "deb https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/x86_64 /" > /etc/apt/sources.list.d/cuda.list \
|
|
&& apt-get update
|
|
|
|
# Install system packages required for audio processing
|
|
# Note: nvcc and build-essential are required for FlashInfer JIT compilation.
|
|
RUN apt-get install -y --no-install-recommends \
|
|
ffmpeg \
|
|
sox \
|
|
libsox-dev \
|
|
libsndfile1 \
|
|
nginx \
|
|
build-essential \
|
|
${CUDA_NVCC_PACKAGE}
|
|
|
|
WORKDIR /app
|
|
|
|
# Install Python dependencies directly into the image runtime Python.
|
|
# The image defaults to CUDA 13.0/cu130 for Blackwell-capable GPUs. Developers
|
|
# can rebuild with a different PyTorch CUDA backend by overriding:
|
|
# PYTORCH_BASE_IMAGE, PYTORCH_CUDA_INDEX, CUDA_NVCC_PACKAGE, TORCH_CUDA_ARCH_LIST.
|
|
COPY pyproject.toml /app/
|
|
RUN python - <<'PY' > /tmp/qwen3-asr-gpu-reqs.txt
|
|
import tomllib
|
|
from pathlib import Path
|
|
|
|
data = tomllib.loads(Path("/app/pyproject.toml").read_text())
|
|
for dep in data["project"]["dependencies"]:
|
|
if dep.startswith("torch=="):
|
|
continue
|
|
if dep.startswith("torchaudio=="):
|
|
continue
|
|
if dep.startswith("torchvision=="):
|
|
continue
|
|
if dep.startswith("vllm=="):
|
|
continue
|
|
print(dep)
|
|
PY
|
|
RUN pip install --no-cache-dir -r /tmp/qwen3-asr-gpu-reqs.txt && \
|
|
pip install --no-cache-dir \
|
|
--index-url "${PYTORCH_CUDA_INDEX}" \
|
|
--extra-index-url https://pypi.org/simple \
|
|
"torch==${TORCH_VERSION}" \
|
|
"torchaudio==${TORCHAUDIO_VERSION}" \
|
|
"torchvision==${TORCHVISION_VERSION}" && \
|
|
pip install --no-cache-dir "vllm==${VLLM_VERSION}" && \
|
|
rm -f /tmp/qwen3-asr-gpu-reqs.txt
|
|
|
|
# Fail the image build early if the runtime dependency chain is inconsistent.
|
|
RUN python - <<'PY'
|
|
import torch
|
|
import torchaudio
|
|
import transformers
|
|
import vllm
|
|
from transformers import PreTrainedModel
|
|
|
|
print("torch", torch.__version__)
|
|
print("torchaudio", torchaudio.__version__)
|
|
print("transformers", transformers.__version__)
|
|
print("vllm", vllm.__version__)
|
|
print("PreTrainedModel", PreTrainedModel)
|
|
PY
|
|
|
|
# Clean apt cache but keep build-essential and nvcc for FlashInfer JIT compilation.
|
|
RUN apt-get clean && rm -rf /var/lib/apt/lists/*
|
|
|
|
# Copy application code
|
|
COPY . .
|
|
|
|
# Create runtime directories
|
|
RUN mkdir -p /app/data/temp /app/data/logs /app/data/tasks \
|
|
&& chmod +x start.py /app/scripts/docker/entrypoint.sh
|
|
|
|
EXPOSE 8000
|
|
|
|
ENTRYPOINT ["/app/scripts/docker/entrypoint.sh"]
|
|
CMD ["python", "start.py"]
|