diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..b3ee9f5 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,71 @@ +# syntax=docker/dockerfile:1.7 +# +# Targets: +# runtime — vLLM only, model from HF cache at runtime (default) +# serve — runtime + model baked in +# +# Runtime (default): +# docker build -t moss-td-vllm:dev . +# +# Serve (reuse local HF cache, no re-download): +# docker build --target serve \ +# --build-context models=$HOME/.cache/huggingface \ +# --build-arg MODEL_REVISION=e5118b411bf5a77d7a90c4941066bec93c967312 \ +# -t moss-td-serve:1.0.0 . + +ARG CUDA_IMAGE=nvidia/cuda:13.3.0-cudnn-devel-ubuntu24.04 +FROM ${CUDA_IMAGE} AS runtime + +ENV DEBIAN_FRONTEND=noninteractive \ + PYTHONUNBUFFERED=1 \ + PIP_DISABLE_PIP_VERSION_CHECK=1 \ + PIP_NO_CACHE_DIR=1 \ + UV_HTTP_TIMEOUT=600 + +RUN apt-get update && apt-get install -y --no-install-recommends \ + ca-certificates \ + curl \ + ffmpeg \ + git \ + libsndfile1 \ + python3.12 \ + python3.12-dev \ + python3-pip \ + && rm -rf /var/lib/apt/lists/* + +RUN curl -LsSf https://astral.sh/uv/install.sh | sh + +ENV PATH="/root/.local/bin:${PATH}" +# README: CUDA 13 → cu130 pinned nightly +ARG VLLM_CUDA=cu130 +ARG VLLM_WHEEL_COMMIT=68b4a1d582818e67adc903bf1b8fc5a5447da2fa + +# Install the CUDA PyTorch wheels explicitly so uv does not resolve to torch+xpu. +# Pin to the 2.11 release family to match the CUDA 13 wheels used by the project docs. +RUN uv pip install --system --break-system-packages -U \ + torch==2.11.0 \ + torchvision==0.26.0 \ + torchaudio==2.11.0 \ + --python /usr/bin/python3.12 \ + --index-url "https://download.pytorch.org/whl/${VLLM_CUDA}" + +RUN uv pip install --system --break-system-packages -U "vllm[audio]" \ + --python /usr/bin/python3.12 \ + --extra-index-url "https://wheels.vllm.ai/${VLLM_WHEEL_COMMIT}/${VLLM_CUDA}" + +WORKDIR /workspace +EXPOSE 8000 +CMD ["vllm", "serve", "OpenMOSS-Team/MOSS-Transcribe-Diarize", "--trust-remote-code", "--host", "0.0.0.0", "--port", "8000"] + +# Model baked in; reuses runtime layer cache when only MODEL_REVISION changes. +FROM runtime AS serve + +ARG MODEL_REPO=models--OpenMOSS-Team--MOSS-Transcribe-Diarize +ARG MODEL_REVISION=e5118b411bf5a77d7a90c4941066bec93c967312 + +# HF snapshots are symlinks into hub/blobs; cp -aL dereferences them into real files. +RUN --mount=type=bind,from=models,source=/,target=/hf,ro \ + mkdir -p /models/MOSS-Transcribe-Diarize && \ + cp -aL "/hf/hub/${MODEL_REPO}/snapshots/${MODEL_REVISION}/." /models/MOSS-Transcribe-Diarize/ + +CMD ["vllm", "serve", "/models/MOSS-Transcribe-Diarize", "--served-model-name", "OpenMOSS-Team/MOSS-Transcribe-Diarize", "--trust-remote-code", "--host", "0.0.0.0", "--port", "8000"] diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..fa50363 --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,58 @@ +# vLLM serving for MOSS-Transcribe-Diarize +# +# Build & run: +# docker compose up --build +# +# GPU (runtime env vars, not build args): +# GPUS=all docker compose up --build +# GPUS=1 docker compose up --build +# NVIDIA_VISIBLE_DEVICES=0,1 docker compose up --build +# +# Image tag: +# IMAGE_TAG=1.0.0 docker compose up --build +# +# Serve image with model baked in: +# docker compose --profile serve up --build + +services: + vllm: + build: + context: . + dockerfile: Dockerfile + target: runtime + image: ${IMAGE_NAME:-moss-td-vllm}:${IMAGE_TAG:-dev} + user: root + volumes: + - ${PWD}:/workspace + - ${HF_HOME:-${HOME}/.cache/huggingface}:/root/.cache/huggingface + working_dir: /workspace + ports: + - "8000:8000" + shm_size: "16gb" + environment: + NVIDIA_VISIBLE_DEVICES: ${NVIDIA_VISIBLE_DEVICES:-all} + NVIDIA_DRIVER_CAPABILITIES: ${NVIDIA_DRIVER_CAPABILITIES:-compute,utility} + gpus: ${GPUS:-all} + stdin_open: true + tty: true + + serve: + profiles: [serve] + build: + context: . + dockerfile: Dockerfile + target: serve + additional_contexts: + models: ${HF_HOME:-${HOME}/.cache/huggingface} + args: + MODEL_REVISION: ${MODEL_REVISION:-e5118b411bf5a77d7a90c4941066bec93c967312} + image: ${SERVE_IMAGE:-moss-td-serve}:${IMAGE_TAG:-latest} + user: root + ports: + - "8000:8000" + shm_size: "16gb" + environment: + NVIDIA_VISIBLE_DEVICES: ${NVIDIA_VISIBLE_DEVICES:-all} + NVIDIA_DRIVER_CAPABILITIES: ${NVIDIA_DRIVER_CAPABILITIES:-compute,utility} + gpus: ${GPUS:-all} + restart: unless-stopped