# Chatterbox hard-pins torch==2.6.0 (cu124 wheels), so it gets its own image.
FROM nvidia/cuda:12.4.1-runtime-ubuntu22.04

# Avoid interactive prompts during package installation
ENV DEBIAN_FRONTEND=noninteractive

# Install Python and system dependencies. git is required at pip-install time because
# chatterbox-tts pulls resemble-perth from a git URL; ffmpeg/libsndfile1 for audio I/O.
RUN apt-get update && apt-get install -y --no-install-recommends \
    python3 \
    python3-pip \
    python3-dev \
    git \
    libsndfile1 \
    ffmpeg \
    && rm -rf /var/lib/apt/lists/*

# Set Python alias
RUN ln -sf /usr/bin/python3 /usr/bin/python
ENV PIP_BREAK_SYSTEM_PACKAGES=1

WORKDIR /app

# Install PyTorch first at Chatterbox's pinned version so its `torch==2.6.0` requirement is
# already satisfied and pip does not try to pull a different (CPU) wheel.
RUN pip install --no-cache-dir \
    torch==2.6.0 \
    torchaudio==2.6.0 \
    --index-url https://download.pytorch.org/whl/cu124

# Chatterbox (brings transformers==5.2.0, diffusers, s3tokenizer, resemble-perth [git], etc.).
RUN pip install --no-cache-dir chatterbox-tts

# datasets + tqdm + soundfile for the eval loop (soundfile saves wavs AND decodes the
# prompt_audio voice-clone reference column). datasets is pinned <4.0 on purpose: 4.0 switched
# the Audio decoder to torchcodec, but chatterbox's torch==2.6 caps torchcodec at 0.2, which
# predates the torchcodec.decoders.AudioDecoder that datasets 4.x imports. Staying on 3.x keeps
# the soundfile backend and sidesteps the torchcodec/torch version bind entirely.
RUN pip install --no-cache-dir "datasets<4.0" tqdm soundfile

# Copy the full repository
COPY . /app

# Default entrypoint
ENTRYPOINT ["bash"]

# Keep-alive CMD so the Space runtime stays healthy; `docker run` overrides it.
EXPOSE 7860
CMD ["-c", "python3 -m http.server 7860"]
