# faster-qwen3-tts: a SECOND ENGINE for the Qwen3-TTS checkpoints the qwen3tts backend already
# evaluates, used only to measure a streaming time-to-first-audio (see run_eval.py's docstring).
#
# It cannot share the qwen3tts image. That one installs `qwen-tts`, which pins
# transformers==4.57.3; faster-qwen3-tts requires transformers>=5.15.1 AND a different base
# package, `qwen-tts-hf`. Hence a separate image, exactly as the qwen3tts image is separate from
# the transformers-from-git `tts-eval` scorer image for the same reason.
#
# -devel (not -runtime) and cu128 to match the qwen3tts image's torch 2.8.0 cu128 wheels.
FROM nvidia/cuda:12.9.0-devel-ubuntu24.04

ENV DEBIAN_FRONTEND=noninteractive

# Fail a stalled mirror fast instead of hanging the build for hours on one package.
RUN printf 'Acquire::Retries "5";\nAcquire::http::Timeout "30";\nAcquire::https::Timeout "30";\n' \
    > /etc/apt/apt.conf.d/99-retries-timeout

RUN apt-get update && apt-get install -y --no-install-recommends \
    python3 \
    python3-pip \
    python3-dev \
    git \
    ffmpeg \
    libsndfile1 \
    && rm -rf /var/lib/apt/lists/*

RUN ln -sf /usr/bin/python3 /usr/bin/python
ENV PIP_BREAK_SYSTEM_PACKAGES=1

WORKDIR /app

# Upgrade pip so it fetches prebuilt manylinux wheels instead of building from source.
RUN pip install --no-cache-dir --upgrade --ignore-installed pip setuptools wheel

# torch FIRST, from the cu128 index, so the CUDA build is the one we chose rather than whatever
# the default PyPI wheel resolves to when faster-qwen3-tts pulls `torch>=2.5.1` transitively.
RUN pip install --no-cache-dir \
    torch==2.8.0 \
    torchaudio==2.8.0 \
    --index-url https://download.pytorch.org/whl/cu128

# The engine under test. Brings qwen-tts-hf (the Transformers-5-compatible build of the Qwen3-TTS
# modelling code) and transformers>=5.15.1. Pinned to the version measured, because chunking and
# the CUDA-graph capture path are exactly what the TTFA number describes — an unpinned upgrade
# would silently change the measurement.
RUN pip install --no-cache-dir "faster-qwen3-tts==0.4.0"

# datasets + tqdm to load the eval set; torchcodec for its Audio() decoding (voice-clone
# references). soundfile comes in via faster-qwen3-tts.
RUN pip install --no-cache-dir datasets tqdm "torchcodec==0.7.*"

# Fail the BUILD, not a GPU job, if the two packages did not co-resolve.
RUN python3 -c "import transformers, faster_qwen3_tts; \
from faster_qwen3_tts import FasterQwen3TTS; \
print('transformers', transformers.__version__, '/ faster_qwen3_tts', faster_qwen3_tts.__version__); \
assert hasattr(FasterQwen3TTS, 'generate_custom_voice_streaming'), 'streaming API missing'; \
assert hasattr(FasterQwen3TTS, 'generate_voice_clone_streaming'), 'streaming API missing'"

# Copy the full repository (build context is the REPO ROOT, so scripts/ttfa_probe.py comes too).
COPY . /app

ENTRYPOINT ["bash"]

# Keep-alive CMD so the Space runtime stays healthy; `docker run` / `hf jobs run` overrides it.
EXPOSE 7860
CMD ["-c", "python3 -m http.server 7860"]
