# -runtime is enough: Paradee is pure ONNX Runtime inference (no CUDA extensions to build). The eval
# runs the CUDA EP (--device=cuda); --device=cpu uses the package's own CPU session.
FROM nvidia/cuda:12.9.0-runtime-ubuntu24.04

# Avoid interactive prompts during package installation
ENV DEBIAN_FRONTEND=noninteractive

# Install Python and system dependencies (libsndfile1 for audio I/O, git for the pip install).
# No apt espeak-ng: misaki's out-of-dictionary fallback loads the espeak-ng shipped in the
# espeakng-loader wheel.
RUN apt-get update && apt-get install -y --no-install-recommends \
    python3 \
    python3-pip \
    python3-dev \
    git \
    libsndfile1 \
    && rm -rf /var/lib/apt/lists/*

# Set Python alias (Ubuntu 24.04 ships Python 3.12)
RUN ln -sf /usr/bin/python3 /usr/bin/python

# Allow pip to install packages system-wide in the container (PEP 668)
ENV PIP_BREAK_SYSTEM_PACKAGES=1

WORKDIR /app

# Upgrade pip so it fetches prebuilt manylinux wheels. Debian-installed pip has no RECORD
# file and cannot be uninstalled, hence --ignore-installed.
RUN pip install --no-cache-dir --upgrade --ignore-installed pip setuptools wheel

# Paradee (pinned to the v1.0 release commit) + its misaki/spaCy G2P, and the eval-loop deps.
# `onnx` is used by run_eval.py to count parameters from the graph initializers.
RUN pip install --no-cache-dir \
    "paradee @ git+https://github.com/sahilmahendrakar/paradee@9c8b4d7504cbee7e64de2d0341bb690f0b0ab708" \
    soundfile datasets tqdm onnx

# spaCy English model used by misaki's G2P. misaki pip-installs it on first use otherwise, which
# would land inside the first (warm-up) generation of every job.
RUN pip install --no-cache-dir \
    "en_core_web_sm @ https://github.com/explosion/spacy-models/releases/download/en_core_web_sm-3.8.0/en_core_web_sm-3.8.0-py3-none-any.whl"

# Swap the CPU onnxruntime that paradee pulls in for onnxruntime-gpu (which also has the CPU EP).
# Done LAST so no later install re-resolves paradee's `onnxruntime` requirement.
# 1.23.x ships CUDA 12.x builds (matches the 12.9 base; 1.24+ needs CUDA 13) and needs cuDNN 9,
# which the -runtime base image does NOT include -> pull it from pip and expose it.
RUN pip uninstall -y onnxruntime && pip install --no-cache-dir \
    onnxruntime-gpu==1.23.0 \
    nvidia-cudnn-cu12
ENV LD_LIBRARY_PATH=/usr/local/lib/python3.12/dist-packages/nvidia/cudnn/lib:${LD_LIBRARY_PATH}

# Bake the model (int8 + fp32 graphs, config) at the revision the package pins, and smoke-test the
# whole CPU path (spaCy, misaki lexicon, espeak fallback on an OOD word, ONNX) so a broken install
# fails the BUILD rather than a job. Also warms everything that is otherwise fetched on first use.
RUN python -c "\
from huggingface_hub import hf_hub_download; \
from paradee.tts import REPO, REVISION; \
[hf_hub_download(REPO, f, revision=REVISION) for f in ('onnx/paradee_int8.onnx', 'onnx/paradee.onnx', 'config.json')]" \
    && python -c "\
from paradee import Paradee, SAMPLE_RATE; \
tts = Paradee(); \
print(tts.phonemize('Zyxomblat is not a word.')); \
a = tts('Paradee is a small voice that runs anywhere. It was distilled from Kokoro.'); \
assert a.size > SAMPLE_RATE, a.shape; print('ok', a.shape)" \
    && python -c "import onnxruntime as ort; p = ort.get_available_providers(); print(p); assert 'CUDAExecutionProvider' in p"

# Copy the full repository
COPY . /app

# Default entrypoint
ENTRYPOINT ["bash"]

# Keep-alive CMD so the Space runtime stays healthy; `docker run` overrides it.
EXPOSE 7860
CMD ["-c", "python3 -m http.server 7860"]
