# faster-qwen3-tts, GGML/CPU variant — a THIRD measurement of the same Qwen3-TTS checkpoint:
#
#   qwen3tts/            qwen-tts,    GPU -> no streaming API at all (whole-utterance TTFA)
#   qwen3tts-fast/       torch,       GPU -> CUDA graphs, streaming TTFA on h200
#   this image           qwentts.cpp, CPU -> streaming TTFA on a cpu-* flavor
#   Dockerfile.ggml-cuda qwentts.cpp, GPU -> the control: same runtime, h200
#
# The CPU number is the point: submit_ttfa_jobs.sh's TTFA_DEVICE=cpu exists so our latency is
# comparable with CPU-measured suites such as tts-bench, and the torch backend cannot serve it
# (from_pretrained raises "CUDA graphs require CUDA device" — graph capture IS its acceleration).
#
# No CUDA base image: this variant is CPU-only by construction, and pulling a cuda runtime would
# just inflate an image whose whole purpose is to run on a cpu-* flavor.
#
# ── WHY THIS BUILDS FROM SOURCE INSTEAD OF INSTALLING THE PUBLISHED +cpu WHEEL ────────────────
# The published qwentts-cpp-python 0.3.1+cpu wheel SIGILLs on our CPU flavor. Its build passes
# only -DGGML_BLAS=OFF (scripts/build_native.py) and never sets -DGGML_NATIVE=OFF, so ggml's
# default -march=native applies and the wheel is compiled for whatever the GitHub `ubuntu-24.04`
# runner had — an Intel part with AVX-512. HF's cpu-upgrade flavor is an AMD EPYC 7R13 (Zen 3):
# avx2 / fma / f16c / bmi2, NO avx512f, NO amx_tile. Weights load fine, then the first AVX-512
# kernel traps:
#     [Pipeline] Ready: hop 1920 samples @ 24000 Hz mono, 16 codebooks @ 12.5 Hz
#     Illegal instruction (core dumped)
# (The "AMX is not ready to be used!" line just above it is a red herring — that CPU has no AMX,
# so ggml probed, failed, and warned; the fatal instruction was something else.)
#
# So the native lib is rebuilt here with an explicitly portable ISA baseline. AVX2/FMA/F16C/BMI2
# is what llama.cpp itself ships portable releases against and is universal on x86 servers; a bare
# GGML_NATIVE=OFF would fall back to baseline x86-64 with no AVX at all and yield an
# unrepresentatively slow CPU TTFA, which is worse than no number.
FROM ubuntu:24.04

ENV DEBIAN_FRONTEND=noninteractive

# Fail a stalled mirror fast instead of hanging the build for hours on one package.
RUN printf 'Acquire::Retries "5";\nAcquire::http::Timeout "30";\nAcquire::https::Timeout "30";\n' \
    > /etc/apt/apt.conf.d/99-retries-timeout

# cmake/ninja/build-essential/patchelf are for the native rebuild above; binutils is for the
# objdump check at the end. The runtime-only deps are python, ffmpeg and libsndfile1.
RUN apt-get update && apt-get install -y --no-install-recommends \
    python3 \
    python3-pip \
    python3-dev \
    git \
    ca-certificates \
    cmake \
    ninja-build \
    build-essential \
    patchelf \
    binutils \
    ffmpeg \
    libsndfile1 \
    && rm -rf /var/lib/apt/lists/*

RUN ln -sf /usr/bin/python3 /usr/bin/python
ENV PIP_BREAK_SYSTEM_PACKAGES=1

WORKDIR /app

RUN pip install --no-cache-dir --upgrade --ignore-installed pip setuptools wheel

# CPU torch, from the cpu index. faster-qwen3-tts declares torch>=2.5.1 and would otherwise pull
# the default CUDA wheel — ~2.5 GB of GPU runtime that this image can never use.
RUN pip install --no-cache-dir torch==2.8.0 --index-url https://download.pytorch.org/whl/cpu

# Both pins match what upstream's own CPU wheel job builds, so this differs from the published
# wheel ONLY in the ISA flags below — not in the code being measured.
#   v0.3.1                                   = the bindings release we would have installed
#   7df559a8ca25f66fee02970514ebe5f01dee9055 = the qwentts.cpp ref that job pins (QWENTTS_REF)
# The qwentts.cpp pin is fetched BY SHA rather than cloned-then-checked-out, which is what
# upstream's CI does. That commit is no longer reachable from any branch of qwentts.cpp — master
# has moved on — so `git clone` never transfers the object and the checkout dies with
#   fatal: reference is not a tree: 7df559a8ca25f66fee02970514ebe5f01dee9055
# even though the sha still resolves through GitHub's API. `git fetch <sha>` gets it regardless
# (GitHub serves unreachable objects on request), and --depth 1 makes it cheaper than the full
# recursive clone it replaces. Upstream's own workflow would fail the same way today.
RUN git clone --depth 1 --branch v0.3.1 \
        https://github.com/andimarafioti/qwentts-cpp-python.git /opt/qwentts-cpp-python \
    && mkdir -p /opt/qwentts-cpp-python/third_party/qwentts.cpp \
    && cd /opt/qwentts-cpp-python/third_party/qwentts.cpp \
    && git init -q . \
    && git remote add origin https://github.com/ServeurpersoCom/qwentts.cpp \
    && git fetch --depth 1 origin 7df559a8ca25f66fee02970514ebe5f01dee9055 \
    && git checkout -q FETCH_HEAD \
    && git submodule update --init --recursive --depth 1

# The whole point of building from source. build_native.py appends $QWENTTS_CPP_CMAKE_ARGS to its
# cmake invocation, so no patching is needed. AVX-512/AMX are forced off explicitly as well as
# implicitly (NATIVE=OFF) so a future ggml default cannot quietly reintroduce the crash.
ENV QWENTTS_CPP_CMAKE_ARGS="-DGGML_NATIVE=OFF -DGGML_AVX=ON -DGGML_AVX2=ON -DGGML_FMA=ON -DGGML_F16C=ON -DGGML_BMI2=ON -DGGML_AVX512=OFF -DGGML_AMX_TILE=OFF -DGGML_AMX_INT8=OFF -DGGML_AMX_BF16=OFF"

# set_local_version.py stamps the "+cpu" PEP 440 local version that the published wheels carry.
# Not cosmetic: run_eval.py reads it back via importlib.metadata to prove which runtime computed
# the number, and refuses --device=cuda on a +cpu build (and vice versa).
RUN cd /opt/qwentts-cpp-python \
    && python scripts/set_local_version.py cpu \
    && python scripts/build_native.py \
        --source third_party/qwentts.cpp \
        --backend cpu \
        --clean \
        --cmake-arg=-G \
        --cmake-arg=Ninja \
    && pip install --no-cache-dir .

# Pinned for the same reason as the other images: chunking and the runtime are exactly what the
# TTFA number describes, so an unpinned upgrade would silently change the measurement.
RUN pip install --no-cache-dir "faster-qwen3-tts==0.4.0"

RUN pip install --no-cache-dir datasets tqdm soundfile

# Fail the BUILD, not a job. The zmm check is the one that matters: it disassembles the freshly
# built native library and asserts no AVX-512 register ever appears, which is the exact failure
# this image exists to avoid — "the cmake flags took" is otherwise only discoverable by crashing
# on the target CPU 20 minutes into a job.
RUN python3 -c "import torch, transformers, faster_qwen3_tts; \
from importlib.metadata import version; \
from qwentts_cpp import QwenTTS, load_speaker_embedding; \
from faster_qwen3_tts.ggml_backend import GGMLQwen3TTS; \
v = version('qwentts-cpp-python'); \
print('torch', torch.__version__, '(cuda', torch.version.cuda, ') / qwentts-cpp-python', v); \
assert not torch.version.cuda, 'expected a CPU torch build'; \
assert v.endswith('+cpu'), f'expected a +cpu local version, got {v}'; \
assert hasattr(GGMLQwen3TTS, 'generate_custom_voice_streaming'), 'streaming API missing'; \
assert hasattr(GGMLQwen3TTS, 'generate_voice_clone_streaming'), 'streaming API missing'"

RUN python3 -c "import glob, os, subprocess, qwentts_cpp; \
pkg = os.path.dirname(qwentts_cpp.__file__); \
libs = glob.glob(os.path.join(pkg, '**', '*.so*'), recursive=True); \
assert libs, f'no shared library found under {pkg}'; \
counts = {os.path.basename(l): subprocess.run(['objdump', '-d', '--no-show-raw-insn', l], capture_output=True, text=True).stdout.count('%zmm') for l in libs}; \
print('AVX-512 (zmm) operands per library:', counts); \
assert sum(counts.values()) == 0, 'AVX-512 found: GGML_NATIVE=OFF did not take, and this will SIGILL on the AMD EPYC 7R13 that HF cpu-* flavors use'; \
print('portable ISA baseline confirmed: no AVX-512 in any built library')"

COPY . /app

ENTRYPOINT ["bash"]

EXPOSE 7860
CMD ["-c", "python3 -m http.server 7860"]
