# Builds a custom llama-server with the relaxed-acceptance MTP speculative
# decoding patch (see README.md). Pinned to the exact llama.cpp commit the
# patch was written against, so `git apply` doesn't hit source drift.

FROM nvidia/cuda:12.6.3-devel-ubuntu24.04 AS build

RUN apt-get update && apt-get install -y --no-install-recommends \
    build-essential cmake git ca-certificates \
    && rm -rf /var/lib/apt/lists/*

WORKDIR /src
ARG LLAMA_CPP_COMMIT=d222767c7a6516559a3f49e7721b6c6b1acc87b4
RUN git init -q . \
    && git remote add origin https://github.com/ggml-org/llama.cpp.git \
    && git fetch --depth 1 origin "$LLAMA_CPP_COMMIT" \
    && git checkout -q FETCH_HEAD

COPY relaxed-top-n.patch .
RUN git apply relaxed-top-n.patch

# CMAKE_CUDA_ARCHITECTURES is hardcoded (rather than `native`) because this
# build has no GPU access to autodetect from. 61 = Pascal / compute 6.1,
# i.e. the Titan Xp this stack actually runs on -- change if the target GPU
# changes.
#
# The devel image ships a link-time libcuda.so *stub* (real libcuda.so.1
# only exists on a machine with the driver installed, which this build
# container isn't). LIBRARY_PATH alone wasn't enough to get it picked up --
# ggml-cuda's CMakeLists doesn't route through FindCUDAToolkit's own stub
# search here -- so force it via explicit linker flags on every link step,
# and symlink the SONAME ld actually looks for (libcuda.so.1) since the
# stub only ships as libcuda.so.
RUN ln -sf libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1
RUN cmake -B build -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=61 \
      -DCMAKE_BUILD_TYPE=Release -DLLAMA_CURL=OFF \
      -DCMAKE_EXE_LINKER_FLAGS="-L/usr/local/cuda/lib64/stubs -lcuda" \
      -DCMAKE_SHARED_LINKER_FLAGS="-L/usr/local/cuda/lib64/stubs -lcuda" \
    && cmake --build build --config Release -j2 --target llama-server

FROM nvidia/cuda:12.6.3-runtime-ubuntu24.04 AS runtime

RUN apt-get update && apt-get install -y --no-install-recommends libgomp1 \
    && rm -rf /var/lib/apt/lists/*

WORKDIR /app
# llama-server + every shared lib it needs land in the same bin/ dir; ggml's
# dynamic backend loader scans the executable's own directory, so copying it
# whole means CUDA is found with no extra env vars.
COPY --from=build /src/build/bin/ /app/
COPY entrypoint.sh /app/entrypoint.sh
RUN chmod +x /app/entrypoint.sh /app/llama-server

EXPOSE 8030
ENTRYPOINT ["/app/entrypoint.sh"]
