# Builds a custom llama-server with the relaxed-acceptance MTP speculative # decoding patch (see README.md). Pinned to the exact llama.cpp commit the # patch was written against, so `git apply` doesn't hit source drift. FROM nvidia/cuda:12.6.3-devel-ubuntu24.04 AS build RUN apt-get update && apt-get install -y --no-install-recommends \ build-essential cmake git ca-certificates \ && rm -rf /var/lib/apt/lists/* WORKDIR /src ARG LLAMA_CPP_COMMIT=d222767c7a6516559a3f49e7721b6c6b1acc87b4 RUN git init -q . \ && git remote add origin https://github.com/ggml-org/llama.cpp.git \ && git fetch --depth 1 origin "$LLAMA_CPP_COMMIT" \ && git checkout -q FETCH_HEAD COPY relaxed-top-n.patch . RUN git apply relaxed-top-n.patch # CMAKE_CUDA_ARCHITECTURES is hardcoded (rather than `native`) because this # build has no GPU access to autodetect from. 61 = Pascal / compute 6.1, # i.e. the Titan Xp this stack actually runs on -- change if the target GPU # changes. # # The devel image ships a link-time libcuda.so *stub* (real libcuda.so.1 # only exists on a machine with the driver installed, which this build # container isn't). LIBRARY_PATH alone wasn't enough to get it picked up -- # ggml-cuda's CMakeLists doesn't route through FindCUDAToolkit's own stub # search here -- so force it via explicit linker flags on every link step, # and symlink the SONAME ld actually looks for (libcuda.so.1) since the # stub only ships as libcuda.so. RUN ln -sf libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1 RUN cmake -B build -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=61 \ -DCMAKE_BUILD_TYPE=Release -DLLAMA_CURL=OFF \ -DCMAKE_EXE_LINKER_FLAGS="-L/usr/local/cuda/lib64/stubs -lcuda" \ -DCMAKE_SHARED_LINKER_FLAGS="-L/usr/local/cuda/lib64/stubs -lcuda" \ && cmake --build build --config Release -j2 --target llama-server FROM nvidia/cuda:12.6.3-runtime-ubuntu24.04 AS runtime RUN apt-get update && apt-get install -y --no-install-recommends libgomp1 \ && rm -rf /var/lib/apt/lists/* WORKDIR /app # llama-server + every shared lib it needs land in the same bin/ dir; ggml's # dynamic backend loader scans the executable's own directory, so copying it # whole means CUDA is found with no extra env vars. COPY --from=build /src/build/bin/ /app/ COPY entrypoint.sh /app/entrypoint.sh RUN chmod +x /app/entrypoint.sh /app/llama-server EXPOSE 8030 ENTRYPOINT ["/app/entrypoint.sh"]