Files
bhethermanandClaude Sonnet 5 627904e130 Add relaxed-MTP llama-server build and wire it into the stack
mtp-relaxed-decoding/ holds the patch, build, and guide for a custom
llama-server with relaxed-acceptance MTP speculative decoding
(see its README for the full writeup). Bumps the open-webui submodule
to the commit that adds it as a new service, connection, and the
ollama-auth proxy swap.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-04 11:33:31 -04:00

55 lines
2.4 KiB
Docker

# Builds a custom llama-server with the relaxed-acceptance MTP speculative
# decoding patch (see README.md). Pinned to the exact llama.cpp commit the
# patch was written against, so `git apply` doesn't hit source drift.
FROM nvidia/cuda:12.6.3-devel-ubuntu24.04 AS build
RUN apt-get update && apt-get install -y --no-install-recommends \
build-essential cmake git ca-certificates \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /src
ARG LLAMA_CPP_COMMIT=d222767c7a6516559a3f49e7721b6c6b1acc87b4
RUN git init -q . \
&& git remote add origin https://github.com/ggml-org/llama.cpp.git \
&& git fetch --depth 1 origin "$LLAMA_CPP_COMMIT" \
&& git checkout -q FETCH_HEAD
COPY relaxed-top-n.patch .
RUN git apply relaxed-top-n.patch
# CMAKE_CUDA_ARCHITECTURES is hardcoded (rather than `native`) because this
# build has no GPU access to autodetect from. 61 = Pascal / compute 6.1,
# i.e. the Titan Xp this stack actually runs on -- change if the target GPU
# changes.
#
# The devel image ships a link-time libcuda.so *stub* (real libcuda.so.1
# only exists on a machine with the driver installed, which this build
# container isn't). LIBRARY_PATH alone wasn't enough to get it picked up --
# ggml-cuda's CMakeLists doesn't route through FindCUDAToolkit's own stub
# search here -- so force it via explicit linker flags on every link step,
# and symlink the SONAME ld actually looks for (libcuda.so.1) since the
# stub only ships as libcuda.so.
RUN ln -sf libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1
RUN cmake -B build -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=61 \
-DCMAKE_BUILD_TYPE=Release -DLLAMA_CURL=OFF \
-DCMAKE_EXE_LINKER_FLAGS="-L/usr/local/cuda/lib64/stubs -lcuda" \
-DCMAKE_SHARED_LINKER_FLAGS="-L/usr/local/cuda/lib64/stubs -lcuda" \
&& cmake --build build --config Release -j2 --target llama-server
FROM nvidia/cuda:12.6.3-runtime-ubuntu24.04 AS runtime
RUN apt-get update && apt-get install -y --no-install-recommends libgomp1 \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
# llama-server + every shared lib it needs land in the same bin/ dir; ggml's
# dynamic backend loader scans the executable's own directory, so copying it
# whole means CUDA is found with no extra env vars.
COPY --from=build /src/build/bin/ /app/
COPY entrypoint.sh /app/entrypoint.sh
RUN chmod +x /app/entrypoint.sh /app/llama-server
EXPOSE 8030
ENTRYPOINT ["/app/entrypoint.sh"]