mtp-relaxed-decoding/ holds the patch, build, and guide for a custom llama-server with relaxed-acceptance MTP speculative decoding (see its README for the full writeup). Bumps the open-webui submodule to the commit that adds it as a new service, connection, and the ollama-auth proxy swap. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
55 lines
2.4 KiB
Docker
55 lines
2.4 KiB
Docker
# Builds a custom llama-server with the relaxed-acceptance MTP speculative
|
|
# decoding patch (see README.md). Pinned to the exact llama.cpp commit the
|
|
# patch was written against, so `git apply` doesn't hit source drift.
|
|
|
|
FROM nvidia/cuda:12.6.3-devel-ubuntu24.04 AS build
|
|
|
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
build-essential cmake git ca-certificates \
|
|
&& rm -rf /var/lib/apt/lists/*
|
|
|
|
WORKDIR /src
|
|
ARG LLAMA_CPP_COMMIT=d222767c7a6516559a3f49e7721b6c6b1acc87b4
|
|
RUN git init -q . \
|
|
&& git remote add origin https://github.com/ggml-org/llama.cpp.git \
|
|
&& git fetch --depth 1 origin "$LLAMA_CPP_COMMIT" \
|
|
&& git checkout -q FETCH_HEAD
|
|
|
|
COPY relaxed-top-n.patch .
|
|
RUN git apply relaxed-top-n.patch
|
|
|
|
# CMAKE_CUDA_ARCHITECTURES is hardcoded (rather than `native`) because this
|
|
# build has no GPU access to autodetect from. 61 = Pascal / compute 6.1,
|
|
# i.e. the Titan Xp this stack actually runs on -- change if the target GPU
|
|
# changes.
|
|
#
|
|
# The devel image ships a link-time libcuda.so *stub* (real libcuda.so.1
|
|
# only exists on a machine with the driver installed, which this build
|
|
# container isn't). LIBRARY_PATH alone wasn't enough to get it picked up --
|
|
# ggml-cuda's CMakeLists doesn't route through FindCUDAToolkit's own stub
|
|
# search here -- so force it via explicit linker flags on every link step,
|
|
# and symlink the SONAME ld actually looks for (libcuda.so.1) since the
|
|
# stub only ships as libcuda.so.
|
|
RUN ln -sf libcuda.so /usr/local/cuda/lib64/stubs/libcuda.so.1
|
|
RUN cmake -B build -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=61 \
|
|
-DCMAKE_BUILD_TYPE=Release -DLLAMA_CURL=OFF \
|
|
-DCMAKE_EXE_LINKER_FLAGS="-L/usr/local/cuda/lib64/stubs -lcuda" \
|
|
-DCMAKE_SHARED_LINKER_FLAGS="-L/usr/local/cuda/lib64/stubs -lcuda" \
|
|
&& cmake --build build --config Release -j2 --target llama-server
|
|
|
|
FROM nvidia/cuda:12.6.3-runtime-ubuntu24.04 AS runtime
|
|
|
|
RUN apt-get update && apt-get install -y --no-install-recommends libgomp1 \
|
|
&& rm -rf /var/lib/apt/lists/*
|
|
|
|
WORKDIR /app
|
|
# llama-server + every shared lib it needs land in the same bin/ dir; ggml's
|
|
# dynamic backend loader scans the executable's own directory, so copying it
|
|
# whole means CUDA is found with no extra env vars.
|
|
COPY --from=build /src/build/bin/ /app/
|
|
COPY entrypoint.sh /app/entrypoint.sh
|
|
RUN chmod +x /app/entrypoint.sh /app/llama-server
|
|
|
|
EXPOSE 8030
|
|
ENTRYPOINT ["/app/entrypoint.sh"]
|