Files
bhethermanandClaude Sonnet 5 0ba2ea45be Add MTP_N_PARALLEL to entrypoint.sh, bump submodule pointers
entrypoint.sh now honors MTP_N_PARALLEL (default 1, unchanged) for
-np instead of hardcoding 1, so open-webui's docker-compose.mtp.yaml
can request multiple llama-server slots.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-05 13:40:59 -04:00

30 lines
956 B
Bash

#!/usr/bin/env bash
set -euo pipefail
: "${MTP_TARGET_GGUF:=/models/gemma-4-E2B-it-Q4_K_M.gguf}"
: "${MTP_DRAFT_GGUF:=/models/mtp-gemma-4-E2B-it.gguf}"
: "${MTP_PORT:=8030}"
: "${MTP_N_MAX:=2}"
: "${MTP_RELAXED_TOP_N:=5}"
: "${MTP_CTX_SIZE:=8192}"
: "${MTP_BATCH_SIZE:=512}"
: "${MTP_TEMP:=0.6}"
: "${MTP_N_PARALLEL:=1}"
args=(
--model "$MTP_TARGET_GGUF"
--port "$MTP_PORT" --host 0.0.0.0 --no-webui --offline --jinja
--log-verbosity 4 --no-log-prefix --no-log-timestamps
--spec-type draft-mtp --spec-draft-n-max "$MTP_N_MAX"
--spec-draft-relaxed-top-n "$MTP_RELAXED_TOP_N" --spec-draft-backend-sampling
--spec-draft-model "$MTP_DRAFT_GGUF" -ngl 999 --spec-draft-ngl 999
--flash-attn auto -c "$MTP_CTX_SIZE" -b "$MTP_BATCH_SIZE" -ub "$MTP_BATCH_SIZE"
--context-shift --keep 4 -np "$MTP_N_PARALLEL" --temp "$MTP_TEMP"
)
if [[ -n "${MTP_AUTH_TOKEN:-}" ]]; then
args+=(--api-key "$MTP_AUTH_TOKEN")
fi
exec /app/llama-server "${args[@]}"