entrypoint.sh now honors MTP_N_PARALLEL (default 1, unchanged) for -np instead of hardcoding 1, so open-webui's docker-compose.mtp.yaml can request multiple llama-server slots. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
30 lines
956 B
Bash
30 lines
956 B
Bash
#!/usr/bin/env bash
|
|
set -euo pipefail
|
|
|
|
: "${MTP_TARGET_GGUF:=/models/gemma-4-E2B-it-Q4_K_M.gguf}"
|
|
: "${MTP_DRAFT_GGUF:=/models/mtp-gemma-4-E2B-it.gguf}"
|
|
: "${MTP_PORT:=8030}"
|
|
: "${MTP_N_MAX:=2}"
|
|
: "${MTP_RELAXED_TOP_N:=5}"
|
|
: "${MTP_CTX_SIZE:=8192}"
|
|
: "${MTP_BATCH_SIZE:=512}"
|
|
: "${MTP_TEMP:=0.6}"
|
|
: "${MTP_N_PARALLEL:=1}"
|
|
|
|
args=(
|
|
--model "$MTP_TARGET_GGUF"
|
|
--port "$MTP_PORT" --host 0.0.0.0 --no-webui --offline --jinja
|
|
--log-verbosity 4 --no-log-prefix --no-log-timestamps
|
|
--spec-type draft-mtp --spec-draft-n-max "$MTP_N_MAX"
|
|
--spec-draft-relaxed-top-n "$MTP_RELAXED_TOP_N" --spec-draft-backend-sampling
|
|
--spec-draft-model "$MTP_DRAFT_GGUF" -ngl 999 --spec-draft-ngl 999
|
|
--flash-attn auto -c "$MTP_CTX_SIZE" -b "$MTP_BATCH_SIZE" -ub "$MTP_BATCH_SIZE"
|
|
--context-shift --keep 4 -np "$MTP_N_PARALLEL" --temp "$MTP_TEMP"
|
|
)
|
|
|
|
if [[ -n "${MTP_AUTH_TOKEN:-}" ]]; then
|
|
args+=(--api-key "$MTP_AUTH_TOKEN")
|
|
fi
|
|
|
|
exec /app/llama-server "${args[@]}"
|