Add MTP_N_PARALLEL to entrypoint.sh, bump submodule pointers
entrypoint.sh now honors MTP_N_PARALLEL (default 1, unchanged) for -np instead of hardcoding 1, so open-webui's docker-compose.mtp.yaml can request multiple llama-server slots. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -9,6 +9,7 @@ set -euo pipefail
|
|||||||
: "${MTP_CTX_SIZE:=8192}"
|
: "${MTP_CTX_SIZE:=8192}"
|
||||||
: "${MTP_BATCH_SIZE:=512}"
|
: "${MTP_BATCH_SIZE:=512}"
|
||||||
: "${MTP_TEMP:=0.6}"
|
: "${MTP_TEMP:=0.6}"
|
||||||
|
: "${MTP_N_PARALLEL:=1}"
|
||||||
|
|
||||||
args=(
|
args=(
|
||||||
--model "$MTP_TARGET_GGUF"
|
--model "$MTP_TARGET_GGUF"
|
||||||
@@ -18,7 +19,7 @@ args=(
|
|||||||
--spec-draft-relaxed-top-n "$MTP_RELAXED_TOP_N" --spec-draft-backend-sampling
|
--spec-draft-relaxed-top-n "$MTP_RELAXED_TOP_N" --spec-draft-backend-sampling
|
||||||
--spec-draft-model "$MTP_DRAFT_GGUF" -ngl 999 --spec-draft-ngl 999
|
--spec-draft-model "$MTP_DRAFT_GGUF" -ngl 999 --spec-draft-ngl 999
|
||||||
--flash-attn auto -c "$MTP_CTX_SIZE" -b "$MTP_BATCH_SIZE" -ub "$MTP_BATCH_SIZE"
|
--flash-attn auto -c "$MTP_CTX_SIZE" -b "$MTP_BATCH_SIZE" -ub "$MTP_BATCH_SIZE"
|
||||||
--context-shift --keep 4 -np 1 --temp "$MTP_TEMP"
|
--context-shift --keep 4 -np "$MTP_N_PARALLEL" --temp "$MTP_TEMP"
|
||||||
)
|
)
|
||||||
|
|
||||||
if [[ -n "${MTP_AUTH_TOKEN:-}" ]]; then
|
if [[ -n "${MTP_AUTH_TOKEN:-}" ]]; then
|
||||||
|
|||||||
+1
-1
Submodule open-webui updated: 6f583deff9...966140c3ef
+1
-1
Submodule spotify-voice-assistant updated: 7249919006...2ed1ca1e99
Reference in New Issue
Block a user