migrate to router mode: use models.ini for multi-model support
This commit is contained in:
+5
-21
@@ -5,25 +5,9 @@ export LD_LIBRARY_PATH=/usr/local/cuda-13.3/lib64${LD_LIBRARY_PATH:+:${LD_LIBRAR
|
|||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
|
DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
|
||||||
/home/bazzite/Документы/llama-cpp-turboquant/build/bin/llama-server \
|
/home/bazzite/Документы/llama-cpp-turboquant/build/bin/llama-server \
|
||||||
-m "models/${MODEL_FILE}" \
|
--models-preset "${DIR}/models.ini" \
|
||||||
--host 0.0.0.0 \
|
--models-dir "${DIR}/models" \
|
||||||
--port 8080 \
|
-np 1
|
||||||
-ngl "${N_GPU_LAYERS}" \
|
|
||||||
--n-cpu-moe "${N_CPU_MOE}" \
|
|
||||||
--spec-type draft-mtp --spec-draft-n-max 2 \
|
|
||||||
-c "${CTX_SIZE}" \
|
|
||||||
-np 1 \
|
|
||||||
-fa on \
|
|
||||||
--cache-type-k "turbo4" \
|
|
||||||
--cache-type-v "turbo3" \
|
|
||||||
--no-mmap \
|
|
||||||
--mlock \
|
|
||||||
--ctx-checkpoints 1 \
|
|
||||||
--cache-ram -1 \
|
|
||||||
--jinja \
|
|
||||||
--reasoning on \
|
|
||||||
--reasoning-budget -1 \
|
|
||||||
-b 2048 \
|
|
||||||
-ub 2048 \
|
|
||||||
--threads "${THREADS}"
|
|
||||||
|
|||||||
+24
@@ -0,0 +1,24 @@
|
|||||||
|
[DEFAULT]
|
||||||
|
host = 0.0.0.0
|
||||||
|
port = 8080
|
||||||
|
flash-attn = on
|
||||||
|
chat-template = jinja
|
||||||
|
reasoning = on
|
||||||
|
reasoning-budget = -1
|
||||||
|
cache-type-k = turbo4
|
||||||
|
cache-type-v = turbo3
|
||||||
|
no-mmap = true
|
||||||
|
mlock = true
|
||||||
|
ctx-checkpoints = true
|
||||||
|
cache-ram = -1
|
||||||
|
batch-size = 2048
|
||||||
|
ubatch-size = 2048
|
||||||
|
|
||||||
|
[Qwen3.6-35B]
|
||||||
|
model = models/Qwen3.6-35B-A3B-MTP-MXFP4_MOE.gguf
|
||||||
|
n-gpu-layers = 99
|
||||||
|
n-cpu-moe = 29
|
||||||
|
ctx-size = 170000
|
||||||
|
threads = 12
|
||||||
|
spec-type = draft-mtp
|
||||||
|
spec-draft-n-max = 2
|
||||||
Reference in New Issue
Block a user