Files
llama-space/local-llama.sh
T
2026-06-22 11:56:45 +07:00

30 lines
699 B
Bash
Executable File

#!/bin/bash
export PATH=/usr/local/cuda-13.3/bin${PATH:+:${PATH}}
export LD_LIBRARY_PATH=/usr/local/cuda-13.3/lib64${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}
set -euo pipefail
/home/bazzite/Документы/llama-cpp-turboquant/build/bin/llama-server \
-m "models/${MODEL_FILE}" \
--host 0.0.0.0 \
--port 8080 \
-ngl "${N_GPU_LAYERS}" \
--n-cpu-moe "${N_CPU_MOE}" \
--spec-type draft-mtp --spec-draft-n-max 2 \
-c "${CTX_SIZE}" \
-np 1 \
-fa on \
--cache-type-k "turbo4" \
--cache-type-v "turbo3" \
--no-mmap \
--mlock \
--ctx-checkpoints 1 \
--cache-ram -1 \
--jinja \
--reasoning on \
--reasoning-budget -1 \
-b 2048 \
-ub 2048 \
--threads "${THREADS}"