mirror of
https://github.com/pewdiepie-archdaemon/odysseus.git
synced 2026-10-06 06:52:20 +02:00
32 lines
1.4 KiB
Bash
32 lines
1.4 KiB
Bash
#!/usr/bin/env bash
|
|
# Run on Ajax. Pre-heretic BF16, four TP2 replicas; no weight modifications.
|
|
set -euo pipefail
|
|
: "${VLLM_BIN:?Set VLLM_BIN to the absolute vLLM executable path}"
|
|
case "$VLLM_BIN" in
|
|
/*) ;;
|
|
*) printf '%s\n' 'VLLM_BIN must be an absolute executable path' >&2; exit 2 ;;
|
|
esac
|
|
case "$VLLM_BIN" in
|
|
*:*|*$'\n'*) printf '%s\n' 'VLLM_BIN must not contain PATH separators or newlines' >&2; exit 2 ;;
|
|
esac
|
|
if [ ! -f "$VLLM_BIN" ] || [ ! -x "$VLLM_BIN" ]; then
|
|
printf '%s\n' 'VLLM_BIN must name an existing executable file' >&2
|
|
exit 2
|
|
fi
|
|
VLLM_BIN_DIR="${VLLM_BIN%/*}"
|
|
export PATH="${VLLM_BIN_DIR:-/}:/usr/local/bin:/usr/bin:/bin"
|
|
export NCCL_P2P_DISABLE=1
|
|
# Installed FlashInfer sampling JIT fails against the installed CUB headers.
|
|
# vLLM's native sampler avoids that optional kernel compilation.
|
|
export VLLM_USE_FLASHINFER_SAMPLER=0
|
|
exec "${VLLM_BIN:-vllm}" serve \
|
|
"${MODEL_PATH:?Set MODEL_PATH explicitly}" \
|
|
--served-model-name odysseus-qwen3.5-tools-pre-heretic \
|
|
--host 0.0.0.0 --port 19184 --dtype bfloat16 \
|
|
--tensor-parallel-size 2 --data-parallel-size 4 --data-parallel-size-local 4 \
|
|
--distributed-executor-backend mp --disable-custom-all-reduce \
|
|
--gpu-memory-utilization 0.9 --max-model-len 16384 --max-num-seqs 8 \
|
|
--enforce-eager --trust-remote-code --enable-auto-tool-choice \
|
|
--tool-call-parser qwen3_coder --limit-mm-per-prompt '{"image":3,"video":0}' \
|
|
--gdn-prefill-backend triton --disable-log-stats
|