Files
odysseus/scripts/serve_ajax_preheretic.sh
T

32 lines
1.4 KiB
Bash

#!/usr/bin/env bash
# Run on Ajax. Pre-heretic BF16, four TP2 replicas; no weight modifications.
set -euo pipefail
: "${VLLM_BIN:?Set VLLM_BIN to the absolute vLLM executable path}"
case "$VLLM_BIN" in
/*) ;;
*) printf '%s\n' 'VLLM_BIN must be an absolute executable path' >&2; exit 2 ;;
esac
case "$VLLM_BIN" in
*:*|*$'\n'*) printf '%s\n' 'VLLM_BIN must not contain PATH separators or newlines' >&2; exit 2 ;;
esac
if [ ! -f "$VLLM_BIN" ] || [ ! -x "$VLLM_BIN" ]; then
printf '%s\n' 'VLLM_BIN must name an existing executable file' >&2
exit 2
fi
VLLM_BIN_DIR="${VLLM_BIN%/*}"
export PATH="${VLLM_BIN_DIR:-/}:/usr/local/bin:/usr/bin:/bin"
export NCCL_P2P_DISABLE=1
# Installed FlashInfer sampling JIT fails against the installed CUB headers.
# vLLM's native sampler avoids that optional kernel compilation.
export VLLM_USE_FLASHINFER_SAMPLER=0
exec "${VLLM_BIN:-vllm}" serve \
"${MODEL_PATH:?Set MODEL_PATH explicitly}" \
--served-model-name odysseus-qwen3.5-tools-pre-heretic \
--host 0.0.0.0 --port 19184 --dtype bfloat16 \
--tensor-parallel-size 2 --data-parallel-size 4 --data-parallel-size-local 4 \
--distributed-executor-backend mp --disable-custom-all-reduce \
--gpu-memory-utilization 0.9 --max-model-len 16384 --max-num-seqs 8 \
--enforce-eager --trust-remote-code --enable-auto-tool-choice \
--tool-call-parser qwen3_coder --limit-mm-per-prompt '{"image":3,"video":0}' \
--gdn-prefill-backend triton --disable-log-stats