GEV-26B-Decide / serve.sh
cloudyu's picture
serve_decide.py: answer read-out that works with speculative decoding (+ fallback), system2_only; serve.sh: --max-logprobs 256, MTP=1 option
ea13f62 verified
Raw History Blame Contribute Delete
1.58 kB
#!/usr/bin/env bash
# Serve autotrust/GEV-26B-Decide with vLLM: one engine, both systems, text and images.
# System 2: the unmodified gemma-4-26B-A4B-it (served name autotrust/GEV-26B-Decide) on the OpenAI endpoints
# System 1: POST /v1/decide (2-256 options, adaptive thinking), or the LoRA module "jev-decision" directly
# serve_decide.py is the standard vLLM OpenAI server (same flags) with /v1/decide added.
# Requirements: a vLLM build with Gemma-4 support plus patches/vllm-gemma4-lm-head-lora.patch (LoRA on Gemma-4's tied
# lm_head, vocabulary 262,144), tested with a vLLM development build from September 2026.
# MAX_MODEL_LEN: up to 262144 (the backbone's native context).
# MTP=1: speculative decoding with google/gemma-4-26B-A4B-it-assistant (System 2 about 1.8-1.9x faster; lower System 1
# throughput at high concurrency).
set -e
MODEL_DIR=${MODEL_DIR:-GEV-26B-Decide}
[ -d "$MODEL_DIR" ] || hf download autotrust/GEV-26B-Decide --local-dir "$MODEL_DIR"
EXTRA=()
if [ "${MTP:-0}" = "1" ]; then
EXTRA=(--speculative-config '{"model": "google/gemma-4-26B-A4B-it-assistant", "num_speculative_tokens": 4}')
fi
exec python3 "$MODEL_DIR/serve_decide.py" --model "$MODEL_DIR" --served-model-name autotrust/GEV-26B-Decide \
--enable-lora --max-lora-rank 32 --lora-modules jev-decision="$MODEL_DIR/adapter_vllm" \
--logprobs-mode processed_logprobs --max-logprobs 256 --max-model-len ${MAX_MODEL_LEN:-65536} --enable-prefix-caching \
--max-num-seqs 256 --trust-request-chat-template --limit-mm-per-prompt '{"image": 8}' \
"${EXTRA[@]}" --port ${PORT:-8000}