wesleysimplicio commited on
Commit
a0fdb27
·
verified ·
1 Parent(s): df67e5e

docs: sync deploy/serve_vllm.sh with updated benchmark numbers

Browse files
Files changed (1) hide show
  1. deploy/serve_vllm.sh +20 -0
deploy/serve_vllm.sh ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ # ==============================================================================
3
+ # Servidor de Produção vLLM (OpenAI-Compatible API) para Simplicio 27B
4
+ # Compatível com OpenRouter, OpenCode, Aider, Continue.dev e Cursor
5
+ # ==============================================================================
6
+
7
+ MODEL_ID=${1:-"wesleysimplicio/Simplicio-27B"}
8
+ PORT=${2:-8000}
9
+ HOST="0.0.0.0"
10
+
11
+ echo "=== Iniciando vLLM OpenAI-Compatible Server para $MODEL_ID na porta $PORT ==="
12
+
13
+ vllm serve "$MODEL_ID" \
14
+ --host "$HOST" \
15
+ --port "$PORT" \
16
+ --tensor-parallel-size 1 \
17
+ --max-model-len 32768 \
18
+ --gpu-memory-utilization 0.95 \
19
+ --chat-template-preset chatml \
20
+ --served-model-name "simplicio-27b"