feat: initial commit with vLLM production stack and benchmark tools
This commit is contained in:
@@ -0,0 +1,10 @@
|
|||||||
|
# Target AWQ-quantized model ID
|
||||||
|
MODEL_ID=Qwen/Qwen2.5-3B-Instruct-AWQ
|
||||||
|
|
||||||
|
# Hugging Face Token (optional, only for gated models)
|
||||||
|
HF_TOKEN=""
|
||||||
|
|
||||||
|
# vLLM inference parameters
|
||||||
|
MAX_MODEL_LEN=4096
|
||||||
|
GPU_MEMORY_UTILIZATION=0.90
|
||||||
|
MAX_NUM_SEQS=64
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
# Model weights cache (multi-GB data)
|
||||||
|
models_cache/
|
||||||
|
*.bin
|
||||||
|
*.safetensors
|
||||||
|
|
||||||
|
# Local environment file
|
||||||
|
.env
|
||||||
|
|
||||||
|
# Docker volumes
|
||||||
|
*-data/
|
||||||
|
|
||||||
|
# Python cache & virtualenv
|
||||||
|
__pycache__/
|
||||||
|
*.py[cod]
|
||||||
|
.venv/
|
||||||
|
venv/
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
# vLLM Production Stack: Concurrency, Resilience & Telemetry
|
||||||
|
|
||||||
|
Production-grade LLM inference deployment featuring **vLLM** (PagedAttention & Continuous Batching), **LiteLLM Gateway** (circuit breakers & fallbacks), and full observability via **Prometheus** and **Grafana**.
|
||||||
|
|
||||||
|
Read the full technical breakdown on my blog: [De Ollama a Producción: Desplegando vLLM con PagedAttention y métricas en tiempo real](https://blog.jrodriiguezg.link)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
- **Engine:** [vLLM](https://github.com/vllm-project/vllm) with AWQ quantization (`Qwen/Qwen2.5-3B-Instruct-AWQ`), PagedAttention, and Continuous Batching.
|
||||||
|
- **Gateway & Resilience:** [LiteLLM Proxy](https://github.com/BerriAI/litellm) providing OpenAI-compatible routing, timeouts, and silent failover.
|
||||||
|
- **Metrics Scraper:** [Prometheus](https://prometheus.io/) scraping the native `/metrics` endpoint every 2s.
|
||||||
|
- **Dashboards:** [Grafana](https://grafana.com/) for real-time visualization of TTFT, TPOT, and KV Cache utilization.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Quick Start
|
||||||
|
|
||||||
|
### 1. Prerequisites
|
||||||
|
- Linux OS (Fedora / RHEL / Debian)
|
||||||
|
- NVIDIA GPU with proprietary drivers (`nvidia-smi`)
|
||||||
|
- Docker Engine & NVIDIA Container Toolkit (`nvidia-ctk`)
|
||||||
|
|
||||||
|
### 2. Configuration
|
||||||
|
Copy the sample environment file:
|
||||||
|
```bash
|
||||||
|
cp .env.example .env
|
||||||
|
```
|
||||||
|
|
||||||
|
### 3. Launch the Stack
|
||||||
|
```bash
|
||||||
|
docker compose up -d
|
||||||
|
```
|
||||||
|
|
||||||
|
Check logs and health status:
|
||||||
|
```bash
|
||||||
|
docker compose logs -f vllm
|
||||||
|
curl http://localhost:8000/health
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Load & Concurrency Benchmark
|
||||||
|
|
||||||
|
Stress-test the deployment with the included asynchronous Python benchmark:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install -r scripts/requirements.txt
|
||||||
|
|
||||||
|
# Run 20 concurrent requests against vLLM
|
||||||
|
python3 scripts/benchmark.py --concurrency 20 --url http://localhost:4000/v1/chat/completions --model production-model
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Service Endpoints
|
||||||
|
|
||||||
|
| Service | Port | Description |
|
||||||
|
| :--- | :--- | :--- |
|
||||||
|
| **LiteLLM Gateway** | `http://localhost:4000` | OpenAI-compatible endpoint with circuit breaker |
|
||||||
|
| **vLLM Engine** | `http://localhost:8000` | Raw inference API & `/metrics` |
|
||||||
|
| **Prometheus** | `http://localhost:9090` | Telemetry scraper & PromQL console |
|
||||||
|
| **Grafana** | `http://localhost:3000` | Dashboards (`admin` / `admin`) |
|
||||||
@@ -0,0 +1,101 @@
|
|||||||
|
version: '3.8'
|
||||||
|
|
||||||
|
services:
|
||||||
|
# 1. MOTOR DE INFERENCIA DE PRODUCCIÓN (vLLM)
|
||||||
|
vllm:
|
||||||
|
image: vllm/vllm-openai:latest
|
||||||
|
container_name: vllm-engine
|
||||||
|
restart: unless-stopped
|
||||||
|
environment:
|
||||||
|
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN}
|
||||||
|
- VLLM_LOGGING_LEVEL=INFO
|
||||||
|
volumes:
|
||||||
|
- ./models_cache:/root/.cache/huggingface:Z # :Z para compatibilidad SELinux en RHEL/Fedora
|
||||||
|
ports:
|
||||||
|
- "8000:8000"
|
||||||
|
command: >
|
||||||
|
--model ${MODEL_ID}
|
||||||
|
--quantization awq
|
||||||
|
--dtype half
|
||||||
|
--gpu-memory-utilization ${GPU_MEMORY_UTILIZATION}
|
||||||
|
--max-model-len ${MAX_MODEL_LEN}
|
||||||
|
--max-num-seqs ${MAX_NUM_SEQS}
|
||||||
|
--block-size 16
|
||||||
|
--port 8000
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
reservations:
|
||||||
|
devices:
|
||||||
|
- driver: nvidia
|
||||||
|
count: all
|
||||||
|
capabilities: [gpu]
|
||||||
|
networks:
|
||||||
|
- vllm-net
|
||||||
|
healthcheck:
|
||||||
|
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
|
||||||
|
interval: 10s
|
||||||
|
timeout: 5s
|
||||||
|
retries: 10
|
||||||
|
start_period: 60s
|
||||||
|
|
||||||
|
# 2. PROXY DE RED, RESILIENCIA Y CIRCUIT BREAKER (LiteLLM)
|
||||||
|
litellm:
|
||||||
|
image: ghcr.io/berriai/litellm:main-latest
|
||||||
|
container_name: litellm-gateway
|
||||||
|
restart: unless-stopped
|
||||||
|
volumes:
|
||||||
|
- ./litellm/config.yaml:/app/config.yaml:ro,Z
|
||||||
|
ports:
|
||||||
|
- "4000:4000"
|
||||||
|
command: ["--config", "/app/config.yaml", "--port", "4000"]
|
||||||
|
depends_on:
|
||||||
|
vllm:
|
||||||
|
condition: service_healthy
|
||||||
|
networks:
|
||||||
|
- vllm-net
|
||||||
|
|
||||||
|
# 3. OBSERVABILIDAD: RECOLECTOR DE TELEMETRÍA (Prometheus)
|
||||||
|
prometheus:
|
||||||
|
image: prom/prometheus:latest
|
||||||
|
container_name: prometheus-telemetry
|
||||||
|
restart: unless-stopped
|
||||||
|
volumes:
|
||||||
|
- ./prometheus/prometheus.yml:/etc/prometheus/prometheus.yml:ro,Z
|
||||||
|
- prometheus-data:/prometheus:Z
|
||||||
|
ports:
|
||||||
|
- "9090:9090"
|
||||||
|
command:
|
||||||
|
- '--config.file=/etc/prometheus/prometheus.yml'
|
||||||
|
- '--storage.tsdb.path=/prometheus'
|
||||||
|
- '--web.console.libraries=/usr/share/prometheus/console_libraries'
|
||||||
|
- '--web.console.templates=/usr/share/prometheus/consoles'
|
||||||
|
networks:
|
||||||
|
- vllm-net
|
||||||
|
|
||||||
|
# 4. OBSERVABILIDAD: VISUALIZACIÓN EN TIEMPO REAL (Grafana)
|
||||||
|
grafana:
|
||||||
|
image: grafana/grafana:latest
|
||||||
|
container_name: grafana-dashboard
|
||||||
|
restart: unless-stopped
|
||||||
|
environment:
|
||||||
|
- GF_SECURITY_ADMIN_USER=admin
|
||||||
|
- GF_SECURITY_ADMIN_PASSWORD=admin
|
||||||
|
- GF_USERS_ALLOW_SIGN_UP=false
|
||||||
|
volumes:
|
||||||
|
- ./grafana/provisioning:/etc/grafana/provisioning:ro,Z
|
||||||
|
- grafana-data:/var/lib/grafana:Z
|
||||||
|
ports:
|
||||||
|
- "3000:3000"
|
||||||
|
depends_on:
|
||||||
|
- prometheus
|
||||||
|
networks:
|
||||||
|
- vllm-net
|
||||||
|
|
||||||
|
networks:
|
||||||
|
vllm-net:
|
||||||
|
driver: bridge
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
prometheus-data:
|
||||||
|
grafana-data:
|
||||||
|
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
apiVersion: 1
|
||||||
|
|
||||||
|
datasources:
|
||||||
|
- name: Prometheus
|
||||||
|
type: prometheus
|
||||||
|
access: proxy
|
||||||
|
url: http://prometheus:9090
|
||||||
|
isDefault: true
|
||||||
|
editable: false
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
model_list:
|
||||||
|
# Modelo primario apuntando a nuestro motor vLLM local
|
||||||
|
- model_name: production-model
|
||||||
|
litellm_params:
|
||||||
|
model: openai/Qwen/Qwen2.5-3B-Instruct-AWQ
|
||||||
|
api_base: http://vllm:8000/v1
|
||||||
|
api_key: "token-local-vllm"
|
||||||
|
request_timeout: 30 # Timeout para evitar colgar al cliente
|
||||||
|
|
||||||
|
# Modelo de fallback (contingencia) en caso de saturación o caída del primario
|
||||||
|
# Puede ser un modelo secundario local o un proveedor externo
|
||||||
|
# - model_name: fallback-backup
|
||||||
|
# litellm_params:
|
||||||
|
# model: openai/gpt-4o-mini # O una segunda instancia vLLM con modelo ligero
|
||||||
|
# api_key: "dummy-key-o-real"
|
||||||
|
|
||||||
|
router_settings:
|
||||||
|
routing_strategy: "latency-based-routing"
|
||||||
|
timeout: 30
|
||||||
|
fallbacks:
|
||||||
|
- "production-model": ["fallback-backup"]
|
||||||
|
num_retries: 2
|
||||||
|
allowed_fails: 3
|
||||||
|
cooldown_time: 15 # Segundos antes de reintentar el modelo primario tras abrir circuito
|
||||||
|
|
||||||
|
general_settings:
|
||||||
|
master_key: "sk-production-admin-key"
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
global:
|
||||||
|
scrape_interval: 2s # Intervalo corto para capturar datos de latencia y saturación
|
||||||
|
evaluation_interval: 2s
|
||||||
|
|
||||||
|
scrape_configs:
|
||||||
|
- job_name: 'vllm'
|
||||||
|
metrics_path: '/metrics'
|
||||||
|
static_configs:
|
||||||
|
- targets: ['vllm:8000']
|
||||||
|
labels:
|
||||||
|
engine: 'vllm'
|
||||||
|
model: 'qwen2.5-3b-awq'
|
||||||
|
|
||||||
|
- job_name: 'litellm'
|
||||||
|
metrics_path: '/metrics'
|
||||||
|
static_configs:
|
||||||
|
- targets: ['litellm:4000']
|
||||||
|
labels:
|
||||||
|
service: 'llm-gateway'
|
||||||
@@ -0,0 +1,84 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Benchmark de Concurrencia para Servidores de Inferencia LLM
|
||||||
|
Compara comportamiento ante ráfagas concurrentes de peticiones.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import time
|
||||||
|
import aiohttp
|
||||||
|
import statistics
|
||||||
|
import argparse
|
||||||
|
|
||||||
|
PROMPT = "Explica en tres párrafos técnicos qué es la memoria virtual y cómo se gestionan las páginas de memoria en el kernel de Linux."
|
||||||
|
|
||||||
|
async def send_request(session, url, model, headers, req_id):
|
||||||
|
# Envia una peticion individual y mide su latencia
|
||||||
|
payload = {
|
||||||
|
"model": model,
|
||||||
|
"messages": [{"role": "user", "content": PROMPT}],
|
||||||
|
"max_tokens": 150,
|
||||||
|
"temperature": 0.7
|
||||||
|
}
|
||||||
|
|
||||||
|
start_time = time.perf_counter()
|
||||||
|
try:
|
||||||
|
async with session.post(url, json=payload, headers=headers, timeout=aiohttp.ClientTimeout(total=60)) as resp:
|
||||||
|
data = await resp.json()
|
||||||
|
latency = time.perf_counter() - start_time
|
||||||
|
if resp.status == 200:
|
||||||
|
tokens = data["usage"]["completion_tokens"]
|
||||||
|
return {"id": req_id, "success": True, "latency": latency, "tokens": tokens}
|
||||||
|
else:
|
||||||
|
return {"id": req_id, "success": False, "latency": latency, "error": resp.status}
|
||||||
|
except Exception as e:
|
||||||
|
latency = time.perf_counter() - start_time
|
||||||
|
return {"id": req_id, "success": False, "latency": latency, "error": str(e)}
|
||||||
|
|
||||||
|
async def run_benchmark(url, model, concurrency, auth_header):
|
||||||
|
# Ejecuta peticiones concurrentes y calcula metricas
|
||||||
|
headers = {"Content-Type": "application/json"}
|
||||||
|
if auth_header:
|
||||||
|
headers["Authorization"] = f"Bearer {auth_header}"
|
||||||
|
|
||||||
|
print("\n=======================================================")
|
||||||
|
print(f"Iniciando Benchmark: {concurrency} peticiones CONCURRENTES")
|
||||||
|
print(f"Target: {url} | Modelo: {model}")
|
||||||
|
print("=======================================================")
|
||||||
|
|
||||||
|
async with aiohttp.ClientSession() as session:
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
tasks = [send_request(session, url, model, headers, i) for i in range(concurrency)]
|
||||||
|
results = await asyncio.gather(*tasks)
|
||||||
|
total_wall_time = time.perf_counter() - t0
|
||||||
|
|
||||||
|
successful = [r for r in results if r["success"]]
|
||||||
|
failed = [r for r in results if not r["success"]]
|
||||||
|
|
||||||
|
if successful:
|
||||||
|
latencies = [r["latency"] for r in successful]
|
||||||
|
total_tokens = sum(r["tokens"] for r in successful)
|
||||||
|
avg_latency = statistics.mean(latencies)
|
||||||
|
p95_latency = statistics.quantiles(latencies, n=20)[18] if len(latencies) >= 20 else max(latencies)
|
||||||
|
throughput_tokens_sec = total_tokens / total_wall_time
|
||||||
|
|
||||||
|
print("\nRESULTADOS:")
|
||||||
|
print(f" - Peticiones exitosas: {len(successful)}/{concurrency}")
|
||||||
|
print(f" - Fallidas / Timeout: {len(failed)}")
|
||||||
|
print(f" - Tiempo total del test: {total_wall_time:.2f} s")
|
||||||
|
print(f" - Throughput global: {throughput_tokens_sec:.2f} tokens/segundo")
|
||||||
|
print(f" - Latencia promedio: {avg_latency:.2f} s")
|
||||||
|
print(f" - Latencia P95: {p95_latency:.2f} s")
|
||||||
|
else:
|
||||||
|
print(f"\nTodas las peticiones fallaron. Errores: {[r.get('error') for r in failed]}")
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
# Parser de argumentos por linea de comandos
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--url", default="http://localhost:4000/v1/chat/completions", help="Endpoint OpenAI-compatible")
|
||||||
|
parser.add_argument("--model", default="production-model", help="Nombre del modelo")
|
||||||
|
parser.add_argument("--concurrency", type=int, default=20, help="Numero de peticiones concurrentes")
|
||||||
|
parser.add_argument("--key", default="sk-production-admin-key", help="API Key si aplica")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
asyncio.run(run_benchmark(args.url, args.model, args.concurrency, args.key))
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
aiohttp
|
||||||
Reference in New Issue
Block a user