Files
qwen3-6-lora/docker-compose.eval.yml
T
aleleba d1e9892911 Fase 5: servicio de diagnostico vllm-eval-nvfp4-nospec (puerto 8003)
Identico a vllm-eval-nvfp4 pero sin --speculative-config, para aislar si la
regresion de calidad observada en las puertas 2-3 (vs. Fase 4) viene del
speculative decoding (MTP) o de la cuantizacion NVFP4 en si -- decision
explicita del usuario antes de invertir tiempo en recalibrar con mas
muestras de calibracion.
2026-07-30 01:10:36 +00:00

136 lines
5.1 KiB
YAML

services:
vllm-eval:
image: vllm/vllm-openai:cu130-nightly-aarch64
container_name: vllm-eval
restart: "no"
ipc: host
ports:
- "8001:8000"
volumes:
- /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-bf16:/model:ro
command:
- "--model=/model"
- "--served-model-name=qwen3.6-35b-a3b-mcp-bf16"
- "--tensor-parallel-size=1"
- "--max-model-len=32768"
- "--enable-auto-tool-choice"
- "--tool-call-parser=qwen3_coder"
- "--reasoning-parser=qwen3"
- "--default-chat-template-kwargs={\"preserve_thinking\": true}"
- "--trust-remote-code"
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: 1
capabilities: [gpu]
# Fase 5: clona 1:1 el docker-compose.yml real de produccion (vllm-qwen36, ver
# PLAN.md) -- mismos flags de vLLM (incluido --speculative-config real, la
# primera vez que se prueba en este proyecto), cambiando solo container_name,
# puerto (8002, produccion usa 8000 y el vllm-eval de Fase 4 usa 8001), volumen
# (checkpoint NVFP4 de Fase 5 en vez de RedHatAI--Qwen3.6-35B-A3B-NVFP4),
# --model/--served-model-name, y restart: "no". Nunca se toca vllm-qwen36 ni su
# compose real de Portainer -- este es un contenedor nuevo y propio del repo.
vllm-eval-nvfp4:
image: vllm/vllm-openai:cu130-nightly-aarch64
container_name: vllm-eval-nvfp4
restart: "no"
runtime: nvidia
environment:
NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
ports:
- "8002:8000"
ipc: host
ulimits:
memlock: -1
stack: 67108864
volumes:
- /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-NVFP4:/models/Qwen3.6-35B-A3B-mcp-NVFP4:ro
command: >
--model /models/Qwen3.6-35B-A3B-mcp-NVFP4
--served-model-name qwen3.6-35b-a3b-mcp-nvfp4
--host 0.0.0.0
--port 8000
--tensor-parallel-size 1
--trust-remote-code
--quantization compressed-tensors
--moe-backend flashinfer_cutlass
--kv-cache-dtype fp8_e4m3
--gpu-memory-utilization 0.45
--max-model-len 524288
--max-num-seqs 8
--max-num-batched-tokens 32768
--enable-chunked-prefill
--enable-prefix-caching
--speculative-config '{"method":"mtp","num_speculative_tokens":1}'
--reasoning-parser qwen3
--tool-call-parser qwen3_coder
--enable-auto-tool-choice
--default-chat-template-kwargs '{"preserve_thinking":true}'
--limit-mm-per-prompt '{"image":4,"video":0,"audio":0}'
--generation-config vllm
--override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}'
--hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}'
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
interval: 30s
timeout: 10s
retries: 5
start_period: 600s
# Fase 5 -- diagnostico de aislamiento: identico a vllm-eval-nvfp4 pero SIN
# --speculative-config, para determinar si la regresion de calidad observada
# en las puertas 2-3 (vs. Fase 4) viene del speculative decoding (MTP) o de
# la cuantizacion NVFP4 en si. Puerto 8003 (distinto de 8000 produccion, 8001
# Fase 4, 8002 Fase 5 con speculative real). Servicio temporal de diagnostico,
# no clona produccion 1:1 a proposito (esa es la variable que se esta aislando).
vllm-eval-nvfp4-nospec:
image: vllm/vllm-openai:cu130-nightly-aarch64
container_name: vllm-eval-nvfp4-nospec
restart: "no"
runtime: nvidia
environment:
NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
ports:
- "8003:8000"
ipc: host
ulimits:
memlock: -1
stack: 67108864
volumes:
- /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-NVFP4:/models/Qwen3.6-35B-A3B-mcp-NVFP4:ro
command: >
--model /models/Qwen3.6-35B-A3B-mcp-NVFP4
--served-model-name qwen3.6-35b-a3b-mcp-nvfp4-nospec
--host 0.0.0.0
--port 8000
--tensor-parallel-size 1
--trust-remote-code
--quantization compressed-tensors
--moe-backend flashinfer_cutlass
--kv-cache-dtype fp8_e4m3
--gpu-memory-utilization 0.45
--max-model-len 524288
--max-num-seqs 8
--max-num-batched-tokens 32768
--enable-chunked-prefill
--enable-prefix-caching
--reasoning-parser qwen3
--tool-call-parser qwen3_coder
--enable-auto-tool-choice
--default-chat-template-kwargs '{"preserve_thinking":true}'
--limit-mm-per-prompt '{"image":4,"video":0,"audio":0}'
--generation-config vllm
--override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}'
--hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}'
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
interval: 30s
timeout: 10s
retries: 5
start_period: 600s