services: vllm-eval: image: vllm/vllm-openai:cu130-nightly-aarch64 container_name: vllm-eval restart: "no" ipc: host ports: - "8001:8000" volumes: - /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-bf16:/model:ro command: - "--model=/model" - "--served-model-name=qwen3.6-35b-a3b-mcp-bf16" - "--tensor-parallel-size=1" - "--max-model-len=32768" - "--enable-auto-tool-choice" - "--tool-call-parser=qwen3_coder" - "--reasoning-parser=qwen3" - "--default-chat-template-kwargs={\"preserve_thinking\": true}" - "--trust-remote-code" deploy: resources: reservations: devices: - driver: nvidia count: 1 capabilities: [gpu] # Fase 5: clona 1:1 el docker-compose.yml real de produccion (vllm-qwen36, ver # PLAN.md) -- mismos flags de vLLM (incluido --speculative-config real, la # primera vez que se prueba en este proyecto), cambiando solo container_name, # puerto (8002, produccion usa 8000 y el vllm-eval de Fase 4 usa 8001), volumen # (checkpoint NVFP4 de Fase 5 en vez de RedHatAI--Qwen3.6-35B-A3B-NVFP4), # --model/--served-model-name, y restart: "no". Nunca se toca vllm-qwen36 ni su # compose real de Portainer -- este es un contenedor nuevo y propio del repo. vllm-eval-nvfp4: image: vllm/vllm-openai:cu130-nightly-aarch64 container_name: vllm-eval-nvfp4 restart: "no" runtime: nvidia environment: NVIDIA_VISIBLE_DEVICES: all NVIDIA_DRIVER_CAPABILITIES: compute,utility ports: - "8002:8000" ipc: host ulimits: memlock: -1 stack: 67108864 volumes: - /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-NVFP4:/models/Qwen3.6-35B-A3B-mcp-NVFP4:ro command: > --model /models/Qwen3.6-35B-A3B-mcp-NVFP4 --served-model-name qwen3.6-35b-a3b-mcp-nvfp4 --host 0.0.0.0 --port 8000 --tensor-parallel-size 1 --trust-remote-code --quantization compressed-tensors --moe-backend flashinfer_cutlass --kv-cache-dtype fp8_e4m3 --gpu-memory-utilization 0.45 --max-model-len 524288 --max-num-seqs 8 --max-num-batched-tokens 32768 --enable-chunked-prefill --enable-prefix-caching --speculative-config '{"method":"mtp","num_speculative_tokens":1}' --reasoning-parser qwen3 --tool-call-parser qwen3_coder --enable-auto-tool-choice --default-chat-template-kwargs '{"preserve_thinking":true}' --limit-mm-per-prompt '{"image":4,"video":0,"audio":0}' --generation-config vllm --override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}' --hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}' healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 5 start_period: 600s