services: vllm-eval: image: vllm/vllm-openai:cu130-nightly-aarch64 container_name: vllm-eval restart: "no" ipc: host ports: - "8001:8000" volumes: # Fase 6.4.25b: apunta al candidato v2b-bf16 (base + LoRA #1 + LoRA #2 # corregido con los 13 seeds del diagnostico de causa raiz), no al v2-bf16 # original (que fallo la puerta 5 modo full por el bug de formato de # tool-call). Es el checkpoint contra el que se remide la puerta 5 tras la # iteracion correctiva, antes de decidir NVFP4 vs fallback. - /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-v2b-bf16:/model:ro command: - "--model=/model" - "--served-model-name=qwen3.6-35b-a3b-mcp-v2b-bf16" - "--tensor-parallel-size=1" - "--max-model-len=32768" - "--enable-auto-tool-choice" - "--tool-call-parser=qwen3_coder" - "--reasoning-parser=qwen3" - "--default-chat-template-kwargs={\"preserve_thinking\": true}" - "--trust-remote-code" deploy: resources: reservations: devices: - driver: nvidia count: 1 capabilities: [gpu] # Fase 5: clona 1:1 el docker-compose.yml real de produccion (vllm-qwen36, ver # PLAN.md) -- mismos flags de vLLM (incluido --speculative-config real, la # primera vez que se prueba en este proyecto), cambiando solo container_name, # puerto (8002, produccion usa 8000 y el vllm-eval de Fase 4 usa 8001), volumen # (checkpoint NVFP4 de Fase 5 en vez de RedHatAI--Qwen3.6-35B-A3B-NVFP4), # --model/--served-model-name, y restart: "no". Nunca se toca vllm-qwen36 ni su # compose real de Portainer -- este es un contenedor nuevo y propio del repo. vllm-eval-nvfp4: image: vllm/vllm-openai:cu130-nightly-aarch64 container_name: vllm-eval-nvfp4 restart: "no" runtime: nvidia environment: NVIDIA_VISIBLE_DEVICES: all NVIDIA_DRIVER_CAPABILITIES: compute,utility ports: - "8002:8000" ipc: host ulimits: memlock: -1 stack: 67108864 volumes: - /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-NVFP4:/models/Qwen3.6-35B-A3B-mcp-NVFP4:ro command: > --model /models/Qwen3.6-35B-A3B-mcp-NVFP4 --served-model-name qwen3.6-35b-a3b-mcp-nvfp4 --host 0.0.0.0 --port 8000 --tensor-parallel-size 1 --trust-remote-code --quantization compressed-tensors --moe-backend flashinfer_cutlass --kv-cache-dtype fp8_e4m3 --gpu-memory-utilization 0.45 --max-model-len 524288 --max-num-seqs 8 --max-num-batched-tokens 32768 --enable-chunked-prefill --enable-prefix-caching --speculative-config '{"method":"mtp","num_speculative_tokens":1}' --reasoning-parser qwen3 --tool-call-parser qwen3_coder --enable-auto-tool-choice --default-chat-template-kwargs '{"preserve_thinking":true}' --limit-mm-per-prompt '{"image":4,"video":0,"audio":0}' --generation-config vllm --override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}' --hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}' healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 5 start_period: 600s # Fase 6 -- candidato v2b (base + LoRA #1 + LoRA #2 de diseno en Penpot, # iteracion correctiva 6.4.25b/c, mergeado y recuantizado a NVFP4). Clon 1:1 de # vllm-eval-nvfp4, que a su vez clona el compose real de produccion, incluido # --speculative-config: es el contenedor contra el que se corren las puertas # 2/3/4/5 del candidato, y solo sirve si replica exactamente los flags con los # que se va a servir. Respecto de vllm-eval-nvfp4 cambian SOLO container_name, # puerto (8004; 8000 produccion, 8001 Fase 4, 8002/8003 Fase 5), volumen, # --model y --served-model-name. v2-NVFP4 (sin la "b") se salteo por completo: # v2-bf16 disparo el fallback y la iteracion correctiva 6.4.25b/c produjo # v2b-bf16 directamente, que es lo que se cuantiza aca. Nunca se toca # vllm-qwen36 ni su compose real de Portainer. vllm-eval-nvfp4-v2: image: vllm/vllm-openai:cu130-nightly-aarch64 container_name: vllm-eval-nvfp4-v2 restart: "no" runtime: nvidia environment: NVIDIA_VISIBLE_DEVICES: all NVIDIA_DRIVER_CAPABILITIES: compute,utility ports: - "8004:8000" ipc: host ulimits: memlock: -1 stack: 67108864 volumes: - /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-v2b-NVFP4:/models/Qwen3.6-35B-A3B-mcp-v2b-NVFP4:ro command: > --model /models/Qwen3.6-35B-A3B-mcp-v2b-NVFP4 --served-model-name qwen3.6-35b-a3b-mcp-v2b-nvfp4 --host 0.0.0.0 --port 8000 --tensor-parallel-size 1 --trust-remote-code --quantization compressed-tensors --moe-backend flashinfer_cutlass --kv-cache-dtype fp8_e4m3 --gpu-memory-utilization 0.45 --max-model-len 524288 --max-num-seqs 8 --max-num-batched-tokens 32768 --enable-chunked-prefill --enable-prefix-caching --speculative-config '{"method":"mtp","num_speculative_tokens":1}' --reasoning-parser qwen3 --tool-call-parser qwen3_coder --enable-auto-tool-choice --default-chat-template-kwargs '{"preserve_thinking":true}' --limit-mm-per-prompt '{"image":4,"video":0,"audio":0}' --generation-config vllm --override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}' --hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}' healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 5 start_period: 600s # Fase 6 -- diagnostico de aislamiento del drift del head de MTP (riesgo #10): # identico a vllm-eval-nvfp4-v2 pero SIN --speculative-config. El draft head de # MTP se copia del linaje base y nunca se fine-tunea, mientras que el target # model ya derivo DOS veces (LoRA #1 y LoRA #2), asi que la tasa de aceptacion # del speculative decoding puede caer y degradar la calidad servida sin que la # cuantizacion ni el dataset tengan nada que ver. Correr las mismas puertas en # 8004 (spec) y 8005 (nospec) separa las dos causas. Es un servicio de # diagnostico: no clona produccion 1:1 a proposito (esa es justo la variable # que se esta aislando) y los flags de produccion NUNCA se cambian por esto, # solo se reporta el hallazgo. vllm-eval-nvfp4-v2-nospec: image: vllm/vllm-openai:cu130-nightly-aarch64 container_name: vllm-eval-nvfp4-v2-nospec restart: "no" runtime: nvidia environment: NVIDIA_VISIBLE_DEVICES: all NVIDIA_DRIVER_CAPABILITIES: compute,utility ports: - "8005:8000" ipc: host ulimits: memlock: -1 stack: 67108864 volumes: - /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-v2b-NVFP4:/models/Qwen3.6-35B-A3B-mcp-v2b-NVFP4:ro command: > --model /models/Qwen3.6-35B-A3B-mcp-v2b-NVFP4 --served-model-name qwen3.6-35b-a3b-mcp-v2b-nvfp4-nospec --host 0.0.0.0 --port 8000 --tensor-parallel-size 1 --trust-remote-code --quantization compressed-tensors --moe-backend flashinfer_cutlass --kv-cache-dtype fp8_e4m3 --gpu-memory-utilization 0.45 --max-model-len 524288 --max-num-seqs 8 --max-num-batched-tokens 32768 --enable-chunked-prefill --enable-prefix-caching --reasoning-parser qwen3 --tool-call-parser qwen3_coder --enable-auto-tool-choice --default-chat-template-kwargs '{"preserve_thinking":true}' --limit-mm-per-prompt '{"image":4,"video":0,"audio":0}' --generation-config vllm --override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}' --hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}' healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 5 start_period: 600s # Fase 5 -- diagnostico de aislamiento: identico a vllm-eval-nvfp4 pero SIN # --speculative-config, para determinar si la regresion de calidad observada # en las puertas 2-3 (vs. Fase 4) viene del speculative decoding (MTP) o de # la cuantizacion NVFP4 en si. Puerto 8003 (distinto de 8000 produccion, 8001 # Fase 4, 8002 Fase 5 con speculative real). Servicio temporal de diagnostico, # no clona produccion 1:1 a proposito (esa es la variable que se esta aislando). vllm-eval-nvfp4-nospec: image: vllm/vllm-openai:cu130-nightly-aarch64 container_name: vllm-eval-nvfp4-nospec restart: "no" runtime: nvidia environment: NVIDIA_VISIBLE_DEVICES: all NVIDIA_DRIVER_CAPABILITIES: compute,utility ports: - "8003:8000" ipc: host ulimits: memlock: -1 stack: 67108864 volumes: - /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-NVFP4:/models/Qwen3.6-35B-A3B-mcp-NVFP4:ro command: > --model /models/Qwen3.6-35B-A3B-mcp-NVFP4 --served-model-name qwen3.6-35b-a3b-mcp-nvfp4-nospec --host 0.0.0.0 --port 8000 --tensor-parallel-size 1 --trust-remote-code --quantization compressed-tensors --moe-backend flashinfer_cutlass --kv-cache-dtype fp8_e4m3 --gpu-memory-utilization 0.45 --max-model-len 524288 --max-num-seqs 8 --max-num-batched-tokens 32768 --enable-chunked-prefill --enable-prefix-caching --reasoning-parser qwen3 --tool-call-parser qwen3_coder --enable-auto-tool-choice --default-chat-template-kwargs '{"preserve_thinking":true}' --limit-mm-per-prompt '{"image":4,"video":0,"audio":0}' --generation-config vllm --override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}' --hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}' healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 5 start_period: 600s