Global weighted loss 0.2725 (vs baseline 0.2750, previous v2-bf16 was 0.2734). All 6/6 non-penpot buckets pass the 0.10 regression threshold, including delegacion_subagentes which had narrowly failed on v2-bf16 (+0.1019 -> now +0.0882). The 13 corrective seeds didn't hurt forgetting. Also points vllm-eval at the new v2b-bf16 checkpoint for the gate 2/5 re-measurement.
254 lines
10 KiB
YAML
254 lines
10 KiB
YAML
services:
|
|
vllm-eval:
|
|
image: vllm/vllm-openai:cu130-nightly-aarch64
|
|
container_name: vllm-eval
|
|
restart: "no"
|
|
ipc: host
|
|
ports:
|
|
- "8001:8000"
|
|
volumes:
|
|
# Fase 6.4.25b: apunta al candidato v2b-bf16 (base + LoRA #1 + LoRA #2
|
|
# corregido con los 13 seeds del diagnostico de causa raiz), no al v2-bf16
|
|
# original (que fallo la puerta 5 modo full por el bug de formato de
|
|
# tool-call). Es el checkpoint contra el que se remide la puerta 5 tras la
|
|
# iteracion correctiva, antes de decidir NVFP4 vs fallback.
|
|
- /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-v2b-bf16:/model:ro
|
|
command:
|
|
- "--model=/model"
|
|
- "--served-model-name=qwen3.6-35b-a3b-mcp-v2b-bf16"
|
|
- "--tensor-parallel-size=1"
|
|
- "--max-model-len=32768"
|
|
- "--enable-auto-tool-choice"
|
|
- "--tool-call-parser=qwen3_coder"
|
|
- "--reasoning-parser=qwen3"
|
|
- "--default-chat-template-kwargs={\"preserve_thinking\": true}"
|
|
- "--trust-remote-code"
|
|
deploy:
|
|
resources:
|
|
reservations:
|
|
devices:
|
|
- driver: nvidia
|
|
count: 1
|
|
capabilities: [gpu]
|
|
|
|
# Fase 5: clona 1:1 el docker-compose.yml real de produccion (vllm-qwen36, ver
|
|
# PLAN.md) -- mismos flags de vLLM (incluido --speculative-config real, la
|
|
# primera vez que se prueba en este proyecto), cambiando solo container_name,
|
|
# puerto (8002, produccion usa 8000 y el vllm-eval de Fase 4 usa 8001), volumen
|
|
# (checkpoint NVFP4 de Fase 5 en vez de RedHatAI--Qwen3.6-35B-A3B-NVFP4),
|
|
# --model/--served-model-name, y restart: "no". Nunca se toca vllm-qwen36 ni su
|
|
# compose real de Portainer -- este es un contenedor nuevo y propio del repo.
|
|
vllm-eval-nvfp4:
|
|
image: vllm/vllm-openai:cu130-nightly-aarch64
|
|
container_name: vllm-eval-nvfp4
|
|
restart: "no"
|
|
runtime: nvidia
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: all
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
ports:
|
|
- "8002:8000"
|
|
ipc: host
|
|
ulimits:
|
|
memlock: -1
|
|
stack: 67108864
|
|
volumes:
|
|
- /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-NVFP4:/models/Qwen3.6-35B-A3B-mcp-NVFP4:ro
|
|
command: >
|
|
--model /models/Qwen3.6-35B-A3B-mcp-NVFP4
|
|
--served-model-name qwen3.6-35b-a3b-mcp-nvfp4
|
|
--host 0.0.0.0
|
|
--port 8000
|
|
--tensor-parallel-size 1
|
|
--trust-remote-code
|
|
--quantization compressed-tensors
|
|
--moe-backend flashinfer_cutlass
|
|
--kv-cache-dtype fp8_e4m3
|
|
--gpu-memory-utilization 0.45
|
|
--max-model-len 524288
|
|
--max-num-seqs 8
|
|
--max-num-batched-tokens 32768
|
|
--enable-chunked-prefill
|
|
--enable-prefix-caching
|
|
--speculative-config '{"method":"mtp","num_speculative_tokens":1}'
|
|
--reasoning-parser qwen3
|
|
--tool-call-parser qwen3_coder
|
|
--enable-auto-tool-choice
|
|
--default-chat-template-kwargs '{"preserve_thinking":true}'
|
|
--limit-mm-per-prompt '{"image":4,"video":0,"audio":0}'
|
|
--generation-config vllm
|
|
--override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}'
|
|
--hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}'
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 5
|
|
start_period: 600s
|
|
|
|
# Fase 6 -- candidato v2 (base + LoRA #1 + LoRA #2 de diseno en Penpot, mergeado
|
|
# y recuantizado a NVFP4). Clon 1:1 de vllm-eval-nvfp4, que a su vez clona el
|
|
# compose real de produccion, incluido --speculative-config: es el contenedor
|
|
# contra el que se corren las puertas 2/3/4/5 del candidato, y solo sirve si
|
|
# replica exactamente los flags con los que se va a servir. Respecto de
|
|
# vllm-eval-nvfp4 cambian SOLO container_name, puerto (8004; 8000 produccion,
|
|
# 8001 Fase 4, 8002/8003 Fase 5), volumen, --model y --served-model-name.
|
|
# Nunca se toca vllm-qwen36 ni su compose real de Portainer.
|
|
vllm-eval-nvfp4-v2:
|
|
image: vllm/vllm-openai:cu130-nightly-aarch64
|
|
container_name: vllm-eval-nvfp4-v2
|
|
restart: "no"
|
|
runtime: nvidia
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: all
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
ports:
|
|
- "8004:8000"
|
|
ipc: host
|
|
ulimits:
|
|
memlock: -1
|
|
stack: 67108864
|
|
volumes:
|
|
- /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-v2-NVFP4:/models/Qwen3.6-35B-A3B-mcp-v2-NVFP4:ro
|
|
command: >
|
|
--model /models/Qwen3.6-35B-A3B-mcp-v2-NVFP4
|
|
--served-model-name qwen3.6-35b-a3b-mcp-v2-nvfp4
|
|
--host 0.0.0.0
|
|
--port 8000
|
|
--tensor-parallel-size 1
|
|
--trust-remote-code
|
|
--quantization compressed-tensors
|
|
--moe-backend flashinfer_cutlass
|
|
--kv-cache-dtype fp8_e4m3
|
|
--gpu-memory-utilization 0.45
|
|
--max-model-len 524288
|
|
--max-num-seqs 8
|
|
--max-num-batched-tokens 32768
|
|
--enable-chunked-prefill
|
|
--enable-prefix-caching
|
|
--speculative-config '{"method":"mtp","num_speculative_tokens":1}'
|
|
--reasoning-parser qwen3
|
|
--tool-call-parser qwen3_coder
|
|
--enable-auto-tool-choice
|
|
--default-chat-template-kwargs '{"preserve_thinking":true}'
|
|
--limit-mm-per-prompt '{"image":4,"video":0,"audio":0}'
|
|
--generation-config vllm
|
|
--override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}'
|
|
--hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}'
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 5
|
|
start_period: 600s
|
|
|
|
# Fase 6 -- diagnostico de aislamiento del drift del head de MTP (riesgo #10):
|
|
# identico a vllm-eval-nvfp4-v2 pero SIN --speculative-config. El draft head de
|
|
# MTP se copia del linaje base y nunca se fine-tunea, mientras que el target
|
|
# model ya derivo DOS veces (LoRA #1 y LoRA #2), asi que la tasa de aceptacion
|
|
# del speculative decoding puede caer y degradar la calidad servida sin que la
|
|
# cuantizacion ni el dataset tengan nada que ver. Correr las mismas puertas en
|
|
# 8004 (spec) y 8005 (nospec) separa las dos causas. Es un servicio de
|
|
# diagnostico: no clona produccion 1:1 a proposito (esa es justo la variable
|
|
# que se esta aislando) y los flags de produccion NUNCA se cambian por esto,
|
|
# solo se reporta el hallazgo.
|
|
vllm-eval-nvfp4-v2-nospec:
|
|
image: vllm/vllm-openai:cu130-nightly-aarch64
|
|
container_name: vllm-eval-nvfp4-v2-nospec
|
|
restart: "no"
|
|
runtime: nvidia
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: all
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
ports:
|
|
- "8005:8000"
|
|
ipc: host
|
|
ulimits:
|
|
memlock: -1
|
|
stack: 67108864
|
|
volumes:
|
|
- /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-v2-NVFP4:/models/Qwen3.6-35B-A3B-mcp-v2-NVFP4:ro
|
|
command: >
|
|
--model /models/Qwen3.6-35B-A3B-mcp-v2-NVFP4
|
|
--served-model-name qwen3.6-35b-a3b-mcp-v2-nvfp4-nospec
|
|
--host 0.0.0.0
|
|
--port 8000
|
|
--tensor-parallel-size 1
|
|
--trust-remote-code
|
|
--quantization compressed-tensors
|
|
--moe-backend flashinfer_cutlass
|
|
--kv-cache-dtype fp8_e4m3
|
|
--gpu-memory-utilization 0.45
|
|
--max-model-len 524288
|
|
--max-num-seqs 8
|
|
--max-num-batched-tokens 32768
|
|
--enable-chunked-prefill
|
|
--enable-prefix-caching
|
|
--reasoning-parser qwen3
|
|
--tool-call-parser qwen3_coder
|
|
--enable-auto-tool-choice
|
|
--default-chat-template-kwargs '{"preserve_thinking":true}'
|
|
--limit-mm-per-prompt '{"image":4,"video":0,"audio":0}'
|
|
--generation-config vllm
|
|
--override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}'
|
|
--hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}'
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 5
|
|
start_period: 600s
|
|
|
|
# Fase 5 -- diagnostico de aislamiento: identico a vllm-eval-nvfp4 pero SIN
|
|
# --speculative-config, para determinar si la regresion de calidad observada
|
|
# en las puertas 2-3 (vs. Fase 4) viene del speculative decoding (MTP) o de
|
|
# la cuantizacion NVFP4 en si. Puerto 8003 (distinto de 8000 produccion, 8001
|
|
# Fase 4, 8002 Fase 5 con speculative real). Servicio temporal de diagnostico,
|
|
# no clona produccion 1:1 a proposito (esa es la variable que se esta aislando).
|
|
vllm-eval-nvfp4-nospec:
|
|
image: vllm/vllm-openai:cu130-nightly-aarch64
|
|
container_name: vllm-eval-nvfp4-nospec
|
|
restart: "no"
|
|
runtime: nvidia
|
|
environment:
|
|
NVIDIA_VISIBLE_DEVICES: all
|
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
ports:
|
|
- "8003:8000"
|
|
ipc: host
|
|
ulimits:
|
|
memlock: -1
|
|
stack: 67108864
|
|
volumes:
|
|
- /home/aleleba/ft-models/Qwen3.6-35B-A3B-mcp-NVFP4:/models/Qwen3.6-35B-A3B-mcp-NVFP4:ro
|
|
command: >
|
|
--model /models/Qwen3.6-35B-A3B-mcp-NVFP4
|
|
--served-model-name qwen3.6-35b-a3b-mcp-nvfp4-nospec
|
|
--host 0.0.0.0
|
|
--port 8000
|
|
--tensor-parallel-size 1
|
|
--trust-remote-code
|
|
--quantization compressed-tensors
|
|
--moe-backend flashinfer_cutlass
|
|
--kv-cache-dtype fp8_e4m3
|
|
--gpu-memory-utilization 0.45
|
|
--max-model-len 524288
|
|
--max-num-seqs 8
|
|
--max-num-batched-tokens 32768
|
|
--enable-chunked-prefill
|
|
--enable-prefix-caching
|
|
--reasoning-parser qwen3
|
|
--tool-call-parser qwen3_coder
|
|
--enable-auto-tool-choice
|
|
--default-chat-template-kwargs '{"preserve_thinking":true}'
|
|
--limit-mm-per-prompt '{"image":4,"video":0,"audio":0}'
|
|
--generation-config vllm
|
|
--override-generation-config '{"temperature":0.6,"top_p":0.80,"top_k":20,"presence_penalty":0.0,"repetition_penalty":1.0}'
|
|
--hf-overrides '{"text_config":{"rope_scaling":{"rope_type":"yarn","factor":2.0,"original_max_position_embeddings":262144}}}'
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 5
|
|
start_period: 600s
|