171 steps (901 examples / 16 grad-accum * 3 epochs, ceil-rounded, not the 168 the plan estimated with floor), trained decoupled on spark against the merged bf16 checkpoint that already has LoRA #1 folded in. No OOM, no aborts: train_runtime 24370s (~6h46m, within the ~7.3h projection), eval_loss 0.6306, peak CUDA 74.58GB, trainable% 0.1220 (matches phase 3 exactly). adapter_config.json confirms the guarded hyperparameters: use_rslora and use_dora both false, lora_bias false, modules_to_save null, r=32, lora_alpha=64 -- the combination 20_merge_lora.py's scaling math depends on. Weights never touched the worktree: OUTPUT_DIR was /workspace/ft-models/lora-adapter-penpot on spark (outside the worktree, root-owned by the training container), copied out via `docker exec cat` piped to a non-root file and verified by matching sha256 (79167dfa...) before landing here. Intermediate checkpoint-*/ directories stay on spark; only the final adapter is versioned, same as phase 3's out/lora-adapter/, which this leaves untouched. .gitignore was missing the negation lines for out/lora-adapter-penpot/ despite already documenting that the final adapter should be committed -- added the same two exceptions that out/lora-adapter/ has.
53 lines
1.2 KiB
JSON
53 lines
1.2 KiB
JSON
{
|
|
"alora_invocation_tokens": null,
|
|
"alpha_pattern": {},
|
|
"arrow_config": null,
|
|
"auto_mapping": null,
|
|
"base_model_name_or_path": "/workspace/ft-models/Qwen3.6-35B-A3B-mcp-bf16",
|
|
"bias": "none",
|
|
"corda_config": null,
|
|
"ensure_weight_tying": false,
|
|
"eva_config": null,
|
|
"exclude_modules": null,
|
|
"fan_in_fan_out": false,
|
|
"inference_mode": true,
|
|
"init_lora_weights": true,
|
|
"layer_replication": null,
|
|
"layers_pattern": null,
|
|
"layers_to_transform": null,
|
|
"loftq_config": {},
|
|
"lora_alpha": 64,
|
|
"lora_bias": false,
|
|
"lora_dropout": 0.05,
|
|
"lora_ga_config": null,
|
|
"megatron_config": null,
|
|
"megatron_core": "megatron.core",
|
|
"modules_to_save": null,
|
|
"peft_type": "LORA",
|
|
"peft_version": "0.19.1",
|
|
"qalora_group_size": 16,
|
|
"r": 32,
|
|
"rank_pattern": {},
|
|
"revision": null,
|
|
"target_modules": [
|
|
"shared_expert.up_proj",
|
|
"in_proj_z",
|
|
"v_proj",
|
|
"shared_expert.down_proj",
|
|
"q_proj",
|
|
"k_proj",
|
|
"in_proj_a",
|
|
"in_proj_b",
|
|
"out_proj",
|
|
"shared_expert.gate_proj",
|
|
"in_proj_qkv",
|
|
"o_proj"
|
|
],
|
|
"target_parameters": null,
|
|
"task_type": "CAUSAL_LM",
|
|
"trainable_token_indices": null,
|
|
"use_bdlora": null,
|
|
"use_dora": false,
|
|
"use_qalora": false,
|
|
"use_rslora": false
|
|
} |