ENV: PYTORCH_CUDA_ALLOC_CONF: 'expandable_segments:True' MAX_PIXELS: '1003520' VIDEO_MAX_PIXELS: '50176' FPS_MAX_FRAMES: '12' # model args model: "Qwen/Qwen3.5-35B-A3B" tuner_type: lora lora_rank: 8 lora_alpha: 32 target_modules: all-linear # dataset args dataset: - 'AI-ModelScope/alpaca-gpt4-data-zh#500' - 'AI-ModelScope/alpaca-gpt4-data-en#500' - 'swift/self-cognition#500' - 'AI-ModelScope/LaTeX_OCR:human_handwrite#2000' load_from_cache_file: true split_dataset_ratio: 0.01 max_length: 2048 dataloader_num_workers: 4 dataset_num_proc: 4 model_author: swift model_name: swift-bot padding_free: true packing: true # template args loss_scale: ignore_empty_think add_non_thinking_prefix: true # training args save_safetensors: true merge_lora: true micro_batch_size: 1 global_batch_size: 4 num_train_epochs: 1 finetune: true lr: 1e-4 lr_warmup_fraction: 0.05 min_lr: 1e-5 output_dir: megatron_output/Qwen3.5-35B-A3B eval_steps: 200 save_steps: 200 no_save_optim: true no_save_rng: true freeze_llm: false freeze_vit: true freeze_aligner: true expert_model_parallel_size: 4 sequence_parallel: true moe_permute_fusion: true moe_grouped_gemm: true moe_shared_expert_overlap: true moe_aux_loss_coeff: 1e-6 recompute_granularity: full recompute_method: uniform recompute_num_layers: 1 cross_entropy_loss_fusion: true attention_backend: flash