model: model_path: /path/to/pretain_ckpt/hf_ckpt tokenizer_path: /path/to/Qwen3-VL-4B-Instruct post_training: true adanorm_time: true config_key: LingbotVLAV2Config moe_implementation: fused data: datasets_type: vla data_name: multi train_path: assets/training_data/robotwin.txt robot_config_root: ./configs/robot_configs joints: - arm.position: 14 - end.position: 14 - effector.position: 2 cameras: - camera_top - camera_wrist_left - camera_wrist_right norm_type: - arm.position: bounds_99_woclip - end.position: bounds_99_woclip - effector.position: bounds_99_woclip num_workers: 8 use_future_image: true train: output_dir: /path/to/save_ckpt moe_monitor_interval: 1000 enable_gradient_checkpointing: false # Set to true if GPU memory is insufficient, but this will slow down training. precompute_grid_thw: true vlm_causal: true vlm_fsdp: true attention_implementation: flex_cached use_moe: true token_moe_layers: [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35] token_num_experts: 32 token_top_k: 4 token_moe_intermediate_size: 512 token_shared_intermediate_size: 704 bias_update_speed: 0 sequence_wise_mode: "per_sequence" sequence_wise_loss_coeff: 1e-3 router_z_loss_coeff: 1e-4 router_activation: "sigmoid" routed_scaling_factor: 4.0 use_shared_expert_gate: false use_moe_expert_lr: false # Distributed Muon uses one learning rate for matrix parameters. loss_type: L1_fm data_parallel_mode: fsdp2 enable_full_shard: false module_fsdp_enable: true use_compile: true use_wandb: false rmpad: false rmpad_with_pos_ids: false ulysses_parallel_size: 1 freeze_vision_encoder: false tokenizer_max_length: 72 action_dim: 55 max_action_dim: 55 max_state_dim: 55 lr: 1.0e-4 lr_min: 5.0e-5 lr_decay_style: cosine # Distributed Muon requires FSDP2 and at least two data-parallel ranks. optimizer: dist_muon muon_momentum: 0.95 muon_nesterov: true muon_ns_steps: 5 muon_adjust_lr_fn: match_rms_adamw dist_muon_attn_per_head: true # Update Q/K/V projections per attention head. dist_muon_layers_per_bucket: 2 # Group adjacent layers for communication/computation overlap. # Use AdamW for these final-layer matrices; adjust names for other architectures. muon_exclude_name_patterns: - qwenvl.model.language_model.layers.35.mlp.down_proj.weight - qwenvl.model.language_model.layers.35.mlp.gate_proj.weight - qwenvl.model.language_model.layers.35.mlp.up_proj.weight - qwenvl.model.language_model.layers.35.self_attn.o_proj.weight num_train_epochs: 29000 micro_batch_size: 32 global_batch_size: 1024 # Training uses 32 GPUs: global_batch_size = micro_batch_size * 32 max_steps: 50000 ckpt_manager: dcp save_steps: 10000 save_epochs: 29000 enable_fp32: true enable_resume: true align_params: mode: 'query' num_task_tokens: 8 depth_loss_weight: 0.004 future_depth_loss_weight: 0.004 use_future_video: true llm: dim_out: 2560 image_token_size: 8 image_input_size: 224 depth: model_type: MoRGBD moge_path: /path/to/depth/moge2-vitb-normal.pt morgbd_path: /path/to/depth/model.pt num_layers: 1 num_heads: 4 dim_head: 32 ff_mult: 1 num_backbone_tokens: 256 token_size: 16 dim_out: 1024 input_size: 224 use_future_depth: true block_future_depth_to_action: false detach_future_image_feats: true video: ckpt_path: /path/to/dino_video/teacher_step_10000.pth config_path: /path/to/dino_video/config.yaml attention_mode: flex_block_causal input_size: 256 block_suffix_to_future_video: false share_future_depth_query: true use_shared_future_task_proj: true use_current_shared_task_proj: true num_future_frames: 1 # Keeps DINO teacher input as [warmup current, current, future]. # Current-DINO target uses current_index=1 in code. use_warmup_frame: true effective_fps: 1.0 n_blocks: 1 cls_pool: last detach_image_feats: true num_layers: 1 num_heads: 4 dim_head: 32 ff_mult: 1 num_backbone_tokens: 256 dim_out: 1024 future_video_loss_weight: 0.004 use_smooth_l1_loss: false use_mse_loss: true mse_loss_weight: 1.0 # Enables current-DINO patch alignment and current depth/DINO shared query projection. use_patch_loss: true use_current_patch_loss: true use_cosine_loss: true cosine_loss_weight: 0.2 use_cls_loss: false cls_loss_type: mse cls_loss_weight: 0.2 log_max_samples: 32 log_scale: 16 visual_steps: 5000