model: model_path: /path/to/pretain_ckpt/hf_ckpt tokenizer_path: /path/to/Qwen3-VL-4B-Instruct post_training: true adanorm_time: true config_key: LingbotVLAV2Config moe_implementation: fused data: datasets_type: vla data_name: multi train_path: path/to/training_data.txt (e.g., assets/training_data/robotwin.txt) robot_config_root: configs/robot_configs joints: - arm.position: 14 - end.position: 14 - effector.position: 2 - waist.position: 4 - head.position: 2 - base.position: 3 - hand.position: 12 cameras: - camera_top - camera_wrist_left - camera_wrist_right norm_type: - arm.position: meanstd - end.position: meanstd - effector.position: meanstd - waist.position: meanstd - head.position: meanstd - base.position: meanstd - hand.position: meanstd num_workers: 8 use_future_image: true train: output_dir: /path/to/save_ckpt moe_monitor_interval: 1000 enable_gradient_checkpointing: false # Set to true if GPU memory is insufficient, but this will slow down training. precompute_grid_thw: true vlm_causal: true vlm_fsdp: true attention_implementation: flex_cached use_moe: true token_moe_layers: [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35] token_num_experts: 32 token_top_k: 4 token_moe_intermediate_size: 512 token_shared_intermediate_size: 704 bias_update_speed: 0 sequence_wise_mode: "per_sequence" sequence_wise_loss_coeff: 1e-3 router_z_loss_coeff: 1e-4 router_activation: "sigmoid" routed_scaling_factor: 4.0 use_shared_expert_gate: false use_moe_expert_lr: true data_parallel_mode: fsdp2 enable_full_shard: false module_fsdp_enable: true use_compile: true use_wandb: false rmpad: false rmpad_with_pos_ids: false ulysses_parallel_size: 1 freeze_vision_encoder: false tokenizer_max_length: 72 action_dim: 55 max_action_dim: 55 max_state_dim: 55 optimizer: muon lr: 5.0e-5 lr_decay_style: constant num_train_epochs: 29000 micro_batch_size: 32 global_batch_size: 256 max_steps: 60000 ckpt_manager: dcp save_steps: 20000 save_epochs: 29000 enable_fp32: true enable_resume: true align_params: mode: 'query' num_task_tokens: 8 depth_loss_weight: 0.004 future_depth_loss_weight: 0.004 use_future_video: true llm: dim_out: 2560 image_token_size: 8 image_input_size: 224 depth: model_type: MoRGBD moge_path: /path/to/depth/moge2-vitb-normal.pt morgbd_path: /path/to/depth/model.pt num_layers: 1 num_heads: 4 dim_head: 32 ff_mult: 1 num_backbone_tokens: 256 token_size: 16 dim_out: 1024 input_size: 224 use_future_depth: true block_future_depth_to_action: false detach_future_image_feats: true video: ckpt_path: /path/to/dino_video/teacher_step_10000.pth config_path: /path/to/dino_video/config.yaml attention_mode: flex_block_causal input_size: 256 block_suffix_to_future_video: false share_future_depth_query: true use_shared_future_task_proj: true use_current_shared_task_proj: true num_future_frames: 1 # Keeps DINO teacher input as [warmup current, current, future]. # Current-DINO target uses current_index=1 in code. use_warmup_frame: true effective_fps: 1.0 n_blocks: 1 cls_pool: last detach_image_feats: true num_layers: 1 num_heads: 4 dim_head: 32 ff_mult: 1 num_backbone_tokens: 256 dim_out: 1024 future_video_loss_weight: 0.004 use_smooth_l1_loss: false use_mse_loss: true mse_loss_weight: 1.0 # Enables current-DINO patch alignment and current depth/DINO shared query projection. use_patch_loss: true use_current_patch_loss: true use_cosine_loss: true cosine_loss_weight: 0.2 use_cls_loss: false cls_loss_type: mse cls_loss_weight: 0.2 log_max_samples: 32 log_scale: 16 visual_steps: 5000