# Train from Qwen-2.5-VL # Model and paths configuration log_name: "robotic_training" log_project: "vla_training" model_type: qwen2_5 pretrained_wallx_path: "/path/to/wallx_model/" # Must set save_path: "/path/to/workspace/" # Must set use_fast_tokenizer: True # True: train FAST, False: train Flow action_tokenizer_path: "/path/to/fast/" # Must set if use_fast_tokenizer is true qwen_vl_act_config_path: "wall-x/workspace/lerobot_example/qwen25_config.json" # Torch Profile profile: False profile_save_path: /path/to/profile/ profile_wait_iters: 10 profile_warmup_iters: 5 profile_active_iters: 2 # Training hyperparameters num_warmup_steps: 100 num_training_steps: 64000000 learning_rate: 0.00009 min_lr: 0.00005 num_epoch: 100 gradient_accumulation_steps: 1 batch_size_per_gpu: 8 padding_side: left epoch_save_interval: 10 # Training optimization settings FSDP2: True torch_compile: False # Robot configuration - Define degrees of freedom for each component dof_config: follow_left_ee_cartesian_pos: 3 # Left end-effector Cartesian position follow_left_ee_rotation: 3 # Left end-effector rotation follow_left_gripper: 1 # Left gripper control follow_right_ee_cartesian_pos: 3 # Right end-effector Cartesian position follow_right_ee_rotation: 3 # Right end-effector rotation follow_right_gripper: 1 # Right gripper control head_actions: 2 # Head/camera movement height: 1 # Mobile base height control car_pose: 3 # Mobile base pose (x, y, theta) # Agent proprioception configuration (typically matches DOF config) agent_pos_config: follow_left_ee_cartesian_pos: 3 follow_left_ee_rotation: 3 follow_left_gripper: 1 follow_right_ee_cartesian_pos: 3 follow_right_ee_rotation: 3 follow_right_gripper: 1 head_actions: 2 height: 1 car_pose: 3 # # Checkpoint resuming configuration # resume: # ckpt: "/path/to/resume_model/" # load_ckpt_only: true norm_stats_path: "/path/to/norm_stats.json" enable_customized_robot_config: true customized_robot_config: name: "physical-intelligence/libero" customized_dof_config: "action_left_shoulder" : 1 "action_left_elbow" : 1 "action_left_forearm_roll" : 1 "action_left_wrist_angle" : 1 "action_left_wrist_rotate" : 1 "action_left_gripper" : 1 "action_right_waist" : 1 "action_right_shoulder" : 1 "action_right_elbow" : 1 "action_right_forearm_roll" : 1 "action_right_wrist_angle" : 1 "action_right_wrist_rotate" : 1 "action_right_gripper" : 1 customized_agent_pos_config: "state_left_shoulder" : 1 "state_left_elbow" : 1 "state_left_forearm_roll" : 1 "state_left_wrist_angle" : 1 "state_left_wrist_rotate" : 1 "state_left_gripper" : 1 "state_right_waist" : 1 "state_right_shoulder" : 1 "state_right_elbow" : 1 "state_right_forearm_roll" : 1 "state_right_wrist_angle" : 1 "state_right_wrist_rotate" : 1 "state_right_gripper" : 1 # Data configuration data: use_lerobot: true # LeRobot dataset configuration lerobot_config: repo_id: "lerobot/aloha_mobile_cabinet" root: null episodes: null image_transforms: null delta_timestamps: null tolerance_s: 1e-4 revision: null force_cache_sync: false download_videos: true video_backend: null action_horizon: 32 train_test_split: 0.95 # Action keys for observation and prediction obs_action_keys: - follow_left_ee_cartesian_pos - follow_left_ee_rotation - follow_left_gripper - follow_right_ee_cartesian_pos - follow_right_ee_rotation - follow_right_gripper - head_actions - height - car_pose predict_action_keys: - follow_left_ee_cartesian_pos - follow_left_ee_rotation - follow_left_gripper - follow_right_ee_cartesian_pos - follow_right_ee_rotation - follow_right_gripper - head_actions - height - car_pose # Image resolution configuration for different camera views resolution: face_view: 256 left_wrist_view: 256 right_wrist_view: 256 move1_view: 256 move2_view: 256 top_view: 256 wall_view: 256 multi_modal: 256