# Training Configuration for Wall-X Robotic Multi-Modal Learning # This configuration supports multi-modal learning with vision, language, and action data # Model and paths configuration log_name: "robotic_training" log_project: "vla_training" model_type: qwen2_5 pretrained_qwen_vl_path: "/path/to/qwen_vl_model/" # whether to enable fast tokenizer use_fast_tokenizer: false action_tokenizer_path: "/path/to/fast/" save_path: "/path/to/workspace/" # Torch Profile profile: False profile_save_path: /path/to/profile/ profile_wait_iters: 10 profile_warmup_iters: 5 profile_active_iters: 2 # Training hyperparameters num_warmup_steps: 100 num_training_steps: 64000000 learning_rate: 0.00009 min_lr: 0.00005 num_epoch: 100 gradient_accumulation_steps: 32 batch_size_per_gpu: 8 padding_side: left # Robot configuration - Define degrees of freedom for each component dof_config: follow_left_ee_cartesian_pos: 3 # Left end-effector Cartesian position follow_left_ee_rotation: 3 # Left end-effector rotation follow_left_gripper: 1 # Left gripper control follow_right_ee_cartesian_pos: 3 # Right end-effector Cartesian position follow_right_ee_rotation: 3 # Right end-effector rotation follow_right_gripper: 1 # Right gripper control head_actions: 2 # Head/camera movement height: 1 # Mobile base height control car_pose: 3 # Mobile base pose (x, y, theta) # Agent proprioception configuration (typically matches DOF config) agent_pos_config: follow_left_ee_cartesian_pos: 3 follow_left_ee_rotation: 3 follow_left_gripper: 1 follow_right_ee_cartesian_pos: 3 follow_right_ee_rotation: 3 follow_right_gripper: 1 head_actions: 2 height: 1 car_pose: 3 # # Checkpoint resuming configuration # resume: # ckpt: "/path/to/resume_model/" # load_ckpt_only: true # Data configuration data: use_lerobot: true # LeRobot dataset configuration lerobot_config: repo_id: "lerobot/aloha_mobile_cabinet" root: null episodes: null image_transforms: null delta_timestamps: null tolerance_s: 1e-4 revision: null force_cache_sync: false download_videos: true video_backend: null action_horizon: 32 train_test_split: 0.95 # Action keys for observation and prediction obs_action_keys: - follow_left_ee_cartesian_pos - follow_left_ee_rotation - follow_left_gripper - follow_right_ee_cartesian_pos - follow_right_ee_rotation - follow_right_gripper - head_actions - height - car_pose predict_action_keys: - follow_left_ee_cartesian_pos - follow_left_ee_rotation - follow_left_gripper - follow_right_ee_cartesian_pos - follow_right_ee_rotation - follow_right_gripper - head_actions - height - car_pose # Image resolution configuration for different camera views resolution: face_view: 256 left_wrist_view: 256 right_wrist_view: 256 move1_view: 256 move2_view: 256 top_view: 256 wall_view: 256 multi_modal: 256