diff --git a/README.md b/README.md index 18d59e5..74907ff 100644 --- a/README.md +++ b/README.md @@ -76,6 +76,7 @@ Training script path configuration - Robot DOF configuration - Training hyperparameters +Download the Flow/FAST pretrained model and run: ```bash bash ./workspace/lerobot_example/run.sh ``` diff --git a/scripts/draw_openloop_plot.py b/scripts/draw_openloop_plot.py index a649392..2be5694 100644 --- a/scripts/draw_openloop_plot.py +++ b/scripts/draw_openloop_plot.py @@ -38,8 +38,7 @@ gt_traj = torch.zeros((total_frames, action_dim)) pred_traj = torch.zeros((total_frames, action_dim)) for idx, batch in enumerate(dataloader): - gt_traj[idx] = batch['action_chunk'][0, 0,:action_dim] - if idx % 32 ==0 and idx + 32 < total_frames: + if idx % pred_horizon ==0 and idx + pred_horizon < total_frames: batch = batch.to("cuda") with torch.no_grad(): outputs = model( @@ -50,7 +49,13 @@ for idx, batch in enumerate(dataloader): predict_mode="fast" ) pred_traj[idx : idx + pred_horizon] = outputs['predict_action'].detach().cpu() - + + # Denormalize ground truth actions + gt_action_chunk = batch['action_chunk'][:, :, :action_dim] + dof_mask = batch["dof_mask"].to(gt_action_chunk.dtype) + denormalized_gt = model.action_preprocessor.normalizer_action.unnormalize_data(gt_action_chunk, ["x2_normal"], dof_mask) + gt_traj[idx : idx + pred_horizon] = denormalized_gt.detach().cpu() + gt_traj_np = gt_traj.numpy() pred_traj_np = pred_traj.numpy() diff --git a/wall_x/data/load_lerobot_dataset.py b/wall_x/data/load_lerobot_dataset.py index d502417..344d567 100644 --- a/wall_x/data/load_lerobot_dataset.py +++ b/wall_x/data/load_lerobot_dataset.py @@ -207,7 +207,7 @@ class DataCollator: self.load_processor() def load_processor(self): - processor_path = self.config["processor_path"] + processor_path = self.config["pretrained_qwen_vl_path"] action_tokenizer_path = self.config["action_tokenizer_path"] # Use cached processors if available diff --git a/wall_x/model/qwen2_5_based/modeling_qwen2_5_vl_act.py b/wall_x/model/qwen2_5_based/modeling_qwen2_5_vl_act.py index 1a0d65b..cf0edb8 100644 --- a/wall_x/model/qwen2_5_based/modeling_qwen2_5_vl_act.py +++ b/wall_x/model/qwen2_5_based/modeling_qwen2_5_vl_act.py @@ -1,6 +1,7 @@ import os import torch import numpy as np +import glob import torch.nn as nn from torchdiffeq import odeint from dataclasses import dataclass @@ -685,7 +686,6 @@ class Qwen2_5_VLMoEForAction(Qwen2_5_VLForConditionalGeneration): """ # Load model components from pretrained path - model_path = os.path.join(pretrained_model_path, "model.safetensors") config_path = os.path.join(pretrained_model_path, "config.json") config = cls.config_class.from_pretrained(config_path) processor = AutoProcessor.from_pretrained(pretrained_model_path, use_fast=True) @@ -703,8 +703,13 @@ class Qwen2_5_VLMoEForAction(Qwen2_5_VLForConditionalGeneration): model.resize_token_embeddings(len(processor.tokenizer)) # Load model state dict from safetensors file - state_dict = load_file(model_path, device="cpu") - msg = model.load_state_dict(state_dict, strict=False) + safetensor_files = glob.glob(os.path.join(pretrained_model_path, "*.safetensors")) + state_dict = {} + for file in safetensor_files: + sd = load_file(file, device="cpu") + state_dict.update(sd) + + model.load_state_dict(state_dict, strict=False) return model diff --git a/wall_x/trainer/qwen_vl_act_trainer.py b/wall_x/trainer/qwen_vl_act_trainer.py index f3301ea..4a04cd7 100644 --- a/wall_x/trainer/qwen_vl_act_trainer.py +++ b/wall_x/trainer/qwen_vl_act_trainer.py @@ -108,7 +108,7 @@ class QwenVlAct_Trainer: ValueError: If required configuration keys are missing """ # Validate required configuration keys - required_keys = ["processor_path", "qwen_vl_act_config_path", "learning_rate", "num_epoch"] + required_keys = ["learning_rate", "num_epoch"] for key in required_keys: if key not in config: raise ValueError(f"Missing required configuration key: {key}") @@ -197,6 +197,9 @@ class QwenVlAct_Trainer: self.train_loop(epoch) self.accelerator.wait_for_everyone() + if (epoch + 1) % self.config.get("epoch_save_interval", 10) == 0: + self.save_checkpoint(epoch) + # Validation after each epoch self.val_loop() self.accelerator.wait_for_everyone() diff --git a/workspace/README.md b/workspace/README.md index ccc7a7a..9a86aec 100644 --- a/workspace/README.md +++ b/workspace/README.md @@ -2,6 +2,12 @@ This document explains the key configuration parameters that can be modified for Wall-X training. +## Enable FAST tokenizer +To fine-tune using the FAST tokenizer, please download the repository and update the `action_tokenizer_path`. Make sure to set `use_fast_tokenizer` to `true`: +```bash +git clone https://huggingface.co/physical-intelligence/fast +``` + ## Quick Start Checklist 1. **Update run.sh**: Set `code_dir` and `config_path` to your actual paths 2. **Configure GPUs**: Set `CUDA_VISIBLE_DEVICES` for your available GPUs diff --git a/workspace/lerobot_example/config_qact.yml b/workspace/lerobot_example/config_qact.yml index 4fa62cd..83e8d50 100644 --- a/workspace/lerobot_example/config_qact.yml +++ b/workspace/lerobot_example/config_qact.yml @@ -5,9 +5,8 @@ log_name: "robotic_training" log_project: "vla_training" model_type: qwen2_5 -processor_path: "/path/to/model/" -pretrained_qwen_vl_path: "/path/to/qwen_vl_model/" -qwen_vl_act_config_path: "/path/to/config.json" +pretrained_qwen_vl_path: "/path/to/wallx_model/" +use_fast_tokenizer: false # True: train FAST, False: train Flow action_tokenizer_path: "/path/to/fast/" save_path: "/path/to/workspace/" @@ -27,6 +26,7 @@ num_epoch: 100 gradient_accumulation_steps: 32 batch_size_per_gpu: 8 padding_side: left +epoch_save_interval: 10 # Robot configuration - Define degrees of freedom for each component dof_config: @@ -52,10 +52,10 @@ agent_pos_config: height: 1 car_pose: 3 -# Checkpoint resuming configuration -resume: - ckpt: "/path/to/resume_model/" - load_ckpt_only: true +# # Checkpoint resuming configuration +# resume: +# ckpt: "/path/to/resume_model/" +# load_ckpt_only: true # Data configuration data: