diff --git a/wall_x/data/load_lerobot_dataset.py b/wall_x/data/load_lerobot_dataset.py index e4e18c8..373dd8a 100644 --- a/wall_x/data/load_lerobot_dataset.py +++ b/wall_x/data/load_lerobot_dataset.py @@ -216,25 +216,15 @@ class DataCollator: if self.config.get("padding_side", "left") == "left": self._processor_cache[processor_path].tokenizer.padding_side = "left" - if action_tokenizer_path not in self._action_tokenizer_cache: + if self.use_fast_tokenizer and action_tokenizer_path not in self._action_tokenizer_cache: self._action_tokenizer_cache[action_tokenizer_path] = AutoProcessor.from_pretrained(action_tokenizer_path, trust_remote_code=True) self.processor = self._processor_cache[processor_path] - self.val_processor = self._processor_cache[processor_path] - self.train_action_tokenizer = self._action_tokenizer_cache[action_tokenizer_path] - self.val_action_tokenizer = self._action_tokenizer_cache[action_tokenizer_path] - new_tokens = ["<|propri|>", "<|action|>"] - - new_tokens += [f"<|action_token_{i}|>" for i in range(self.train_action_tokenizer.vocab_size)] if not self.use_fast_tokenizer: self.train_action_tokenizer = None - self.val_action_tokenizer = None - - # Only add tokens if not already added - if "<|propri|>" not in self.processor.tokenizer.get_vocab(): - num_added_tokens = self.processor.tokenizer.add_tokens(new_tokens) - self.val_processor.tokenizer.add_tokens(new_tokens) + else: + self.train_action_tokenizer = self._action_tokenizer_cache[action_tokenizer_path] if self.use_fast_tokenizer: self.action_mapper = {} diff --git a/workspace/README.md b/workspace/README.md index f174a98..a74fdf4 100644 --- a/workspace/README.md +++ b/workspace/README.md @@ -19,10 +19,10 @@ git clone https://huggingface.co/physical-intelligence/fast ## Required Paths (Must Modify) ```yaml -pretrained_wallx_path: "/path/to/wallx_model/" # Path to pretrained Qwen VL model -use_fast_tokenizer: false # True: train FAST, False: train Flow -action_tokenizer_path: "/path/to/fast/" # Path to action tokenizer +pretrained_wallx_path: "/path/to/wallx_model/" # Path to pretrained wallx model save_path: "/path/to/workspace/" # Path to save training outputs +use_fast_tokenizer: False # True: train FAST, False: train Flow +action_tokenizer_path: "/path/to/fast/" # Must set if use_fast_tokenizer is True ``` ## Training Parameters (Commonly Modified) diff --git a/workspace/lerobot_example/config_qact.yml b/workspace/lerobot_example/config_qact.yml index 76de0cc..60c9d77 100644 --- a/workspace/lerobot_example/config_qact.yml +++ b/workspace/lerobot_example/config_qact.yml @@ -5,10 +5,10 @@ log_name: "robotic_training" log_project: "vla_training" model_type: qwen2_5 -pretrained_wallx_path: "/path/to/wallx_model/" -use_fast_tokenizer: false # True: train FAST, False: train Flow -action_tokenizer_path: "/path/to/fast/" -save_path: "/path/to/workspace/" +pretrained_wallx_path: "/path/to/wallx_model/" # Must set +save_path: "/path/to/workspace/" # Must set +use_fast_tokenizer: False # True: train FAST, False: train Flow +action_tokenizer_path: "/path/to/fast/" # Must set if use_fast_tokenizer is true # Torch Profile profile: False