This commit is contained in:
Starrick
2025-09-07 14:59:17 +08:00
commit 24dbdbd24b
40 changed files with 10754 additions and 0 deletions
View File
+400
View File
@@ -0,0 +1,400 @@
import math
import torch
import torch.nn as nn
from torch.distributions import Beta
from wall_x.utils.constant import action_statistic_dof
class Normalizer(nn.Module):
"""
Action data normalizer for multi-robot systems.
This module handles normalization and denormalization of action data for different robot
configurations. It maintains per-robot statistics (min values and deltas) and applies
normalization to map actions to the [-1, 1] range.
"""
def __init__(self, action_statistic_dof, dof_config):
"""
Initialize the normalizer with robot-specific action statistics.
Args:
action_statistic_dof (dict): Statistical data for each robot's degrees of freedom
dof_config (dict): Configuration mapping for degrees of freedom per robot
"""
super(Normalizer, self).__init__()
action_statistic = {}
# Process statistics for each robot
for robot_name in action_statistic_dof.keys():
action_statistic[robot_name] = {}
all_dof_min = []
all_dof_delta = []
# Collect min and delta values for all DOFs
for k in dof_config:
if k in action_statistic_dof[robot_name]:
all_dof_min.extend(action_statistic_dof[robot_name][k]["min"])
all_dof_delta.extend(action_statistic_dof[robot_name][k]["delta"])
else:
# Use default values if statistics not available
all_dof_min.extend([0.0] * dof_config[k])
all_dof_delta.extend([1.0] * dof_config[k])
all_dof_min = torch.tensor(all_dof_min)
all_dof_delta = torch.tensor(all_dof_delta)
action_statistic[robot_name]["min"] = all_dof_min
action_statistic[robot_name]["delta"] = all_dof_delta
# Register statistics as non-trainable parameters
self.min = nn.ParameterDict({
k: nn.Parameter(action_statistic[k]["min"], requires_grad=False)
for k in action_statistic.keys()
})
self.delta = nn.ParameterDict({
k: nn.Parameter(action_statistic[k]["delta"], requires_grad=False)
for k in action_statistic.keys()
})
def normalize_data(self, xs, dataset_names):
"""
Normalize action data to [-1, 1] range using robot-specific statistics.
Args:
xs: Input action data tensors
dataset_names: List of dataset/robot names corresponding to each tensor
Returns:
torch.Tensor: Normalized action data in [-1, 1] range
"""
new_xs = []
# Filter out multimodal dataset entries
dataset_names = [name for name in dataset_names if name != "x2_multimodal"]
for x, dataset_name in zip(xs, dataset_names):
# Apply min-max normalization
x = (x - self.min[dataset_name]) / (self.delta[dataset_name])
# Scale to [-1, 1] range
x = x * 2 - 1
# Clamp to ensure bounds
x = torch.clamp(x, -1, 1)
new_xs.append(x)
new_xs = torch.stack(new_xs)
return new_xs
def unnormalize_data(self, xs, dataset_names, dof_mask=None):
"""
Convert normalized data back to original action space.
Args:
xs: Normalized action data in [-1, 1] range
dataset_names: List of dataset/robot names
dof_mask: Optional mask to select specific degrees of freedom
Returns:
torch.Tensor: Denormalized action data in original scale
"""
new_xs = []
# Filter out multimodal dataset entries
dataset_names = [name for name in dataset_names if name != "x2_multimodal"]
dof_mask = dof_mask if dof_mask is not None else [None] * len(xs)
for x, dataset_name, mask in zip(xs, dataset_names, dof_mask):
# Convert from [-1, 1] to [0, 1] range
x = (x + 1) / 2
# Apply DOF mask if provided
if mask is not None:
mask = mask[0].bool()
action_space_delta = self.delta[dataset_name][mask]
action_space_min = self.min[dataset_name][mask]
else:
action_space_delta = self.delta[dataset_name]
action_space_min = self.min[dataset_name]
# Scale back to original range
x = x * action_space_delta + action_space_min
new_xs.append(x)
new_xs = torch.stack(new_xs)
return new_xs
class SinusoidalPosEmb(nn.Module):
"""
Sinusoidal positional embedding for diffusion timesteps.
Generates sinusoidal embeddings commonly used in diffusion models to encode
timestep information with different frequencies.
"""
def __init__(self, dim):
"""
Initialize sinusoidal positional embedding.
Args:
dim (int): Embedding dimension (must be even)
"""
super().__init__()
self.dim = dim
def forward(self, x):
"""
Generate sinusoidal embeddings for input timesteps.
Args:
x (torch.Tensor): Input timesteps
Returns:
torch.Tensor: Sinusoidal embeddings of shape (..., dim)
"""
device = x.device
half_dim = self.dim // 2
emb = math.log(10000) / (half_dim - 1)
emb = torch.exp(torch.arange(half_dim, device=device) * -emb)
emb = x[:, None] * emb[None, :]
emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
return emb
class ActionProcessor(nn.Module):
"""
Action sequence processor for robotic control with flow matching.
This module handles action sequence processing for robotic systems with the following capabilities:
1. Adds controlled noise to action sequences using Beta distribution scheduling
2. Generates temporal embeddings for timestep conditioning
3. Projects actions to model hidden space for transformer processing
4. Supports proprioceptive data integration and multi-robot configurations
The Beta distribution provides more flexible noise injection strategies compared to
traditional linear schedules, allowing better control over the noise scheduling process.
"""
def __init__(self, config):
"""
Initialize the action processor with multi-robot support.
Args:
config: Configuration object containing:
- dof_config (dict): Degrees of freedom configuration per robot type
- agent_pos_config (dict): Agent position/proprioception configuration
- hidden_size (int): Model hidden layer dimension
- noise_scheduler (dict): Noise scheduler configuration with Beta parameters
"""
super().__init__()
# Calculate action and proprioception dimensions from configuration
self.dof_config = config.dof_config
self.agent_pos_config = config.agent_pos_config
self.action_dim = sum([v for k, v in self.dof_config.items()])
self.propri_dim = sum([v for k, v in self.agent_pos_config.items()])
# Log configuration details for debugging
print("ActionProcessor Configuration:", flush=True)
print(f" Action dimension: {self.action_dim}", flush=True)
print(f" Proprioception dimension: {self.propri_dim}", flush=True)
print(" DOF configuration:", flush=True)
for key, value in self.dof_config.items():
print(f" {key}: {value}", flush=True)
print(" Agent position configuration:", flush=True)
for key, value in self.agent_pos_config.items():
print(f" {key}: {value}", flush=True)
self.hidden_size = config.hidden_size
# Initialize data normalizers for actions and proprioception
self.normalizer_action = Normalizer(action_statistic_dof, config.dof_config)
self.normalizer_propri = Normalizer(action_statistic_dof, config.agent_pos_config)
# Proprioception projection layer (includes history/current state)
self.propri_proj = nn.Linear(self.propri_dim * 2, self.hidden_size, bias=False)
# Beta distribution noise scheduler configuration
noise_scheduler_config = config.noise_scheduler
self.beta_alpha = noise_scheduler_config.get('beta_alpha', 1.5) # Beta distribution α parameter
self.beta_beta = noise_scheduler_config.get('beta_beta', 1.0) # Beta distribution β parameter
self.s = noise_scheduler_config.get('s', 0.999) # Scaling factor
# Initialize Beta distribution for noise scheduling
alpha_tensor = torch.tensor(self.beta_alpha, dtype=torch.float32).to("cuda")
beta_tensor = torch.tensor(self.beta_beta, dtype=torch.float32).to("cuda")
self.beta_dist = Beta(alpha_tensor, beta_tensor)
# Sinusoidal positional embedding for timesteps
self.time_embed = SinusoidalPosEmb(config.hidden_size)
# Action embedding network: project to hidden space
self.w1 = nn.Linear(self.action_dim * 2, self.hidden_size, bias=False) # *2 for action + DOF mask
self.w2 = nn.Linear(self.hidden_size * 2, self.hidden_size, bias=False) # *2 for action + time embeddings
self.w3 = nn.Linear(self.hidden_size, self.hidden_size, bias=False)
self.act_fn = nn.SiLU()
# Project back to action space for flow matching loss
self.action_proj_back = nn.Linear(self.hidden_size, self.action_dim, bias=False)
self.mse_loss = nn.MSELoss(reduction='none')
def sample_time(self, batch_size, device, dtype):
"""
Sample timesteps using Beta distribution for noise scheduling.
Generates random timesteps in [0,1] range using Beta distribution, then scales them.
This provides more flexible control over the noise injection schedule compared to
uniform sampling.
Args:
batch_size (int): Number of timesteps to sample
device: Target device for tensors
dtype: Target data type for tensors
Returns:
torch.Tensor: Sampled timesteps of shape [batch_size]
"""
sample = self.beta_dist.sample([batch_size]).to(device=device, dtype=dtype)
time = (self.s - sample) / self.s
return time
def proprioception_proj(self, proprioception, dataset_names=None, dof_mask=None, use_history=False):
"""
Project proprioceptive data (joint positions, orientations) to hidden space.
Args:
proprioception (torch.Tensor): Proprioceptive data of shape [batch_size, seq_len, propri_dim]
dataset_names (list, optional): Dataset names for normalization. Defaults to None.
dof_mask (torch.Tensor, optional): DOF mask of shape [batch_size, propri_dim]. Defaults to None.
use_history (bool, optional): Whether to use historical proprioceptive data. Defaults to False.
Returns:
torch.Tensor: Projected proprioceptive features of shape [batch_size, seq_len, hidden_size]
"""
# Ensure proper device and dtype alignment
proprioception = proprioception.to(device=self.propri_proj.weight.device).to(dtype=self.propri_proj.weight.dtype)
if dof_mask is not None:
# Concatenate proprioception with DOF mask
# TODO: Use variable-based dimension checking for better flexibility
if use_history:
proprioception = torch.cat([proprioception, dof_mask], dim=-1)
else:
proprioception = torch.cat([proprioception, dof_mask], dim=-1)
proprioception = proprioception.to(device=self.propri_proj.weight.device).to(dtype=self.propri_proj.weight.dtype)
return self.propri_proj(proprioception)
def forward(self, action_chunk, dataset_names, dof_mask=None):
"""
Process action sequences with noise injection and temporal embedding.
This method implements the forward pass for flow matching training:
1. Adds Beta-distributed noise to action sequences
2. Generates sinusoidal timestep embeddings
3. Projects noisy actions to hidden space
4. Combines action and temporal features
Args:
action_chunk (torch.Tensor): Action sequences of shape [batch_size, seq_len, action_dim]
dataset_names (list): Dataset names for normalization
dof_mask (torch.Tensor, optional): DOF mask of shape [batch_size, seq_len, action_dim].
Defaults to None.
Returns:
tuple: (action_embeddings, flow_target) where:
- action_embeddings: Processed action features of shape [batch_size, seq_len, hidden_size]
- flow_target: Flow matching target (action_chunk - noise) for loss computation
"""
batch_size = action_chunk.shape[0]
device = action_chunk.device
dtype = action_chunk.dtype
# 1. Add noise to action sequences using flow matching
noise = torch.randn_like(action_chunk)
time = self.sample_time(batch_size, device, dtype)
t = time.unsqueeze(-1).unsqueeze(-1) # Broadcast to match action dimensions
# Linear interpolation between noise and action (flow matching)
noisy_action = (1 - t) * noise + t * action_chunk
flow = action_chunk - noise # Flow target for loss computation
# 2. Generate sinusoidal positional encoding for timesteps
time_embed = self.time_embed(time)
# 3. Project noisy actions with DOF mask to hidden space
if dof_mask is not None:
noisy_action = torch.cat([noisy_action, dof_mask], dim=-1)
noisy_action = noisy_action.to(dtype=self.w1.weight.dtype)
action_embed = self.w1(noisy_action)
# Repeat time embedding for each sequence position
time_embed = time_embed.unsqueeze(1).repeat(1, action_embed.shape[1], 1).to(dtype=self.w2.weight.dtype)
# Combine action and temporal embeddings
concat_embed = torch.cat([action_embed, time_embed], dim=-1)
concat_embed = self.w2(concat_embed)
embed = self.w3(self.act_fn(concat_embed))
return embed, flow
def step(self, timestep, noisy_action, dof_mask=None):
"""
Single denoising step for diffusion inference.
Processes noisy actions at a specific timestep for iterative denoising during inference.
Args:
timestep (torch.Tensor): Current timesteps of shape [batch_size]
noisy_action (torch.Tensor): Noisy actions of shape [batch_size, seq_len, action_dim]
dof_mask (torch.Tensor, optional): DOF mask for action space. Defaults to None.
Returns:
torch.Tensor: Processed action embeddings of shape [batch_size, seq_len, hidden_size]
"""
# Concatenate noisy action with DOF mask if provided
if dof_mask is not None:
noisy_action = torch.cat([noisy_action, dof_mask], dim=-1)
# Generate timestep embeddings
time_embed = self.time_embed(timestep) # [batch_size, hidden_size]
# Project noisy actions
action_embed = self.w1(noisy_action)
# Broadcast time embeddings to sequence length
time_embed = time_embed.unsqueeze(1).repeat(1, action_embed.shape[1], 1)
time_embed = time_embed.to(device=noisy_action.device).to(dtype=noisy_action.dtype)
# Combine embeddings and process through MLP
concat_embed = torch.cat([action_embed, time_embed], dim=-1)
concat_embed = self.w2(concat_embed)
embed = self.w3(self.act_fn(concat_embed))
return embed
def flow_loss(self, action_hidden_states, flow, dof_mask=None):
"""
Compute flow matching loss between predicted and target actions.
Args:
action_hidden_states (torch.Tensor): Hidden states from transformer
flow (torch.Tensor): Target flow (action - noise) for matching
dof_mask (torch.Tensor, optional): DOF mask to weight loss per dimension. Defaults to None.
Returns:
torch.Tensor: Flow matching loss (no reduction for channel loss computation)
"""
# Project hidden states back to action space
action_pred = self.action_proj_back(action_hidden_states)
# Compute MSE loss between predicted and target flow
loss = self.mse_loss(action_pred, flow)
# Apply DOF mask if provided
if dof_mask is not None:
dof_mask = dof_mask.reshape(-1, dof_mask.shape[-1])
loss = loss * dof_mask
# Return loss without reduction for channel-wise loss computation
return loss
+2
View File
@@ -0,0 +1,2 @@
from .modeling_qwen2_5_vl_act import Qwen2_5_VLMoEModel,Qwen2_5_VLMoEForAction
from .configuration_qwen2_5_vl import Qwen2_5_VLConfig
@@ -0,0 +1,248 @@
from transformers.configuration_utils import PretrainedConfig
from transformers.modeling_rope_utils import rope_config_validation
class Qwen2_5_VLVisionConfig(PretrainedConfig):
model_type = "qwen2_5_vl"
base_config_key = "vision_config"
def __init__(
self,
depth=32,
hidden_size=3584,
hidden_act="silu",
intermediate_size=3420,
num_heads=16,
in_channels=3,
patch_size=14,
spatial_merge_size=2,
temporal_patch_size=2,
tokens_per_second=4,
window_size=112,
out_hidden_size=3584,
fullatt_block_indexes=[7, 15, 23, 31],
**kwargs,
):
super().__init__(**kwargs)
self.depth = depth
self.hidden_size = hidden_size
self.hidden_act = hidden_act
self.intermediate_size = intermediate_size
self.num_heads = num_heads
self.in_channels = in_channels
self.patch_size = patch_size
self.spatial_merge_size = spatial_merge_size
self.temporal_patch_size = temporal_patch_size
self.tokens_per_second = tokens_per_second
self.window_size = window_size
self.fullatt_block_indexes = fullatt_block_indexes
self.out_hidden_size = out_hidden_size
class Qwen2_5_VLConfig(PretrainedConfig):
r"""
This is the configuration class to store the configuration of a [`Qwen2_5_VLModel`]. It is used to instantiate a
Qwen2-VL model according to the specified arguments, defining the model architecture. Instantiating a configuration
with the defaults will yield a similar configuration to that of
Qwen2-VL-7B-Instruct [Qwen/Qwen2-VL-7B-Instruct](https://huggingface.co/Qwen/Qwen2-VL-7B-Instruct).
Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
documentation from [`PretrainedConfig`] for more information.
Args:
vocab_size (`int`, *optional*, defaults to 152064):
Vocabulary size of the Qwen2_5_VL model. Defines the number of different tokens that can be represented by the
`inputs_ids` passed when calling [`Qwen2_5_VLModel`]
hidden_size (`int`, *optional*, defaults to 8192):
Dimension of the hidden representations.
intermediate_size (`int`, *optional*, defaults to 29568):
Dimension of the MLP representations.
num_hidden_layers (`int`, *optional*, defaults to 80):
Number of hidden layers in the Transformer encoder.
num_attention_heads (`int`, *optional*, defaults to 64):
Number of attention heads for each attention layer in the Transformer encoder.
num_key_value_heads (`int`, *optional*, defaults to 8):
This is the number of key_value heads that should be used to implement Grouped Query Attention. If
`num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
`num_key_value_heads=1` the model will use Multi Query Attention (MQA) otherwise GQA is used. When
converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
by meanpooling all the original heads within that group. For more details checkout [this
paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to `32`.
hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
The non-linear activation function (function or string) in the decoder.
max_position_embeddings (`int`, *optional*, defaults to 32768):
The maximum sequence length that this model might ever be used with.
initializer_range (`float`, *optional*, defaults to 0.02):
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
rms_norm_eps (`float`, *optional*, defaults to 1e-05):
The epsilon used by the rms normalization layers.
use_cache (`bool`, *optional*, defaults to `True`):
Whether or not the model should return the last key/values attentions (not used by all models). Only
relevant if `config.is_decoder=True`.
tie_word_embeddings (`bool`, *optional*, defaults to `False`):
Whether the model's input and output word embeddings should be tied.
rope_theta (`float`, *optional*, defaults to 1000000.0):
The base period of the RoPE embeddings.
use_sliding_window (`bool`, *optional*, defaults to `False`):
Whether to use sliding window attention.
sliding_window (`int`, *optional*, defaults to 4096):
Sliding window attention (SWA) window size. If not specified, will default to `4096`.
max_window_layers (`int`, *optional*, defaults to 80):
The number of layers that use SWA (Sliding Window Attention). The bottom layers use SWA while the top use full attention.
attention_dropout (`float`, *optional*, defaults to 0.0):
The dropout ratio for the attention probabilities.
vision_config (`Dict`, *optional*):
The config for the visual encoder initialization.
rope_scaling (`Dict`, *optional*):
Dictionary containing the scaling configuration for the RoPE embeddings. NOTE: if you apply new rope type
and you expect the model to work on longer `max_position_embeddings`, we recommend you to update this value
accordingly.
Expected contents:
`rope_type` (`str`):
The sub-variant of RoPE to use. Can be one of ['default', 'linear', 'dynamic', 'yarn', 'longrope',
'llama3'], with 'default' being the original RoPE implementation.
`factor` (`float`, *optional*):
Used with all rope types except 'default'. The scaling factor to apply to the RoPE embeddings. In
most scaling types, a `factor` of x will enable the model to handle sequences of length x *
original maximum pre-trained length.
`original_max_position_embeddings` (`int`, *optional*):
Used with 'dynamic', 'longrope' and 'llama3'. The original max position embeddings used during
pretraining.
`attention_factor` (`float`, *optional*):
Used with 'yarn' and 'longrope'. The scaling factor to be applied on the attention
computation. If unspecified, it defaults to value recommended by the implementation, using the
`factor` field to infer the suggested value.
`beta_fast` (`float`, *optional*):
Only used with 'yarn'. Parameter to set the boundary for extrapolation (only) in the linear
ramp function. If unspecified, it defaults to 32.
`beta_slow` (`float`, *optional*):
Only used with 'yarn'. Parameter to set the boundary for interpolation (only) in the linear
ramp function. If unspecified, it defaults to 1.
`short_factor` (`List[float]`, *optional*):
Only used with 'longrope'. The scaling factor to be applied to short contexts (<
`original_max_position_embeddings`). Must be a list of numbers with the same length as the hidden
size divided by the number of attention heads divided by 2
`long_factor` (`List[float]`, *optional*):
Only used with 'longrope'. The scaling factor to be applied to long contexts (<
`original_max_position_embeddings`). Must be a list of numbers with the same length as the hidden
size divided by the number of attention heads divided by 2
`low_freq_factor` (`float`, *optional*):
Only used with 'llama3'. Scaling factor applied to low frequency components of the RoPE
`high_freq_factor` (`float`, *optional*):
Only used with 'llama3'. Scaling factor applied to high frequency components of the RoPE
```python
>>> from transformers import Qwen2_5_VLForConditionalGeneration, Qwen2_5_VLConfig
>>> # Initializing a Qwen2_5_VL style configuration
>>> configuration = Qwen2_5_VLConfig()
>>> # Initializing a model from the Qwen2-VL-7B style configuration
>>> model = Qwen2_5_VLForConditionalGeneration(configuration)
>>> # Accessing the model configuration
>>> configuration = model.config
```"""
model_type = "qwen2_5_vl"
sub_configs = {"vision_config": Qwen2_5_VLVisionConfig}
keys_to_ignore_at_inference = ["past_key_values"]
# Default tensor parallel plan for base model `Qwen2_5_VL`
base_model_tp_plan = {
"layers.*.self_attn.q_proj": "colwise",
"layers.*.self_attn.k_proj": "colwise",
"layers.*.self_attn.v_proj": "colwise",
"layers.*.self_attn.o_proj": "rowwise",
"layers.*.mlp.gate_proj": "colwise",
"layers.*.mlp.up_proj": "colwise",
"layers.*.mlp.down_proj": "rowwise",
}
base_model_pp_plan = {
"embed_tokens": (["input_ids"], ["inputs_embeds"]),
"layers": (["hidden_states", "attention_mask"], ["hidden_states"]),
"norm": (["hidden_states"], ["hidden_states"]),
}
def __init__(
self,
vocab_size=152064,
hidden_size=8192,
intermediate_size=29568,
num_hidden_layers=80,
num_attention_heads=64,
num_key_value_heads=8,
hidden_act="silu",
max_position_embeddings=32768,
initializer_range=0.02,
rms_norm_eps=1e-05,
use_cache=True,
tie_word_embeddings=False,
rope_theta=1000000.0,
use_sliding_window=False,
sliding_window=4096,
max_window_layers=80,
attention_dropout=0.0,
vision_config=None,
rope_scaling=None,
num_experts=4,
experts=None,
dof_config=None,
noise_scheduler=None,
dim_inputs=(1536,1536),
attention_moe=False,
mlp_moe=False,
**kwargs,
):
if isinstance(vision_config, dict):
self.vision_config = self.sub_configs["vision_config"](**vision_config)
elif vision_config is None:
self.vision_config = self.sub_configs["vision_config"]()
self.vocab_size = vocab_size
self.max_position_embeddings = max_position_embeddings
self.hidden_size = hidden_size
self.intermediate_size = intermediate_size
self.num_hidden_layers = num_hidden_layers
self.num_attention_heads = num_attention_heads
self.use_sliding_window = use_sliding_window
self.sliding_window = sliding_window
self.max_window_layers = max_window_layers
# for backward compatibility
if num_key_value_heads is None:
num_key_value_heads = num_attention_heads
self.num_key_value_heads = num_key_value_heads
self.hidden_act = hidden_act
self.initializer_range = initializer_range
self.rms_norm_eps = rms_norm_eps
self.use_cache = use_cache
self.rope_theta = rope_theta
self.attention_dropout = attention_dropout
self.rope_scaling = rope_scaling
self.num_experts = num_experts
self.experts = experts
self.dof_config = dof_config
self.noise_scheduler = noise_scheduler
self.dim_inputs = tuple(dim_inputs)
self.attention_moe = attention_moe
self.mlp_moe = mlp_moe
# Validate the correctness of rotary position embeddings parameters
# BC: if there is a 'type' field, move it to 'rope_type'.
# and change type from 'mrope' to 'default' because `mrope` does defeault RoPE calculations
# one can set it to "linear"/"dynamic" etc. to have scaled RoPE
# TODO: @raushan update config in the hub
if self.rope_scaling is not None and "type" in self.rope_scaling:
if self.rope_scaling["type"] == "mrope":
self.rope_scaling["type"] = "default"
self.rope_scaling["rope_type"] = self.rope_scaling["type"]
rope_config_validation(self, ignore_keys={"mrope_section"})
super().__init__(tie_word_embeddings=tie_word_embeddings, **kwargs)
__all__ = ["Qwen2_5_VLConfig"]
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff