Upload processor

Browse files

Files changed (12) hide show

.gitattributes +1 -0
added_tokens.json +24 -0
chat_template.jinja +7 -0
merges.txt +0 -0
modeling_yangjian.py +709 -0
preprocessor_config.json +32 -0
processor_config.json +6 -0
special_tokens_map.json +31 -0
tokenizer.json +3 -0
tokenizer_config.json +215 -0
video_preprocessor_config.json +46 -0
vocab.json +0 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

added_tokens.json ADDED Viewed

	@@ -0,0 +1,24 @@

+{
+  "</tool_call>": 151658,
+  "<tool_call>": 151657,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

chat_template.jinja ADDED Viewed

	@@ -0,0 +1,7 @@

+{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system
+You are a helpful assistant.<|im_end|>
+{% endif %}<|im_start|>{{ message['role'] }}
+{% if message['content'] is string %}{{ message['content'] }}<|im_end|>
+{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>
+{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant
+{% endif %}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

modeling_yangjian.py ADDED Viewed

	@@ -0,0 +1,709 @@

+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+import numpy as np
+from typing import Any, Callable, Optional, Union
+from transformers import Qwen2_5_VLForConditionalGeneration
+from transformers.models.qwen2_5_vl.modeling_qwen2_5_vl import (
+    Qwen2_5_VisionTransformerPretrainedModel,
+    Qwen2_5_VLModel,
+    Qwen2_5_VLModelOutputWithPast,
+    is_torchdynamo_compiling,
+    Qwen2RMSNorm,
+    Qwen2_5_VLMLP,
+    eager_attention_forward,
+    ALL_ATTENTION_FUNCTIONS
+)
+from transformers.image_utils import ImageInput
+from transformers.tokenization_utils import TextInput, PreTokenizedInput
+from transformers.video_utils import VideoInput
+from transformers.feature_extraction_utils import BatchFeature
+from transformers import Qwen2_5_VLProcessor, Qwen2_5_VLConfig
+from transformers.models.qwen2_5_vl.processing_qwen2_5_vl import Qwen2_5_VLProcessorKwargs
+class YangJianConfig(Qwen2_5_VLConfig):
+    model_type = "yangjian"
+    def __init__(self, **kwargs):
+        super().__init__(**kwargs)
+        self.vision_config.compare_token_size = 100
+        self.architectures = ["YangJianVLForConditionalGeneration"]
+class YangJianProcessor(Qwen2_5_VLProcessor):
+    def __init__(self, image_processor=None, tokenizer=None, video_processor=None, chat_template=None, **kwargs):
+        super().__init__(image_processor, tokenizer, video_processor, chat_template, **kwargs)
+        self.compare_token_size = 100 if "compare_token_size" not in kwargs else kwargs["compare_token_size"]
+    def __call__(
+        self,
+        images: ImageInput = None,
+        text: Union[TextInput, PreTokenizedInput, list[TextInput], list[PreTokenizedInput]] = None,
+        videos: VideoInput = None,
+        **kwargs,
+    ) -> BatchFeature:
+        """
+        Main method to prepare for the model one or several sequences(s) and image(s). This method forwards the `text`
+        and `kwargs` arguments to Qwen2TokenizerFast's [`~Qwen2TokenizerFast.__call__`] if `text` is not `None` to encode
+        the text. To prepare the vision inputs, this method forwards the `vision_infos` and `kwrags` arguments to
+        Qwen2VLImageProcessor's [`~Qwen2VLImageProcessor.__call__`] if `vision_infos` is not `None`.
+        Args:
+            images (`PIL.Image.Image`, `np.ndarray`, `torch.Tensor`, `list[PIL.Image.Image]`, `list[np.ndarray]`, `list[torch.Tensor]`):
+                The image or batch of images to be prepared. Each image can be a PIL image, NumPy array or PyTorch
+                tensor. Both channels-first and channels-last formats are supported.
+            text (`str`, `list[str]`, `list[list[str]]`):
+                The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings
+                (pretokenized string). If the sequences are provided as list of strings (pretokenized), you must set
+                `is_split_into_words=True` (to lift the ambiguity with a batch of sequences).
+            videos (`np.ndarray`, `torch.Tensor`, `list[np.ndarray]`, `list[torch.Tensor]`):
+                The image or batch of videos to be prepared. Each video can be a 4D NumPy array or PyTorch
+                tensor, or a nested list of 3D frames. Both channels-first and channels-last formats are supported.
+            return_tensors (`str` or [`~utils.TensorType`], *optional*):
+                If set, will return tensors of a particular framework. Acceptable values are:
+                - `'tf'`: Return TensorFlow `tf.constant` objects.
+                - `'pt'`: Return PyTorch `torch.Tensor` objects.
+                - `'np'`: Return NumPy `np.ndarray` objects.
+                - `'jax'`: Return JAX `jnp.ndarray` objects.
+        Returns:
+            [`BatchFeature`]: A [`BatchFeature`] with the following fields:
+            - **input_ids** -- List of token ids to be fed to a model. Returned when `text` is not `None`.
+            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
+              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
+              `None`).
+            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
+            - **pixel_values_videos** -- Pixel values of videos to be fed to a model. Returned when `videos` is not `None`.
+            - **image_grid_thw** -- List of image 3D grid in LLM. Returned when `images` is not `None`.
+            - **video_grid_thw** -- List of video 3D grid in LLM. Returned when `videos` is not `None`.
+            - **second_per_grid_ts** -- List of video seconds per time grid. Returned when `videos` is not `None`.
+        """
+        output_kwargs = self._merge_kwargs(
+            Qwen2_5_VLProcessorKwargs,
+            tokenizer_init_kwargs=self.tokenizer.init_kwargs,
+            **kwargs,
+        )
+        image_inputs = videos_inputs = {}
+        if images is not None:
+            image_inputs = self.image_processor(images=images, **output_kwargs["images_kwargs"])
+            image_grid_thw = image_inputs["image_grid_thw"]
+        if videos is not None:
+            fps = output_kwargs["videos_kwargs"].get("fps", 2.0)
+            videos_inputs = self.video_processor(videos=videos, **output_kwargs["videos_kwargs"])
+            video_grid_thw = videos_inputs["video_grid_thw"]
+            if isinstance(fps, (int, float)):
+                second_per_grid_ts = [self.video_processor.temporal_patch_size / fps] * len(video_grid_thw)
+            elif hasattr(fps, "__len__") and len(fps) == len(video_grid_thw):
+                second_per_grid_ts = [self.video_processor.temporal_patch_size / tmp for tmp in fps]
+            else:
+                raise ValueError(
+                    f"The length of fps ({len(fps) if hasattr(fps, '__len__') else fps}) must be equal to the length of video_grid_thw ({len(video_grid_thw)}) or fps should be a single number."
+                )
+            videos_inputs.update({"second_per_grid_ts": second_per_grid_ts})
+        if not isinstance(text, list):
+            text = [text]
+        text = text.copy()  # below lines change text in-place
+        if images is not None:
+            merge_length = self.image_processor.merge_size**2
+            index = 0
+            for i in range(len(text)):
+                while self.image_token in text[i]:
+                    num_image_tokens = image_grid_thw[index].prod() // merge_length
+                    text[i] = text[i].replace(self.image_token, "<|placeholder|>" * (num_image_tokens + self.compare_token_size), 1)
+                    index += 1
+                text[i] = text[i].replace("<|placeholder|>", self.image_token)
+        if videos is not None:
+            merge_length = self.video_processor.merge_size**2
+            index = 0
+            for i in range(len(text)):
+                while self.video_token in text[i]:
+                    num_video_tokens = video_grid_thw[index].prod() // merge_length
+                    text[i] = text[i].replace(self.video_token, "<|placeholder|>" * num_video_tokens, 1)
+                    index += 1
+                text[i] = text[i].replace("<|placeholder|>", self.video_token)
+        return_tensors = output_kwargs["text_kwargs"].pop("return_tensors", None)
+        return_mm_token_type_ids = output_kwargs["text_kwargs"].pop("return_mm_token_type_ids", None)
+        text_inputs = self.tokenizer(text, **output_kwargs["text_kwargs"])
+        self._check_special_mm_tokens(text, text_inputs, modalities=["image", "video"])
+        if return_mm_token_type_ids:
+            array_ids = np.array(text_inputs["input_ids"])
+            mm_token_type_ids = np.zeros_like(text_inputs["input_ids"])
+            mm_token_type_ids[array_ids == self.image_token_id] = 1
+            text_inputs["mm_token_type_ids"] = mm_token_type_ids.tolist()
+        return BatchFeature(data={**text_inputs, **image_inputs, **videos_inputs}, tensor_type=return_tensors)
+class OptimizedCrossAttention(nn.Module):
+    """
+    仿照 Qwen2_5_VLVisionAttention 结构的优化 Cross Attention
+    """
+    def __init__(self, config, is_cross_attention=False):
+        super().__init__()
+        self.config = config
+        self.dim = config.hidden_size
+        self.num_heads = config.num_heads
+        self.head_dim = self.dim // self.num_heads
+        self.num_key_value_groups = 1  # 对于 cross attention，通常设为 1
+        self.scaling = self.head_dim**-0.5
+        self.attention_dropout = 0.0
+        self.is_causal = False  # cross attention 不需要因果掩码
+        self.is_cross_attention = is_cross_attention
+        if is_cross_attention:
+            # Cross attention: Q 来自一个序列，K、V 来自另一个序列
+            self.q_proj = nn.Linear(self.dim, self.dim, bias=True)
+            self.kv = nn.Linear(self.dim, self.dim * 2, bias=True)  # 融合 K、V
+        else:
+            # Self attention: Q、K、V 来自同一个序列
+            self.qkv = nn.Linear(self.dim, self.dim * 3, bias=True)  # 融合 Q、K、V
+        self.proj = nn.Linear(self.dim, self.dim, bias=True)
+    def forward(
+        self,
+        query_states: torch.Tensor,
+        key_value_states: Optional[torch.Tensor] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        **kwargs,
+    ) -> torch.Tensor:
+        """
+        Args:
+            query_states: [seq_len_q, hidden_size] 或 [batch_size, seq_len_q, hidden_size]
+            key_value_states: [seq_len_kv, hidden_size] 或 [batch_size, seq_len_kv, hidden_size]
+                             如果为 None，则执行 self attention
+        """
+        # 处理输入维度
+        if query_states.dim() == 2:
+            query_states = query_states.unsqueeze(0)  # [1, seq_len_q, hidden_size]
+            squeeze_output = True
+        else:
+            squeeze_output = False
+        batch_size, seq_len_q, _ = query_states.shape
+        if self.is_cross_attention and key_value_states is not None:
+            # Cross Attention
+            if key_value_states.dim() == 2:
+                key_value_states = key_value_states.unsqueeze(0)  # [1, seq_len_kv, hidden_size]
+            # 计算 Q
+            q = self.q_proj(query_states)  # [batch_size, seq_len_q, hidden_size]
+            # 计算 K、V（融合计算）
+            kv = self.kv(key_value_states)  # [batch_size, seq_len_kv, hidden_size * 2]
+            seq_len_kv = kv.shape[1]
+            # 分离 K、V
+            k, v = kv.reshape(batch_size, seq_len_kv, 2, self.num_heads, self.head_dim).permute(2, 0, 3, 1, 4).unbind(0)
+            # k, v: [batch_size, num_heads, seq_len_kv, head_dim]
+            # 重塑 Q
+            q = q.reshape(batch_size, seq_len_q, self.num_heads, self.head_dim).transpose(1, 2)
+            # q: [batch_size, num_heads, seq_len_q, head_dim]
+        else:
+            # Self Attention
+            if key_value_states is None:
+                key_value_states = query_states
+            # 融合计算 Q、K、V
+            qkv = self.qkv(query_states)  # [batch_size, seq_len, hidden_size * 3]
+            # 分离 Q、K、V
+            q, k, v = qkv.reshape(batch_size, seq_len_q, 3, self.num_heads, self.head_dim).permute(2, 0, 3, 1, 4).unbind(0)
+            # q, k, v: [batch_size, num_heads, seq_len, head_dim]
+        # 选择 attention 实现
+        attention_interface: Callable = eager_attention_forward
+        if hasattr(self.config, '_attn_implementation') and self.config._attn_implementation != "eager":
+            attention_interface = ALL_ATTENTION_FUNCTIONS[self.config._attn_implementation]
+        # 执行 attention 计算
+        attn_output, _ = attention_interface(
+            self,
+            q,
+            k,
+            v,
+            attention_mask=attention_mask,
+            dropout=0.0 if not self.training else self.attention_dropout,
+            scaling=self.scaling,
+            is_causal=False,
+            **kwargs,
+        )
+        # 重塑输出
+        attn_output = attn_output.transpose(1, 2).contiguous()  # [batch_size, seq_len_q, num_heads, head_dim]
+        attn_output = attn_output.reshape(batch_size, seq_len_q, self.dim)  # [batch_size, seq_len_q, hidden_size]
+        # 输出投影
+        attn_output = self.proj(attn_output)
+        # 如果输入是 2D，则输出也应该是 2D
+        if squeeze_output:
+            attn_output = attn_output.squeeze(0)  # [seq_len_q, hidden_size]
+        return attn_output
+class YangJianCompareVisualEncoder(nn.Module):
+    def __init__(self, config):
+        super().__init__()
+        self.config = config
+        self.hidden_size = config.hidden_size
+        self.token_size = 100  * (config.spatial_merge_size**2) if "compare_token_size" not in config else config.compare_token_size  * (config.spatial_merge_size**2)
+        # Encoder 部分：双向图像特征交互
+        # 第一个cross attention: previous attend to current
+        self.encoder_cross_attn1 = OptimizedCrossAttention(config, is_cross_attention=True)
+        # 第二个cross attention: current attend to previous
+        self.encoder_cross_attn2 = OptimizedCrossAttention(config, is_cross_attention=True)
+        self.encoder_norm1 = Qwen2RMSNorm(self.hidden_size, eps=1e-6)
+        self.encoder_norm2 = Qwen2RMSNorm(self.hidden_size, eps=1e-6)
+        self.encoder_norm3 = Qwen2RMSNorm(self.hidden_size, eps=1e-6)
+        self.encoder_norm4 = Qwen2RMSNorm(self.hidden_size, eps=1e-6)
+        self.encoder_mlp1 = Qwen2_5_VLMLP(config)
+        self.encoder_mlp2 = Qwen2_5_VLMLP(config)
+        # Decoder 部分：Query 与编码特征交互
+        # 可学习的 Query Embeddings
+        self.query_embeddings = nn.Parameter(
+            torch.randn(self.token_size, self.hidden_size) * 0.02
+        )
+        # 只保留 Cross Attention for queries to attend to encoded features
+        self.decoder_cross_attn = OptimizedCrossAttention(config, is_cross_attention=True)
+        self.decoder_norm1 = Qwen2RMSNorm(self.hidden_size, eps=1e-6)
+        self.decoder_norm2 = Qwen2RMSNorm(self.hidden_size, eps=1e-6)
+        self.decoder_mlp = Qwen2_5_VLMLP(config)
+    def _ensure_device_dtype_consistency(self, target_tensor):
+        """
+        确保所有模块组件都在目标张量的设备上并使用相同的数据类型
+        """
+        device = target_tensor.device
+        dtype = target_tensor.dtype
+        # 移动 attention 模块到正确设备
+        self.encoder_cross_attn1 = self.encoder_cross_attn1.to(device=device, dtype=dtype)
+        self.encoder_cross_attn2 = self.encoder_cross_attn2.to(device=device, dtype=dtype)
+        self.decoder_cross_attn = self.decoder_cross_attn.to(device=device, dtype=dtype)
+        # 移动 norm 层到正确设备
+        self.encoder_norm1 = self.encoder_norm1.to(device=device, dtype=dtype)
+        self.encoder_norm2 = self.encoder_norm2.to(device=device, dtype=dtype)
+        self.encoder_norm3 = self.encoder_norm3.to(device=device, dtype=dtype)
+        self.encoder_norm4 = self.encoder_norm4.to(device=device, dtype=dtype)
+        self.decoder_norm1 = self.decoder_norm1.to(device=device, dtype=dtype)
+        self.decoder_norm2 = self.decoder_norm2.to(device=device, dtype=dtype)
+        # 移动 MLP 到正确设备
+        self.encoder_mlp1 = self.encoder_mlp1.to(device=device, dtype=dtype)
+        self.encoder_mlp2 = self.encoder_mlp2.to(device=device, dtype=dtype)
+        self.decoder_mlp = self.decoder_mlp.to(device=device, dtype=dtype)
+    def forward(self, images_hidden_states: list) -> list:
+        """
+        Args:
+            images_hidden_states: List of tensor, each tensor has shape [seq_len, hidden_size]
+        Returns:
+            List of compare visual embeddings, each has shape [token_size, hidden_size]
+        """
+        if not images_hidden_states:
+            return []
+        # 确保所有组件的设备和数据类型一致
+        self._ensure_device_dtype_consistency(images_hidden_states[0])
+        compare_visual_embeds = []
+        for i in range(len(images_hidden_states)):
+            current_hidden_state = images_hidden_states[i]  # [seq_len_current, hidden_size]
+            previous_hidden_state = images_hidden_states[i-1] if i > 0 else current_hidden_state  # [seq_len_prev, hidden_size]
+            # Encoder 部分：双向图像特征交互
+            encoded_features = self._encoder_forward(current_hidden_state, previous_hidden_state)
+            # Decoder 部分：Query 与编码特征交互
+            compare_visual_embed = self._decoder_forward(encoded_features)
+            compare_visual_embeds.append(compare_visual_embed)
+        return compare_visual_embeds
+    def _encoder_forward(self, current_features, previous_features):
+        """
+        Encoder: 双向图像特征交互
+        1. previous attend to current
+        2. current attend to previous
+        """
+        # 确保数据类型和设备一致
+        device = current_features.device
+        dtype = current_features.dtype
+        previous_features = previous_features.to(device=device, dtype=dtype)
+        # 第一步：previous attend to current
+        residual = previous_features
+        # Layer norm
+        previous_normed = self.encoder_norm1(previous_features)
+        current_normed1 = self.encoder_norm1(current_features)
+        # Cross attention: previous attend to current
+        cross_attn_output1 = self.encoder_cross_attn1(
+            query_states=previous_normed,
+            key_value_states=current_normed1
+        )
+        # Residual connection
+        previous_features = residual + cross_attn_output1
+        # MLP for previous features
+        residual = previous_features
+        mlp_input1 = self.encoder_norm2(previous_features)
+        mlp_output1 = self.encoder_mlp1(mlp_input1)
+        previous_features = residual + mlp_output1
+        # 第二步：current attend to previous (enhanced)
+        residual = current_features
+        # Layer norm
+        current_normed2 = self.encoder_norm3(current_features)
+        previous_normed2 = self.encoder_norm3(previous_features)  # 使用增强后的 previous features
+        # Cross attention: current attend to previous
+        cross_attn_output2 = self.encoder_cross_attn2(
+            query_states=current_normed2,
+            key_value_states=previous_normed2
+        )
+        # Residual connection
+        current_features = residual + cross_attn_output2
+        # MLP for current features
+        residual = current_features
+        mlp_input2 = self.encoder_norm4(current_features)
+        mlp_output2 = self.encoder_mlp2(mlp_input2)
+        current_features = residual + mlp_output2
+        return current_features
+    def _decoder_forward(self, encoded_features):
+        """
+        Decoder: Query 与编码特征交互（仅使用 cross attention）
+        """
+        # 获取设备和数据类型
+        device = encoded_features.device
+        dtype = encoded_features.dtype
+        # 初始化 queries 并确保设备和数据类型一致
+        queries = self.query_embeddings.to(device=device, dtype=dtype)
+        # Cross attention: queries attend to encoded features
+        residual = queries
+        queries_normed = self.decoder_norm1(queries)
+        encoded_normed = self.decoder_norm1(encoded_features)
+        cross_attn_output = self.decoder_cross_attn(
+            query_states=queries_normed,
+            key_value_states=encoded_normed
+        )
+        queries = residual + cross_attn_output
+        # MLP
+        residual = queries
+        mlp_input = self.decoder_norm2(queries)
+        mlp_output = self.decoder_mlp(mlp_input)
+        queries = residual + mlp_output
+        return queries  # [token_size, hidden_size]
+# 先把组件继承出来方便修改
+class YangJianVisionTransformerPretrainedModel(Qwen2_5_VisionTransformerPretrainedModel):
+    def __init__(self, config, *inputs, **kwargs) -> None:
+        super().__init__(config, *inputs, **kwargs)
+        self.compare_visual_encoder = YangJianCompareVisualEncoder(config)
+    def forward(self, hidden_states: torch.Tensor, grid_thw: torch.Tensor, **kwargs) -> torch.Tensor:
+        """
+        Args:
+            hidden_states (`torch.Tensor` of shape `(seq_len, hidden_size)`):
+                The final hidden states of the model.
+            grid_thw (`torch.Tensor` of shape `(num_images_or_videos, 3)`):
+                The temporal, height and width of feature shape of each image in LLM.
+        Returns:
+            `torch.Tensor`: hidden_states, compare_visual_embeds.
+        """
+        hidden_states = self.patch_embed(hidden_states)
+        rotary_pos_emb = self.rot_pos_emb(grid_thw)
+        window_index, cu_window_seqlens = self.get_window_index(grid_thw)
+        cu_window_seqlens = torch.tensor(
+            cu_window_seqlens,
+            device=hidden_states.device,
+            dtype=grid_thw.dtype if torch.jit.is_tracing() else torch.int32,
+        )
+        cu_window_seqlens = torch.unique_consecutive(cu_window_seqlens)
+        seq_len, _ = hidden_states.size()
+        hidden_states = hidden_states.reshape(seq_len // self.spatial_merge_unit, self.spatial_merge_unit, -1)
+        hidden_states = hidden_states[window_index, :, :]
+        hidden_states = hidden_states.reshape(seq_len, -1)
+        rotary_pos_emb = rotary_pos_emb.reshape(seq_len // self.spatial_merge_unit, self.spatial_merge_unit, -1)
+        rotary_pos_emb = rotary_pos_emb[window_index, :, :]
+        rotary_pos_emb = rotary_pos_emb.reshape(seq_len, -1)
+        emb = torch.cat((rotary_pos_emb, rotary_pos_emb), dim=-1)
+        position_embeddings = (emb.cos(), emb.sin())
+        cu_seqlens = torch.repeat_interleave(grid_thw[:, 1] * grid_thw[:, 2], grid_thw[:, 0]).cumsum(
+            dim=0,
+            # Select dtype based on the following factors:
+            #  - FA2 requires that cu_seqlens_q must have dtype int32
+            #  - torch.onnx.export requires that cu_seqlens_q must have same dtype as grid_thw
+            # See https://github.com/huggingface/transformers/pull/34852 for more information
+            dtype=grid_thw.dtype if torch.jit.is_tracing() else torch.int32,
+        )
+        cu_seqlens = F.pad(cu_seqlens, (1, 0), value=0)
+        for layer_num, blk in enumerate(self.blocks):
+            if layer_num in self.fullatt_block_indexes:
+                cu_seqlens_now = cu_seqlens
+            else:
+                cu_seqlens_now = cu_window_seqlens
+            attention_mask = self._prepare_attention_mask(hidden_states, cu_seqlens_now)
+            hidden_states = blk(
+                hidden_states,
+                cu_seqlens=cu_seqlens_now,
+                position_embeddings=position_embeddings,
+                attention_mask=attention_mask,
+                **kwargs,
+            )
+        split_sizes = grid_thw.prod(-1).tolist()
+        splited_hidden_states_before_merger = torch.split(hidden_states, split_sizes)
+        compare_visual_embeds = self.compare_visual_encoder(splited_hidden_states_before_merger)
+        # compare_visual_embeds = self.merger(compare_visual_embeds)
+        for i, embeds in enumerate(compare_visual_embeds):
+            compare_visual_embeds[i] = self.merger(embeds)
+        hidden_states = self.merger(hidden_states)
+        reverse_indices = torch.argsort(window_index)
+        hidden_states = hidden_states[reverse_indices, :]
+        return hidden_states, compare_visual_embeds
+class YangJianVLModel(Qwen2_5_VLModel):
+    def __init__(self, config):
+        super().__init__(config)
+        self.visual = YangJianVisionTransformerPretrainedModel._from_config(config.vision_config)
+        # self.learnable_image_embeddings = nn.Parameter(
+        #     torch.randn(100, config.hidden_size) * 0.02  # 使用小的初始化值
+        # )
+    def get_image_features(self, pixel_values: torch.FloatTensor, image_grid_thw: Optional[torch.LongTensor] = None):
+        """
+        Encodes images into continuous embeddings that can be forwarded to the language model.
+        Args:
+            pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
+                The tensors corresponding to the input images.
+            image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
+                The temporal, height and width of feature shape of each image in LLM.
+        """
+        pixel_values = pixel_values.type(self.visual.dtype)
+        image_embeds, compare_visual_embeds = self.visual(pixel_values, grid_thw=image_grid_thw)
+        # 每个图像添加了对比感知token
+        split_sizes = (image_grid_thw.prod(-1) // self.visual.spatial_merge_size**2).tolist()
+        image_embeds = torch.split(image_embeds, split_sizes)
+        # 将图像嵌入和对比视觉嵌入拼接
+        enhanced_image_embeds = []
+        for i, embeds in enumerate(image_embeds):
+            # 确保 compare_visual_embeds[i] 与 embeds 在相同设备和数据类型
+            compare_embed = compare_visual_embeds[i].to(device=embeds.device, dtype=embeds.dtype)
+            enhanced_embeds = torch.cat([embeds, compare_embed], dim=0)
+            enhanced_image_embeds.append(enhanced_embeds)
+        # image_embeds = torch.cat(enhanced_image_embeds, dim=0)
+        return enhanced_image_embeds
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[list[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+        pixel_values: Optional[torch.Tensor] = None,
+        pixel_values_videos: Optional[torch.FloatTensor] = None,
+        image_grid_thw: Optional[torch.LongTensor] = None,
+        video_grid_thw: Optional[torch.LongTensor] = None,
+        rope_deltas: Optional[torch.LongTensor] = None,
+        cache_position: Optional[torch.LongTensor] = None,
+        second_per_grid_ts: Optional[torch.Tensor] = None,
+        **kwargs,
+    ) -> Union[tuple, Qwen2_5_VLModelOutputWithPast]:
+        r"""
+        pixel_values_videos (`torch.FloatTensor` of shape `(seq_length, num_channels * temporal_size * image_size * image_size)):
+            The tensors corresponding to the input videos. Pixel values can be obtained using
+            [`AutoImageProcessor`]. See [`Qwen2VLImageProcessor.__call__`] for details. [`Qwen2_5_VLProcessor`] uses
+            [`Qwen2VLImageProcessor`] for processing videos.
+        image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
+            The temporal, height and width of feature shape of each image in LLM.
+        video_grid_thw (`torch.LongTensor` of shape `(num_videos, 3)`, *optional*):
+            The temporal, height and width of feature shape of each video in LLM.
+        rope_deltas (`torch.LongTensor` of shape `(batch_size, )`, *optional*):
+            The rope index difference between sequence length and multimodal rope.
+        second_per_grid_ts (`torch.Tensor` of shape `(num_videos)`, *optional*):
+            The time interval (in seconds) for each grid along the temporal dimension in the 3D position IDs.
+        """
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        if inputs_embeds is None:
+            inputs_embeds = self.get_input_embeddings()(input_ids)
+            if pixel_values is not None:
+                image_embeds = self.get_image_features(pixel_values, image_grid_thw)
+                # # 为每个图像添加 100 个可学习的 embedding
+                # learnable_embeddings = self.learnable_image_embeddings.to(image_embeds[0].device, image_embeds[0].dtype)
+                # enhanced_image_embeds = []
+                # for i, embeds in enumerate(image_embeds):
+                #     # 为每个图像添加 100 个可学习的 embedding
+                #     enhanced_embeds = torch.cat([embeds, learnable_embeddings], dim=0)
+                #     enhanced_image_embeds.append(enhanced_embeds)
+                image_embeds = torch.cat(image_embeds, dim=0)
+                n_image_tokens = (input_ids == self.config.image_token_id).sum()
+                n_image_features = image_embeds.shape[0]
+                if not is_torchdynamo_compiling() and n_image_tokens != n_image_features:
+                    raise ValueError(
+                        f"Image features and image tokens do not match: tokens: {n_image_tokens}, features {n_image_features}"
+                    )
+                mask = input_ids == self.config.image_token_id
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                image_mask = mask_expanded.to(inputs_embeds.device)
+                image_embeds = image_embeds.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(image_mask, image_embeds)
+            if pixel_values_videos is not None:
+                video_embeds = self.get_video_features(pixel_values_videos, video_grid_thw)
+                video_embeds = torch.cat(video_embeds, dim=0)
+                n_video_tokens = (input_ids == self.config.video_token_id).sum()
+                n_video_features = video_embeds.shape[0]
+                if not is_torchdynamo_compiling() and n_video_tokens != n_video_features:
+                    raise ValueError(
+                        f"Video features and video tokens do not match: tokens: {n_video_tokens}, features {n_video_features}"
+                    )
+                mask = input_ids == self.config.video_token_id
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                video_mask = mask_expanded.to(inputs_embeds.device)
+                video_embeds = video_embeds.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(video_mask, video_embeds)
+        if position_ids is None:
+            attention_mask_tensor = (
+                attention_mask if not isinstance(attention_mask, dict) else attention_mask["full_attention"]
+            )
+            if attention_mask_tensor is not None and attention_mask_tensor.ndim == 4:
+                attention_mask_tensor = torch.diagonal(attention_mask_tensor[:, 0], dim1=1, dim2=2)
+                attention_mask_tensor = attention_mask_tensor / torch.finfo(attention_mask_tensor.dtype).min
+                attention_mask_tensor = (1.0 - attention_mask_tensor).int()
+            # Calculate RoPE index once per generation in the pre-fill stage only.
+            # When compiling, we can't check tensor values thus we check only input length
+            # It is safe to assume that `length!=1` means we're in pre-fill because compiled
+            # models currently cannot do asssisted decoding
+            prefill_compiled_stage = is_torchdynamo_compiling() and (
+                (input_ids is not None and input_ids.shape[1] != 1)
+                or (inputs_embeds is not None and inputs_embeds.shape[1] != 1)
+            )
+            prefill_noncompiled_stage = not is_torchdynamo_compiling() and (
+                (cache_position is not None and cache_position[0] == 0)
+                or (past_key_values is None or past_key_values.get_seq_length() == 0)
+            )
+            if (prefill_compiled_stage or prefill_noncompiled_stage) or self.rope_deltas is None:
+                position_ids, rope_deltas = self.get_rope_index(
+                    input_ids,
+                    image_grid_thw,
+                    video_grid_thw,
+                    second_per_grid_ts=second_per_grid_ts,
+                    attention_mask=attention_mask_tensor,
+                )
+                self.rope_deltas = rope_deltas
+            # then use the prev pre-calculated rope-deltas to get the correct position ids
+            else:
+                batch_size, seq_length, _ = inputs_embeds.shape
+                delta = (
+                    (cache_position[0] + self.rope_deltas).to(inputs_embeds.device)
+                    if cache_position is not None
+                    else 0
+                )
+                position_ids = torch.arange(seq_length, device=inputs_embeds.device)
+                position_ids = position_ids.view(1, -1).expand(batch_size, -1)
+                if cache_position is not None:  # otherwise `deltas` is an int `0`
+                    delta = delta.repeat_interleave(batch_size // delta.shape[0], dim=0)
+                position_ids = position_ids.add(delta)
+                position_ids = position_ids.unsqueeze(0).expand(3, -1, -1)
+        outputs = self.language_model(
+            input_ids=None,
+            position_ids=position_ids,
+            attention_mask=attention_mask,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=True,
+            cache_position=cache_position,
+            **kwargs,
+        )
+        output = Qwen2_5_VLModelOutputWithPast(
+            last_hidden_state=outputs.last_hidden_state,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+            rope_deltas=self.rope_deltas,
+        )
+        return output if return_dict else output.to_tuple()
+class YangJianVLForConditionalGeneration(Qwen2_5_VLForConditionalGeneration):
+    config_class = YangJianConfig
+    def __init__(self, config):
+        super().__init__(config)
+        self.model = YangJianVLModel(config)

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,32 @@

+{
+  "auto_map": {
+    "AutoProcessor": "modeling_yangjian.YangJianProcessor"
+  },
+  "do_convert_rgb": true,
+  "do_normalize": true,
+  "do_rescale": true,
+  "do_resize": true,
+  "image_mean": [
+    0.48145466,
+    0.4578275,
+    0.40821073
+  ],
+  "image_processor_type": "Qwen2VLImageProcessor",
+  "image_std": [
+    0.26862954,
+    0.26130258,
+    0.27577711
+  ],
+  "max_pixels": 12845056,
+  "merge_size": 2,
+  "min_pixels": 3136,
+  "patch_size": 14,
+  "processor_class": "YangJianProcessor",
+  "resample": 3,
+  "rescale_factor": 0.00392156862745098,
+  "size": {
+    "longest_edge": 12845056,
+    "shortest_edge": 3136
+  },
+  "temporal_patch_size": 2
+}

processor_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "auto_map": {
+    "AutoProcessor": "modeling_yangjian.YangJianProcessor"
+  },
+  "processor_class": "YangJianProcessor"
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>"
+  ],
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ba0c439f7be467bf47d12a7e6f9adc6116201056fc60c67f431c679b7c16afc8
+size 11422064

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,215 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>"
+  ],
+  "auto_map": {
+    "AutoProcessor": "modeling_yangjian.YangJianProcessor"
+  },
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "max_length": null,
+  "model_max_length": 131072,
+  "pad_to_multiple_of": null,
+  "pad_token": "<|endoftext|>",
+  "pad_token_type_id": 0,
+  "padding_side": "right",
+  "processor_class": "YangJianProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null
+}

video_preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,46 @@

+{
+  "auto_map": {
+    "AutoProcessor": "modeling_yangjian.YangJianProcessor"
+  },
+  "crop_size": null,
+  "data_format": "channels_first",
+  "default_to_square": true,
+  "device": null,
+  "do_center_crop": null,
+  "do_convert_rgb": true,
+  "do_normalize": true,
+  "do_pad": null,
+  "do_rescale": true,
+  "do_resize": true,
+  "do_sample_frames": false,
+  "fps": null,
+  "image_mean": [
+    0.48145466,
+    0.4578275,
+    0.40821073
+  ],
+  "image_std": [
+    0.26862954,
+    0.26130258,
+    0.27577711
+  ],
+  "input_data_format": null,
+  "max_frames": 768,
+  "max_pixels": 12845056,
+  "merge_size": 2,
+  "min_frames": 4,
+  "min_pixels": 3136,
+  "num_frames": null,
+  "patch_size": 14,
+  "processor_class": "YangJianProcessor",
+  "resample": 3,
+  "rescale_factor": 0.00392156862745098,
+  "size": {
+    "longest_edge": 12845056,
+    "shortest_edge": 3136
+  },
+  "size_divisor": null,
+  "temporal_patch_size": 2,
+  "video_metadata": null,
+  "video_processor_type": "Qwen2VLVideoProcessor"
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff