xvla log fix

Merge branch 'chore/bump_transformers_v5' into ci/add_hf_account
fix(ci): temp fix for pi0 rtc test
2026-06-11 05:39:49 +00:00 · 2026-02-26 09:55:25 +00:00 · 2026-02-25 11:56:52 +01:00 · 2026-02-25 11:42:04 +01:00 · 2026-02-24 21:16:37 +03:00 · 2026-02-24 17:50:48 +01:00
13 changed files with 59 additions and 53 deletions
@@ -61,6 +61,7 @@ jobs:
      MUJOCO_GL: egl
      HF_HOME: /mnt/cache/.cache/huggingface
      HF_LEROBOT_HOME: /mnt/cache/.cache/huggingface/lerobot
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    steps:
      - uses: actions/checkout@v6
        with:
@@ -89,5 +90,10 @@ jobs:
      - name: Install lerobot with test extras
        run: uv sync --extra "test"

+      - name: Login to Hugging Face
+        run: |
+          uv run hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          uv run hf auth whoami
+
      - name: Run pytest
        run: uv run pytest tests -vv --maxfail=10
@@ -60,6 +60,7 @@ jobs:
      MUJOCO_GL: egl
      HF_HOME: /mnt/cache/.cache/huggingface
      HF_LEROBOT_HOME: /mnt/cache/.cache/huggingface/lerobot
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    steps:
      - uses: actions/checkout@v6
        with:
@@ -87,6 +88,11 @@ jobs:
      - name: Install lerobot with all extras
        run: uv sync --extra all # TODO(Steven): Make flash-attn optional

+      - name: Login to Hugging Face
+        run: |
+          uv run hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          uv run hf auth whoami
+
      - name: Run pytest (all extras)
        run: uv run pytest tests -vv --maxfail=10

@@ -162,6 +168,7 @@ jobs:
      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
      TORCH_HOME: /home/user_lerobot/.cache/torch
      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    container:
      image: ${{ needs.build-and-push-docker.outputs.image_tag }} # zizmor: ignore[unpinned-images]
      options: --gpus all --shm-size "16gb"
@@ -173,6 +180,10 @@ jobs:
        shell: bash
        working-directory: /lerobot
    steps:
+      - name: Login to Hugging Face
+        run: |
+          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          hf auth whoami
      - name: Run pytest on GPU
        run: pytest tests -vv --maxfail=10
      - name: Run end-to-end tests
@@ -261,10 +261,15 @@ class Qwen2_5_VLMoEForAction(Qwen2_5_VLForConditionalGeneration):
    and optional LoRA fine-tuning support.
    """

-    _tied_weights_keys = ["lm_head.weight"]
+    _tied_weights_keys = {"lm_head.weight": "model.embed_tokens.weight"}
    config_class = Qwen2_5_VLConfig
    _no_split_modules = ["Qwen2_5_VLDecoderLayer_with_MoE", "Qwen2_5_VLVisionBlock"]

+    def init_weights(self):
+        if getattr(self.model, "language_model", None) is not None:
+            return
+        super().init_weights()
+
    @classmethod
    def from_pretrained(
        cls,
@@ -312,6 +317,11 @@ class Qwen2_5_VLMoEForAction(Qwen2_5_VLForConditionalGeneration):
            processor.action_processor = action_tokenizer
        else:
            action_tokenizer = None
+
+        # add pad_token_id to config
+        config.pad_token_id = processor.tokenizer.pad_token_id
+        config.text_config.pad_token_id = processor.tokenizer.pad_token_id
+
        # Initialize model with configuration and processor
        model = cls(config, processor=processor, action_tokenizer=action_tokenizer, **kwargs)

@@ -21,6 +21,7 @@ class Qwen2_5_VLVisionConfig(PretrainedConfig):
        window_size=112,
        out_hidden_size=3584,
        fullatt_block_indexes=[7, 15, 23, 31],
+        initializer_range=0.02,
        **kwargs,
    ):
        super().__init__(**kwargs)
@@ -38,6 +39,7 @@ class Qwen2_5_VLVisionConfig(PretrainedConfig):
        self.window_size = window_size
        self.fullatt_block_indexes = fullatt_block_indexes
        self.out_hidden_size = out_hidden_size
+        self.initializer_range = initializer_range


 class Qwen2_5_VLConfig(PretrainedConfig):
@@ -602,19 +602,40 @@ class Qwen2_5_VisionTransformerPretrainedModel(Qwen2_5_VLPreTrainedModel):
        return hidden_states


+def _compute_default_rope_parameters_qwen2_5_vl(config, device=None):
+    """
+    compute default rope parameters for Qwen2_5_VL
+    """
+    base = config.text_config.rope_parameters["rope_theta"]
+    dim = config.hidden_size // config.num_attention_heads
+    inv_freq = 1.0 / (
+        base ** (torch.arange(0, dim, 2, dtype=torch.int64).to(device=device, dtype=torch.float) / dim)
+    )
+    return inv_freq, 1.0
+
+
 class Qwen2_5_VLRotaryEmbedding(nn.Module):
    def __init__(self, config: Qwen2_5_VLConfig, device=None):
        super().__init__()
        # BC: "rope_type" was originally "type"
        if hasattr(config, "rope_scaling") and config.rope_scaling is not None:
            self.rope_type = config.rope_scaling.get("rope_type", config.rope_scaling.get("type"))
+        elif hasattr(config, "rope_parameters") and config.rope_parameters is not None:
+            self.rope_type = config.rope_parameters.get("rope_type", "default")
        else:
            self.rope_type = "default"
        self.max_seq_len_cached = config.max_position_embeddings
        self.original_max_seq_len = config.max_position_embeddings

        self.config = config
-        self.rope_init_fn = ROPE_INIT_FUNCTIONS[self.rope_type]
+
+        if self.rope_type == "default":
+            self.rope_init_fn = _compute_default_rope_parameters_qwen2_5_vl
+            self.rope_kwargs = {}
+        else:
+            rope_type_key = "linear" if self.rope_type == "linear" else self.rope_type
+            self.rope_init_fn = ROPE_INIT_FUNCTIONS[rope_type_key]
+            self.rope_kwargs = {}

        inv_freq, self.attention_scaling = self.rope_init_fn(self.config, device)
        self.register_buffer("inv_freq", inv_freq, persistent=False)
@@ -144,7 +144,7 @@ def preprocesser_call(
    """
    # Process image inputs
    if images is not None and len(images) > 0:
-        image_inputs = processor.image_processor(images=images, videos=None, return_tensors=return_tensors)
+        image_inputs = processor.image_processor(images=images, return_tensors=return_tensors)
        image_grid_thw = image_inputs["image_grid_thw"]
    else:
        image_inputs = {}
@@ -152,7 +152,7 @@ def preprocesser_call(

    # Process video inputs
    if videos is not None:
-        videos_inputs = processor.image_processor(images=None, videos=videos, return_tensors=return_tensors)
+        videos_inputs = processor.image_processor(videos=videos, return_tensors=return_tensors)
        video_grid_thw = videos_inputs["video_grid_thw"]
    else:
        videos_inputs = {}
@@ -13,12 +13,9 @@
 import warnings

 from transformers.configuration_utils import PretrainedConfig
-from transformers.utils import logging

 """ Florence-2 configuration"""

-logger = logging.get_logger(__name__)
-

 class Florence2VisionConfig(PretrainedConfig):
    r"""
@@ -46,7 +46,6 @@ from transformers.utils import (
    add_start_docstrings_to_model_forward,
    is_flash_attn_2_available,
    is_flash_attn_greater_or_equal_2_10,
-    logging,
    replace_return_docstrings,
 )

@@ -57,8 +56,6 @@ if is_flash_attn_2_available():
    from flash_attn import flash_attn_func, flash_attn_varlen_func
    from flash_attn.bert_padding import index_first_axis, pad_input, unpad_input  # noqa

-logger = logging.get_logger(__name__)
-
 _CONFIG_FOR_DOC = "Florence2Config"


@@ -992,12 +989,6 @@ class Florence2FlashAttention2(Florence2Attention):
            else:
                target_dtype = self.q_proj.weight.dtype

-            logger.warning_once(
-                f"The input hidden states seems to be silently casted in float32, this might be related to"
-                f" the fact you have upcasted embedding or layer norm layers in float32. We will cast back the input in"
-                f" {target_dtype}."
-            )
-
            query_states = query_states.to(target_dtype)
            key_states = key_states.to(target_dtype)
            value_states = value_states.to(target_dtype)
@@ -1135,11 +1126,6 @@ class Florence2SdpaAttention(Florence2Attention):
    ) -> tuple[torch.Tensor, torch.Tensor | None, tuple[torch.Tensor] | None]:
        """Input shape: Batch x Time x Channel"""
        if output_attentions or layer_head_mask is not None:
-            # TODO: Improve this warning with e.g. `model.config._attn_implementation = "manual"` once this is implemented.
-            logger.warning_once(
-                "Florence2Model is using Florence2SdpaAttention, but `torch.nn.functional.scaled_dot_product_attention` does not support `output_attentions=True` or `layer_head_mask` not None. Falling back to the manual attention"
-                ' implementation, but specifying the manual implementation will be required from Transformers version v5.0.0 onwards. This warning can be removed using the argument `attn_implementation="eager"` when loading the model.'
-            )
            return super().forward(
                hidden_states,
                key_value_states=key_value_states,
@@ -1860,9 +1846,6 @@ class Florence2Decoder(Florence2LanguagePreTrainedModel):
        hidden_states = nn.functional.dropout(hidden_states, p=self.dropout, training=self.training)

        if self.gradient_checkpointing and self.training and use_cache:
-            logger.warning_once(
-                "`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`..."
-            )
            use_cache = False

        # decoder layers
@@ -2160,8 +2143,6 @@ class Florence2LanguageForConditionalGeneration(Florence2LanguagePreTrainedModel
        return_dict = return_dict if return_dict is not None else self.config.use_return_dict

        if labels is not None:
-            if use_cache:
-                logger.warning("The `use_cache` argument is changed to `False` since `labels` is provided.")
            use_cache = False
            if decoder_input_ids is None and decoder_inputs_embeds is None:
                decoder_input_ids = shift_tokens_right(
@@ -17,7 +17,6 @@
 """Test script to verify PI0Fast policy integration with LeRobot vs the original implementation"""
 # ruff: noqa: E402

-import os
 import random
 from copy import deepcopy
 from typing import Any
@@ -28,10 +27,6 @@ import torch

 pytest.importorskip("transformers")
 pytest.importorskip("scipy")
-pytestmark = pytest.mark.skipif(
-    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
-    reason="This test requires accepting the model license",
-)

 from lerobot.policies.pi0_fast.configuration_pi0_fast import PI0FastConfig
 from lerobot.policies.pi0_fast.modeling_pi0_fast import PI0FastPolicy
@@ -16,17 +16,8 @@

 """Test script to verify PI0 policy integration with LeRobot, only meant to be run locally!"""

-import os
-
-import pytest
 import torch

-# Skip this entire module in CI
-pytestmark = pytest.mark.skipif(
-    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
-    reason="This test requires accepting the model license",
-)
-
 from lerobot.policies.factory import make_policy_config  # noqa: E402
 from lerobot.policies.pi0 import (  # noqa: E402
    PI0Config,
@@ -16,25 +16,15 @@

 """Test script to verify PI0.5 (pi05) support in PI0 policy, only meant to be run locally!"""

-import os
-
-import pytest
 import torch

-from lerobot.utils.random_utils import set_seed
-
-# Skip this entire module in CI
-pytestmark = pytest.mark.skipif(
-    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
-    reason="This test requires accepting the model license",
-)
-
 from lerobot.policies.factory import make_policy_config  # noqa: E402
 from lerobot.policies.pi05 import (  # noqa: E402
    PI05Config,
    PI05Policy,
    make_pi05_pre_post_processors,  # noqa: E402
 )
+from lerobot.utils.random_utils import set_seed
 from tests.utils import require_cuda  # noqa: E402


@@ -24,9 +24,10 @@ import torch
 # Skip this entire module in CI
 pytestmark = pytest.mark.skipif(
    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
-    reason="This test requires local OpenPI installation and is not meant for CI",
+    reason="TODO: This test seems to hang the CI",
 )

+
 from lerobot.configs.types import FeatureType, PolicyFeature, RTCAttentionSchedule  # noqa: E402
 from lerobot.policies.pi05 import PI05Config, PI05Policy, make_pi05_pre_post_processors  # noqa: E402
 from lerobot.policies.rtc.configuration_rtc import RTCConfig  # noqa: E402
@@ -24,9 +24,10 @@ import torch
 # Skip this entire module in CI
 pytestmark = pytest.mark.skipif(
    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
-    reason="This test requires local OpenPI installation and is not meant for CI",
+    reason="TODO: This test seems to hang the CI",
 )

+
 from lerobot.configs.types import FeatureType, PolicyFeature, RTCAttentionSchedule  # noqa: E402
 from lerobot.policies.pi0 import PI0Config, PI0Policy, make_pi0_pre_post_processors  # noqa: E402
 from lerobot.policies.rtc.configuration_rtc import RTCConfig  # noqa: E402
Author	SHA1	Message	Date
Jade Choghari	0bda187268	xvla log fix	2026-02-26 09:55:25 +00:00
Steven Palma	59b33c0ea3	Merge branch 'chore/bump_transformers_v5' into ci/add_hf_account	2026-02-25 11:56:52 +01:00
Steven Palma	4419901e6b	fix(ci): temp fix for pi0 rtc test	2026-02-25 11:42:04 +01:00
Jade Choghari	3f3a159cff	fix wall x for transformer v5 (#3008 ) * tv5 fix * various wall x fixes * Delete tests/policies/pi0_pi05/print_pi05_output_logits.py Signed-off-by: Jade Choghari <chogharijade@gmail.com> * sync modeling_florence2.py with chore/bump_transformers_v5 * more * more fixes * more * remove comment * more --------- Signed-off-by: Jade Choghari <chogharijade@gmail.com>	2026-02-24 21:16:37 +03:00
Steven Palma	6deebf1e47	Merge branch 'chore/bump_transformers_v5' into ci/add_hf_account	2026-02-24 17:50:48 +01:00
Steven Palma	481a956100	Merge branch 'chore/bump_transformers_v5' into ci/add_hf_account Signed-off-by: Steven Palma <imstevenpmwork@ieee.org>	2026-02-24 13:18:07 +01:00
Steven Palma	ffde29be49	chore(ci): change hf call + secret name	2026-02-24 12:11:31 +01:00
Steven Palma	2504d00707	feat(ci): log into HF to unblock some CI tests	2026-02-24 11:56:13 +01:00