refactor(normalization): Remove unused state dict transformation methods and streamline imports

- Eliminated the _transform_state_dict_keys and _load_as_safetensor methods from PI0Policy, simplifying the model loading process. - Cleaned up imports in modeling_pi0.py by removing log_model_loading_keys and init_logging. - Updated TDMPCPolicy and VQBeTPolicy to handle action removal from batches during offline evaluation. - Introduced hotswap_stats function in normalize_processor.py to update normalization statistics dynamically, with corresponding tests to ensure functionality.
2026-07-09 19:11:44 +00:00 · 2025-08-01 08:55:38 +02:00
parent f02ce69df0
commit 8ff95be04c
6 changed files with 398 additions and 98 deletions
@@ -65,8 +65,7 @@ from lerobot.policies.pi0.paligemma_with_expert import (
    PaliGemmaWithExpertModel,
 )
 from lerobot.policies.pretrained import PreTrainedPolicy
-from lerobot.policies.utils import log_model_loading_keys
-from lerobot.utils.utils import get_safe_dtype, init_logging
+from lerobot.utils.utils import get_safe_dtype


 def create_sinusoidal_pos_embedding(
@@ -245,99 +244,6 @@ class PI0Policy(PreTrainedPolicy):
        """This should be called whenever the environment is reset."""
        self._action_queue = deque([], maxlen=self.config.n_action_steps)

-    @classmethod
-    def _transform_state_dict_keys(cls, state_dict: dict) -> dict:
-        """
-        Transform state dict keys to match expected model structure.
-
-        Transformations:
-        - model.paligemma_with_expert.paligemma.language_model.lm_head ->
-          model.paligemma_with_expert.paligemma.lm_head
-        - model.paligemma_with_expert.paligemma.language_model.model ->
-          model.paligemma_with_expert.paligemma.model.language_model
-        - model.paligemma_with_expert.paligemma.vision_tower ->
-          model.paligemma_with_expert.paligemma.model.vision_tower
-        - model.paligemma_with_expert.paligemma.multi_modal_projector ->
-          model.paligemma_with_expert.paligemma.model.multi_modal_projector
-
-        Also handles tied weights between lm_head.weight and
-        embed_tokens.weight.
-        """
-        import re
-
-        transformed_dict = {}
-
-        transformations = [
-            (
-                re.compile(r"\.paligemma_with_expert\.paligemma\.language_model\.lm_head"),
-                ".paligemma_with_expert.paligemma.lm_head",
-            ),
-            (
-                re.compile(r"\.paligemma_with_expert\.paligemma\.language_model\.model"),
-                ".paligemma_with_expert.paligemma.model.language_model",
-            ),
-            (
-                re.compile(r"\.paligemma_with_expert\.paligemma\.vision_tower"),
-                ".paligemma_with_expert.paligemma.model.vision_tower",
-            ),
-            (
-                re.compile(r"\.paligemma_with_expert\.paligemma\.multi_modal_projector"),
-                ".paligemma_with_expert.paligemma.model.multi_modal_projector",
-            ),
-        ]
-
-        for key, value in state_dict.items():
-            new_key = key
-            for pattern, replacement in transformations:
-                new_key = pattern.sub(replacement, new_key)
-            transformed_dict[new_key] = value
-
-        # Handle tied weights: lm_head.weight and embed_tokens.weight share memory
-        lm_head_key = None
-        embed_tokens_key = None
-
-        for key in transformed_dict:
-            if key.endswith(".paligemma_with_expert.paligemma.lm_head.weight"):
-                lm_head_key = key
-            elif key.endswith(".paligemma_with_expert.paligemma.model.language_model.embed_tokens.weight"):
-                embed_tokens_key = key
-            if lm_head_key and embed_tokens_key:
-                break
-
-        if lm_head_key and not embed_tokens_key:
-            embed_tokens_key = lm_head_key.replace(
-                ".lm_head.weight", ".model.language_model.embed_tokens.weight"
-            )
-            transformed_dict[embed_tokens_key] = transformed_dict[lm_head_key]
-        elif embed_tokens_key and not lm_head_key:
-            lm_head_key = embed_tokens_key.replace(
-                ".model.language_model.embed_tokens.weight", ".lm_head.weight"
-            )
-            transformed_dict[lm_head_key] = transformed_dict[embed_tokens_key]
-
-        return transformed_dict
-
-    @classmethod
-    def _load_as_safetensor(
-        cls, model: "PI0Policy", model_file: str, map_location: str, strict: bool
-    ) -> "PI0Policy":
-        """Override to apply key transformations before loading."""
-        from safetensors.torch import load_file
-
-        init_logging()
-        # Load the state dict from file safely
-        state_dict = load_file(model_file, device=map_location)
-
-        # Apply key transformations
-        transformed_state_dict = cls._transform_state_dict_keys(state_dict)
-
-        # Load the transformed state dict
-        msg = model.load_state_dict(transformed_state_dict, strict=strict)
-
-        # Log message
-        log_model_loading_keys(msg.missing_keys, msg.unexpected_keys)
-        return model
-
    def get_optim_params(self) -> dict:
        return self.parameters()

@@ -141,6 +141,9 @@ class TDMPCPolicy(PreTrainedPolicy):
        if self.config.image_features:
            batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
            batch[OBS_IMAGE] = batch[next(iter(self.config.image_features))]
+        # NOTE: for offline evaluation, we have action in the batch, so we need to pop it out
+        if ACTION in batch:
+            batch.pop(ACTION)

        self._queues = populate_queues(self._queues, batch)

@@ -134,6 +134,9 @@ class VQBeTPolicy(PreTrainedPolicy):
        batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
        # NOTE: It's important that this happens after stacking the images into a single key.
        batch["observation.images"] = torch.stack([batch[key] for key in self.config.image_features], dim=-4)
+        # NOTE: for offline evaluation, we have action in the batch, so we need to pop it out
+        if ACTION in batch:
+            batch.pop(ACTION)

        self._queues = populate_queues(self._queues, batch)

@@ -16,7 +16,7 @@

 from .batch_processor import ToBatchProcessor
 from .device_processor import DeviceProcessor
-from .normalize_processor import NormalizerProcessor, UnnormalizerProcessor
+from .normalize_processor import NormalizerProcessor, UnnormalizerProcessor, hotswap_stats
 from .observation_processor import VanillaObservationProcessor
 from .pipeline import (
    ActionProcessor,
@@ -43,6 +43,7 @@ __all__ = [
    "InfoProcessor",
    "NormalizerProcessor",
    "UnnormalizerProcessor",
+    "hotswap_stats",
    "ObservationProcessor",
    "ProcessorStep",
    "ProcessorStepRegistry",
@@ -1,6 +1,8 @@
 from __future__ import annotations

+
 from collections.abc import Mapping
+from copy import deepcopy
 from dataclasses import dataclass, field
 from typing import Any

@@ -10,7 +12,7 @@ from torch import Tensor

 from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
 from lerobot.datasets.lerobot_dataset import LeRobotDataset
-from lerobot.processor.pipeline import EnvTransition, ProcessorStepRegistry, TransitionKey
+from lerobot.processor.pipeline import EnvTransition, ProcessorStepRegistry, TransitionKey, RobotProcessor


 def _convert_stats_to_tensors(stats: dict[str, dict[str, Any]]) -> dict[str, dict[str, Tensor]]:
@@ -402,3 +404,13 @@ class UnnormalizerProcessor:

    def feature_contract(self, features: dict[str, PolicyFeature]) -> dict[str, PolicyFeature]:
        return features
+
+
+def hotswap_stats(robot_processor: RobotProcessor, stats: dict[str, dict[str, Any]]) -> RobotProcessor:
+    robot_processor = deepcopy(robot_processor)
+    for step in robot_processor.steps:
+        if isinstance(step, NormalizerProcessor) or isinstance(step, UnnormalizerProcessor):
+            step: NormalizerProcessor | UnnormalizerProcessor
+            step.stats = stats
+            step._tensor_stats = _convert_stats_to_tensors(stats)
+    return robot_processor