From 3c302e39cbdbcf46f6157c0f8c25acd41b2d1b98 Mon Sep 17 00:00:00 2001 From: Khalil Meftah Date: Thu, 12 Mar 2026 15:24:38 +0100 Subject: [PATCH] test(rewards): add reward model tests and update existing test imports --- tests/rewards/__init__.py | 0 .../test_classifier_processor.py | 200 ++---------------- .../test_modeling_classifier.py | 24 ++- tests/rewards/test_reward_model_base.py | 101 +++++++++ .../test_sarm_processor.py | 18 +- .../{policies => rewards}/test_sarm_utils.py | 2 +- 6 files changed, 141 insertions(+), 204 deletions(-) create mode 100644 tests/rewards/__init__.py rename tests/{processor => rewards}/test_classifier_processor.py (51%) rename tests/{policies/hilserl => rewards}/test_modeling_classifier.py (86%) create mode 100644 tests/rewards/test_reward_model_base.py rename tests/{policies => rewards}/test_sarm_processor.py (97%) rename tests/{policies => rewards}/test_sarm_utils.py (99%) diff --git a/tests/rewards/__init__.py b/tests/rewards/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/tests/processor/test_classifier_processor.py b/tests/rewards/test_classifier_processor.py similarity index 51% rename from tests/processor/test_classifier_processor.py rename to tests/rewards/test_classifier_processor.py index e1567bf29..e01515929 100644 --- a/tests/processor/test_classifier_processor.py +++ b/tests/rewards/test_classifier_processor.py @@ -17,12 +17,9 @@ import tempfile -import pytest import torch from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature -from lerobot.policies.sac.reward_model.configuration_classifier import RewardClassifierConfig -from lerobot.policies.sac.reward_model.processor_classifier import make_classifier_processor from lerobot.processor import ( DataProcessorPipeline, DeviceProcessorStep, @@ -31,6 +28,8 @@ from lerobot.processor import ( TransitionKey, ) from lerobot.processor.converters import create_transition, transition_to_batch +from lerobot.rewards.classifier.configuration_classifier import RewardClassifierConfig +from lerobot.rewards.classifier.processor_classifier import make_classifier_processor from lerobot.utils.constants import OBS_IMAGE, OBS_STATE @@ -42,12 +41,12 @@ def create_default_config(): OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)), } config.output_features = { - "reward": PolicyFeature(type=FeatureType.ACTION, shape=(1,)), # Classifier output + "reward": PolicyFeature(type=FeatureType.ACTION, shape=(1,)), } config.normalization_mapping = { FeatureType.STATE: NormalizationMode.MEAN_STD, FeatureType.VISUAL: NormalizationMode.IDENTITY, - FeatureType.ACTION: NormalizationMode.IDENTITY, # No normalization for classifier output + FeatureType.ACTION: NormalizationMode.IDENTITY, } config.device = "cpu" return config @@ -57,8 +56,8 @@ def create_default_stats(): """Create default dataset statistics for testing.""" return { OBS_STATE: {"mean": torch.zeros(10), "std": torch.ones(10)}, - OBS_IMAGE: {}, # No normalization for images - "reward": {}, # No normalization for classifier output + OBS_IMAGE: {}, + "reward": {}, } @@ -69,17 +68,14 @@ def test_make_classifier_processor_basic(): preprocessor, postprocessor = make_classifier_processor(config, stats) - # Check processor names assert preprocessor.name == "classifier_preprocessor" assert postprocessor.name == "classifier_postprocessor" - # Check steps in preprocessor assert len(preprocessor.steps) == 3 - assert isinstance(preprocessor.steps[0], NormalizerProcessorStep) # For input features - assert isinstance(preprocessor.steps[1], NormalizerProcessorStep) # For output features + assert isinstance(preprocessor.steps[0], NormalizerProcessorStep) + assert isinstance(preprocessor.steps[1], NormalizerProcessorStep) assert isinstance(preprocessor.steps[2], DeviceProcessorStep) - # Check steps in postprocessor assert len(postprocessor.steps) == 2 assert isinstance(postprocessor.steps[0], DeviceProcessorStep) assert isinstance(postprocessor.steps[1], IdentityProcessorStep) @@ -90,128 +86,21 @@ def test_classifier_processor_normalization(): config = create_default_config() stats = create_default_stats() - preprocessor, postprocessor = make_classifier_processor( - config, - stats, - ) + preprocessor, postprocessor = make_classifier_processor(config, stats) - # Create test data - observation = { - OBS_STATE: torch.randn(10), - OBS_IMAGE: torch.randn(3, 224, 224), - } - action = torch.randn(1) # Dummy action/reward - transition = create_transition(observation, action) - batch = transition_to_batch(transition) - - # Process through preprocessor - processed = preprocessor(batch) - - # Check that data is processed - assert processed[OBS_STATE].shape == (10,) - assert processed[OBS_IMAGE].shape == (3, 224, 224) - assert processed[TransitionKey.ACTION.value].shape == (1,) - - -@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available") -def test_classifier_processor_cuda(): - """Test Classifier processor with CUDA device.""" - config = create_default_config() - config.device = "cuda" - stats = create_default_stats() - - preprocessor, postprocessor = make_classifier_processor( - config, - stats, - ) - - # Create CPU data observation = { OBS_STATE: torch.randn(10), OBS_IMAGE: torch.randn(3, 224, 224), } action = torch.randn(1) transition = create_transition(observation, action) - batch = transition_to_batch(transition) - # Process through preprocessor - processed = preprocessor(batch) - # Check that data is on CUDA - assert processed[OBS_STATE].device.type == "cuda" - assert processed[OBS_IMAGE].device.type == "cuda" - assert processed[TransitionKey.ACTION.value].device.type == "cuda" - - # Process through postprocessor - postprocessed = postprocessor(processed[TransitionKey.ACTION.value]) - - # Check that output is back on CPU - assert postprocessed.device.type == "cpu" - - -@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available") -def test_classifier_processor_accelerate_scenario(): - """Test Classifier processor in simulated Accelerate scenario.""" - config = create_default_config() - config.device = "cuda:0" - stats = create_default_stats() - - preprocessor, postprocessor = make_classifier_processor( - config, - stats, - ) - - # Simulate Accelerate: data already on GPU - device = torch.device("cuda:0") - observation = { - OBS_STATE: torch.randn(10).to(device), - OBS_IMAGE: torch.randn(3, 224, 224).to(device), - } - action = torch.randn(1).to(device) - transition = create_transition(observation, action) - - batch = transition_to_batch(transition) - - # Process through preprocessor - - processed = preprocessor(batch) - - # Check that data stays on same GPU - assert processed[OBS_STATE].device == device - assert processed[OBS_IMAGE].device == device - assert processed[TransitionKey.ACTION.value].device == device - - -@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs") -def test_classifier_processor_multi_gpu(): - """Test Classifier processor with multi-GPU setup.""" - config = create_default_config() - config.device = "cuda:0" - stats = create_default_stats() - - preprocessor, postprocessor = make_classifier_processor(config, stats) - - # Simulate data on different GPU - device = torch.device("cuda:1") - observation = { - OBS_STATE: torch.randn(10).to(device), - OBS_IMAGE: torch.randn(3, 224, 224).to(device), - } - action = torch.randn(1).to(device) - transition = create_transition(observation, action) - - batch = transition_to_batch(transition) - - # Process through preprocessor - - processed = preprocessor(batch) - - # Check that data stays on cuda:1 - assert processed[OBS_STATE].device == device - assert processed[OBS_IMAGE].device == device - assert processed[TransitionKey.ACTION.value].device == device + assert processed[OBS_STATE].shape == (10,) + assert processed[OBS_IMAGE].shape == (3, 224, 224) + assert processed[TransitionKey.ACTION.value].shape == (1,) def test_classifier_processor_without_stats(): @@ -220,18 +109,15 @@ def test_classifier_processor_without_stats(): preprocessor, postprocessor = make_classifier_processor(config, dataset_stats=None) - # Should still create processors assert preprocessor is not None assert postprocessor is not None - # Process should still work observation = { OBS_STATE: torch.randn(10), OBS_IMAGE: torch.randn(3, 224, 224), } action = torch.randn(1) transition = create_transition(observation, action) - batch = transition_to_batch(transition) processed = preprocessor(batch) @@ -246,15 +132,12 @@ def test_classifier_processor_save_and_load(): preprocessor, postprocessor = make_classifier_processor(config, stats) with tempfile.TemporaryDirectory() as tmpdir: - # Save preprocessor preprocessor.save_pretrained(tmpdir) - # Load preprocessor loaded_preprocessor = DataProcessorPipeline.from_pretrained( tmpdir, config_filename="classifier_preprocessor.json" ) - # Test that loaded processor works observation = { OBS_STATE: torch.randn(10), OBS_IMAGE: torch.randn(3, 224, 224), @@ -269,55 +152,13 @@ def test_classifier_processor_save_and_load(): assert processed[TransitionKey.ACTION.value].shape == (1,) -@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available") -def test_classifier_processor_mixed_precision(): - """Test Classifier processor with mixed precision.""" - config = create_default_config() - config.device = "cuda" - stats = create_default_stats() - - preprocessor, postprocessor = make_classifier_processor(config, stats) - - # Replace DeviceProcessorStep with one that uses float16 - modified_steps = [] - for step in preprocessor.steps: - if isinstance(step, DeviceProcessorStep): - modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="float16")) - else: - modified_steps.append(step) - preprocessor.steps = modified_steps - - # Create test data - observation = { - OBS_STATE: torch.randn(10, dtype=torch.float32), - OBS_IMAGE: torch.randn(3, 224, 224, dtype=torch.float32), - } - action = torch.randn(1, dtype=torch.float32) - transition = create_transition(observation, action) - - batch = transition_to_batch(transition) - - # Process through preprocessor - - processed = preprocessor(batch) - - # Check that data is converted to float16 - assert processed[OBS_STATE].dtype == torch.float16 - assert processed[OBS_IMAGE].dtype == torch.float16 - assert processed[TransitionKey.ACTION.value].dtype == torch.float16 - - def test_classifier_processor_batch_data(): """Test Classifier processor with batched data.""" config = create_default_config() stats = create_default_stats() - preprocessor, postprocessor = make_classifier_processor( - config, - stats, - ) + preprocessor, postprocessor = make_classifier_processor(config, stats) - # Test with batched data batch_size = 16 observation = { OBS_STATE: torch.randn(batch_size, 10), @@ -325,14 +166,10 @@ def test_classifier_processor_batch_data(): } action = torch.randn(batch_size, 1) transition = create_transition(observation, action) - batch = transition_to_batch(transition) - # Process through preprocessor - processed = preprocessor(batch) - # Check that batch dimension is preserved assert processed[OBS_STATE].shape == (batch_size, 10) assert processed[OBS_IMAGE].shape == (batch_size, 3, 224, 224) assert processed[TransitionKey.ACTION.value].shape == (batch_size, 1) @@ -343,20 +180,13 @@ def test_classifier_processor_postprocessor_identity(): config = create_default_config() stats = create_default_stats() - preprocessor, postprocessor = make_classifier_processor( - config, - stats, - ) + preprocessor, postprocessor = make_classifier_processor(config, stats) - # Create test data for postprocessor - reward = torch.tensor([[0.8], [0.3], [0.9]]) # Batch of rewards/predictions + reward = torch.tensor([[0.8], [0.3], [0.9]]) transition = create_transition(action=reward) - _ = transition_to_batch(transition) - # Process through postprocessor processed = postprocessor(reward) - # IdentityProcessor should leave values unchanged (except device) assert torch.allclose(processed.cpu(), reward.cpu()) assert processed.device.type == "cpu" diff --git a/tests/policies/hilserl/test_modeling_classifier.py b/tests/rewards/test_modeling_classifier.py similarity index 86% rename from tests/policies/hilserl/test_modeling_classifier.py rename to tests/rewards/test_modeling_classifier.py index a62ef3ebb..7e3d272b5 100644 --- a/tests/policies/hilserl/test_modeling_classifier.py +++ b/tests/rewards/test_modeling_classifier.py @@ -1,5 +1,3 @@ -# !/usr/bin/env python - # Copyright 2025 The HuggingFace Inc. team. All rights reserved. # # Licensed under the Apache License, Version 2.0 (the "License"); @@ -18,8 +16,8 @@ import pytest import torch from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature -from lerobot.policies.sac.reward_model.configuration_classifier import RewardClassifierConfig -from lerobot.policies.sac.reward_model.modeling_classifier import ClassifierOutput +from lerobot.rewards.classifier.configuration_classifier import RewardClassifierConfig +from lerobot.rewards.classifier.modeling_classifier import ClassifierOutput from lerobot.utils.constants import OBS_IMAGE, REWARD from tests.utils import require_package @@ -42,7 +40,7 @@ def test_classifier_output(): reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers" ) def test_binary_classifier_with_default_params(): - from lerobot.policies.sac.reward_model.modeling_classifier import Classifier + from lerobot.rewards.classifier.modeling_classifier import Classifier config = RewardClassifierConfig() config.input_features = { @@ -86,7 +84,7 @@ def test_binary_classifier_with_default_params(): reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers" ) def test_multiclass_classifier(): - from lerobot.policies.sac.reward_model.modeling_classifier import Classifier + from lerobot.rewards.classifier.modeling_classifier import Classifier num_classes = 5 config = RewardClassifierConfig() @@ -128,11 +126,15 @@ def test_multiclass_classifier(): reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers" ) def test_default_device(): - from lerobot.policies.sac.reward_model.modeling_classifier import Classifier + from lerobot.rewards.classifier.modeling_classifier import Classifier config = RewardClassifierConfig() - assert config.device == "cpu" + assert config.device is None or config.device == "cpu" + config.input_features = { + OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)), + } + config.num_cameras = 1 classifier = Classifier(config) for p in classifier.parameters(): assert p.device == torch.device("cpu") @@ -143,11 +145,15 @@ def test_default_device(): reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers" ) def test_explicit_device_setup(): - from lerobot.policies.sac.reward_model.modeling_classifier import Classifier + from lerobot.rewards.classifier.modeling_classifier import Classifier config = RewardClassifierConfig(device="cpu") assert config.device == "cpu" + config.input_features = { + OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)), + } + config.num_cameras = 1 classifier = Classifier(config) for p in classifier.parameters(): assert p.device == torch.device("cpu") diff --git a/tests/rewards/test_reward_model_base.py b/tests/rewards/test_reward_model_base.py new file mode 100644 index 000000000..96557359f --- /dev/null +++ b/tests/rewards/test_reward_model_base.py @@ -0,0 +1,101 @@ +# Copyright 2026 The HuggingFace Inc. team. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests for the reward model base classes and registry.""" + +import pytest +import torch + +from lerobot.configs.rewards import RewardModelConfig +from lerobot.rewards.pretrained import PreTrainedRewardModel + + +def test_reward_model_config_registry(): + """Verify that classifier and sarm are registered.""" + known = RewardModelConfig.get_known_choices() + assert "reward_classifier" in known + assert "sarm" in known + + +def test_reward_model_config_lookup(): + """Verify that we can look up configs by name.""" + cls = RewardModelConfig.get_choice_class("reward_classifier") + from lerobot.rewards.classifier.configuration_classifier import RewardClassifierConfig + + assert cls is RewardClassifierConfig + + +def test_factory_get_reward_model_class(): + """Test the get_reward_model_class factory.""" + from lerobot.rewards.factory import get_reward_model_class + + cls = get_reward_model_class("sarm") + from lerobot.rewards.sarm.modeling_sarm import SARMRewardModel + + assert cls is SARMRewardModel + + +def test_factory_unknown_raises(): + """Unknown name should raise ValueError.""" + from lerobot.rewards.factory import get_reward_model_class + + with pytest.raises(ValueError, match="not available"): + get_reward_model_class("nonexistent_reward_model") + + +def test_pretrained_reward_model_requires_config_class(): + """Subclass without config_class should fail.""" + with pytest.raises(TypeError, match="must define 'config_class'"): + + class BadModel(PreTrainedRewardModel): + name = "bad" + + def compute_reward(self, batch): + pass + + +def test_pretrained_reward_model_requires_name(): + """Subclass without name should fail.""" + with pytest.raises(TypeError, match="must define 'name'"): + + class BadModel(PreTrainedRewardModel): + config_class = RewardModelConfig + + def compute_reward(self, batch): + pass + + +def test_non_trainable_forward_raises(): + """Non-trainable model should raise on forward().""" + from dataclasses import dataclass + + from lerobot.optim.optimizers import AdamWConfig + + @dataclass + class DummyConfig(RewardModelConfig): + def get_optimizer_preset(self): + return AdamWConfig(lr=1e-4) + + class DummyReward(PreTrainedRewardModel): + config_class = DummyConfig + name = "dummy_test" + + def compute_reward(self, batch): + return torch.zeros(1) + + config = DummyConfig() + model = DummyReward(config) + + with pytest.raises(NotImplementedError, match="not trainable"): + model.forward({"x": torch.zeros(1)}) diff --git a/tests/policies/test_sarm_processor.py b/tests/rewards/test_sarm_processor.py similarity index 97% rename from tests/policies/test_sarm_processor.py rename to tests/rewards/test_sarm_processor.py index 66404f663..c75442951 100644 --- a/tests/policies/test_sarm_processor.py +++ b/tests/rewards/test_sarm_processor.py @@ -104,8 +104,8 @@ class TestSARMEncodingProcessorStepEndToEnd: def mock_clip_model(self): """Mock CLIP model to avoid loading real weights.""" with ( - patch("lerobot.policies.sarm.processor_sarm.CLIPModel") as mock_model_cls, - patch("lerobot.policies.sarm.processor_sarm.CLIPProcessor") as mock_processor_cls, + patch("lerobot.rewards.sarm.processor_sarm.CLIPModel") as mock_model_cls, + patch("lerobot.rewards.sarm.processor_sarm.CLIPProcessor") as mock_processor_cls, ): # Mock the CLIP model - return embeddings based on input batch size mock_model = MagicMock() @@ -142,7 +142,7 @@ class TestSARMEncodingProcessorStepEndToEnd: @pytest.fixture def processor_with_mocks(self, mock_clip_model): """Create a processor with mocked CLIP and dataset metadata for dual mode.""" - from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep + from lerobot.rewards.sarm.processor_sarm import SARMEncodingProcessorStep # Dual mode config with both sparse and dense annotations config = MockConfig( @@ -256,7 +256,7 @@ class TestSARMEncodingProcessorStepEndToEnd: def test_call_with_batched_input(self, mock_clip_model): """Test processor __call__ with a batched input (multiple frames) in dual mode.""" - from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep + from lerobot.rewards.sarm.processor_sarm import SARMEncodingProcessorStep config = MockConfig( n_obs_steps=8, @@ -332,7 +332,7 @@ class TestSARMEncodingProcessorStepEndToEnd: def test_targets_increase_with_progress(self, mock_clip_model): """Test that both sparse and dense targets increase as frame index progresses.""" - from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep + from lerobot.rewards.sarm.processor_sarm import SARMEncodingProcessorStep config = MockConfig( n_obs_steps=8, @@ -404,7 +404,7 @@ class TestSARMEncodingProcessorStepEndToEnd: def test_progress_labels_exact_values(self, mock_clip_model): """Test that progress labels (stage.tau) are computed correctly for known positions.""" - from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep + from lerobot.rewards.sarm.processor_sarm import SARMEncodingProcessorStep # Simple setup: 2 sparse stages, 4 dense stages, 100 frame episode config = MockConfig( @@ -495,7 +495,7 @@ class TestSARMEncodingProcessorStepEndToEnd: """Test that rewind augmentation correctly extends sequence and generates targets.""" import random - from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep + from lerobot.rewards.sarm.processor_sarm import SARMEncodingProcessorStep config = MockConfig( n_obs_steps=8, @@ -587,8 +587,8 @@ class TestSARMEncodingProcessorStepEndToEnd: def test_full_sequence_target_consistency(self, mock_clip_model): """Test that the full sequence of targets is consistent with frame positions.""" - from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep - from lerobot.policies.sarm.sarm_utils import find_stage_and_tau + from lerobot.rewards.sarm.processor_sarm import SARMEncodingProcessorStep + from lerobot.rewards.sarm.sarm_utils import find_stage_and_tau config = MockConfig( n_obs_steps=8, diff --git a/tests/policies/test_sarm_utils.py b/tests/rewards/test_sarm_utils.py similarity index 99% rename from tests/policies/test_sarm_utils.py rename to tests/rewards/test_sarm_utils.py index 510477ec8..9ee542909 100644 --- a/tests/policies/test_sarm_utils.py +++ b/tests/rewards/test_sarm_utils.py @@ -18,7 +18,7 @@ import numpy as np import pytest import torch -from lerobot.policies.sarm.sarm_utils import ( +from lerobot.rewards.sarm.sarm_utils import ( apply_rewind_augmentation, compute_absolute_indices, compute_tau,