add various experiments for wavelet

2026-05-12 07:09:43 +00:00 · 2026-02-13 10:27:02 +00:00
270 changed files with 5459 additions and 9813 deletions
@@ -44,7 +44,7 @@ permissions:
 # Sets up the environment variables
 env:
  UV_VERSION: "0.8.0"
-  PYTHON_VERSION: "3.12"
+  PYTHON_VERSION: "3.10"

 # Ensures that only the latest commit for a PR or branch is built, canceling older runs.
 concurrency:
@@ -61,7 +61,6 @@ jobs:
      MUJOCO_GL: egl
      HF_HOME: /mnt/cache/.cache/huggingface
      HF_LEROBOT_HOME: /mnt/cache/.cache/huggingface/lerobot
-      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    steps:
      - uses: actions/checkout@v6
        with:
@@ -90,11 +89,5 @@ jobs:
      - name: Install lerobot with test extras
        run: uv sync --extra "test"

-      - name: Login to Hugging Face
-        if: env.HF_USER_TOKEN != ''
-        run: |
-          uv run hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
-          uv run hf auth whoami
-
      - name: Run pytest
        run: uv run pytest tests -vv --maxfail=10
@@ -37,7 +37,7 @@ permissions:
 # Sets up the environment variables
 env:
  UV_VERSION: "0.8.0"
-  PYTHON_VERSION: "3.12"
+  PYTHON_VERSION: "3.10"
  DOCKER_IMAGE_NAME: huggingface/lerobot-gpu

 # Ensures that only the latest action is built, canceling older runs.
@@ -60,7 +60,6 @@ jobs:
      MUJOCO_GL: egl
      HF_HOME: /mnt/cache/.cache/huggingface
      HF_LEROBOT_HOME: /mnt/cache/.cache/huggingface/lerobot
-      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    steps:
      - uses: actions/checkout@v6
        with:
@@ -88,12 +87,6 @@ jobs:
      - name: Install lerobot with all extras
        run: uv sync --extra all # TODO(Steven): Make flash-attn optional

-      - name: Login to Hugging Face
-        if: env.HF_USER_TOKEN != ''
-        run: |
-          uv run hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
-          uv run hf auth whoami
-
      - name: Run pytest (all extras)
        run: uv run pytest tests -vv --maxfail=10

@@ -169,7 +162,6 @@ jobs:
      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
      TORCH_HOME: /home/user_lerobot/.cache/torch
      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
-      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    container:
      image: ${{ needs.build-and-push-docker.outputs.image_tag }} # zizmor: ignore[unpinned-images]
      options: --gpus all --shm-size "16gb"
@@ -181,13 +173,6 @@ jobs:
        shell: bash
        working-directory: /lerobot
    steps:
-      - name: Login to Hugging Face
-        if: env.HF_USER_TOKEN != ''
-        run: |
-          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
-          hf auth whoami
-      - name: Fix ptxas permissions
-        run: chmod +x /lerobot/.venv/lib/python3.12/site-packages/triton/backends/nvidia/bin/ptxas
      - name: Run pytest on GPU
        run: pytest tests -vv --maxfail=10
      - name: Run end-to-end tests
@@ -28,7 +28,7 @@ on:
 # Sets up the environment variables
 env:
  UV_VERSION: "0.8.0"
-  PYTHON_VERSION: "3.12"
+  PYTHON_VERSION: "3.10"
  DOCKER_IMAGE_NAME_CPU: huggingface/lerobot-cpu:latest
  DOCKER_IMAGE_NAME_GPU: huggingface/lerobot-gpu:latest

@@ -119,7 +119,6 @@ jobs:
      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
      TORCH_HOME: /home/user_lerobot/.cache/torch
      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
-      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    container:
      image: ${{ needs.build-docker-cpu-nightly.outputs.image_tag }} # zizmor: ignore[unpinned-images]
      options: --shm-size "16gb"
@@ -131,11 +130,6 @@ jobs:
        shell: bash
        working-directory: /lerobot
    steps:
-      - name: Login to Hugging Face
-        if: env.HF_USER_TOKEN != ''
-        run: |
-          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
-          hf auth whoami
      - name: Run pytest on CPU
        run: pytest tests -vv --maxfail=10
      - name: Run end-to-end tests
@@ -152,7 +146,6 @@ jobs:
      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
      TORCH_HOME: /home/user_lerobot/.cache/torch
      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
-      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    container:
      image: ${{ needs.build-docker-gpu-nightly.outputs.image_tag }} # zizmor: ignore[unpinned-images]
      options: --gpus all --shm-size "16gb"
@@ -164,11 +157,6 @@ jobs:
        shell: bash
        working-directory: /lerobot
    steps:
-      - name: Login to Hugging Face
-        if: env.HF_USER_TOKEN != ''
-        run: |
-          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
-          hf auth whoami
      - name: Run pytest on GPU
        run: pytest tests -vv --maxfail=10
      - name: Run end-to-end tests
@@ -186,7 +174,6 @@ jobs:
      TORCH_HOME: /home/user_lerobot/.cache/torch
      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
      CUDA_VISIBLE_DEVICES: "0,1,2,3"
-      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    container:
      image: ${{ needs.build-docker-gpu-nightly.outputs.image_tag }} # zizmor: ignore[unpinned-images]
      options: --gpus all --shm-size "16gb"
@@ -198,15 +185,12 @@ jobs:
        shell: bash
        working-directory: /lerobot
    steps:
-      - name: Login to Hugging Face
-        if: env.HF_USER_TOKEN != ''
-        run: |
-          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
-          hf auth whoami
      - name: Verify GPU availability
        run: |
          nvidia-smi
          python -c "import torch; print(f'PyTorch CUDA available: {torch.cuda.is_available()}'); print(f'Number of GPUs: {torch.cuda.device_count()}')"

      - name: Run multi-GPU training tests
-        run: pytest -vv tests/training/
+      # TODO(Steven): Investigate why motors tests are failing in multi-GPU setup
+        run: pytest tests -vv --maxfail=10 --ignore=tests/motors/
+        timeout-minutes: 10
@@ -50,7 +50,7 @@ jobs:
      - name: Set up Python
        uses: actions/setup-python@v6
        with:
-          python-version: '3.12'
+          python-version: '3.10'

      - name: Run pre-commit hooks
        uses: pre-commit/action@v3.0.1 # zizmor: ignore[unpinned-uses]
@@ -22,7 +22,7 @@ on:
 # Sets up the environment variables
 env:
  UV_VERSION: "0.8.0"
-  PYTHON_VERSION: "3.12"
+  PYTHON_VERSION: "3.10"

 jobs:
  # This job builds the Python package and publishes it to PyPI
@@ -45,7 +45,7 @@ jobs:
      - name: Set up Python
        uses: actions/setup-python@v6
        with:
-          python-version: '3.12'
+          python-version: '3.10'

      - name: Extract Version
        id: extract_info
@@ -83,6 +83,14 @@ jobs:
            exit 1
          fi

+      - name: Remove Tags with Git dependencies
+        # TODO(Steven): Temporary patch to remove pi from PyPi 0.4.0 release due to its reliance on git dependencies.
+        run: |
+          echo "::info:: Checking for Git dependencies to remove from pyproject.toml..."
+          grep -E '@ git\+https|lerobot\[pi\]' pyproject.toml | sed 's/^/::warning:: Removing line: /' || true
+          sed -E -i '/@ git\+https|lerobot\[pi\]/d' pyproject.toml
+          echo "::info:: Git dependencies removed. Proceeding with build."
+
      - name: Install build dependencies
        run: python -m pip install build

@@ -29,7 +29,7 @@ permissions:
 # Sets up the environment variables
 env:
  UV_VERSION: "0.8.0"
-  PYTHON_VERSION: "3.12"
+  PYTHON_VERSION: "3.10"
  DOCKER_IMAGE_NAME: huggingface/lerobot-gpu:unbound

 # Ensures that only the latest action is built, canceling older runs.
@@ -48,7 +48,6 @@ jobs:
      MUJOCO_GL: egl
      HF_HOME: /mnt/cache/.cache/huggingface
      HF_LEROBOT_HOME: /mnt/cache/.cache/huggingface/lerobot
-      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    steps:
      - uses: actions/checkout@v6
        with:
@@ -80,11 +79,7 @@ jobs:

      - name: Install lerobot with all extras
        run: uv sync --extra all # TODO(Steven): Make flash-attn optional
-      - name: Login to Hugging Face
-        if: env.HF_USER_TOKEN != ''
-        run: |
-          uv run hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
-          uv run hf auth whoami
+
      - name: Run pytest (all extras)
        run: uv run pytest tests -vv

@@ -142,7 +137,6 @@ jobs:
      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
      TORCH_HOME: /home/user_lerobot/.cache/torch
      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
-      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
    container:
      image: ${{ needs.build-and-push-docker.outputs.image_tag }} # zizmor: ignore[unpinned-images]
      options: --gpus all --shm-size "16gb"
@@ -154,11 +148,6 @@ jobs:
        shell: bash
        working-directory: /lerobot
    steps:
-      - name: Login to Hugging Face
-        if: env.HF_USER_TOKEN != ''
-        run: |
-          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
-          hf auth whoami
      - name: Run pytest on GPU
        run: pytest tests -vv
      - name: Run end-to-end tests
@@ -13,7 +13,7 @@
 # limitations under the License.

 default_language_version:
-    python: python3.12
+    python: python3.10

 exclude: "tests/artifacts/.*\\.safetensors$"

@@ -55,7 +55,7 @@ repos:
    rev: v3.21.0
    hooks:
    -   id: pyupgrade
-        args: [--py312-plus]
+        args: [--py310-plus]

  ##### Markdown Quality #####
  - repo: https://github.com/rbubley/mirrors-prettier
@@ -1,25 +0,0 @@
-# AI Usage Policy
-
-The LeRobot project welcomes contributions from everyone, and we have a few guidelines regarding AI usage to ensure high code quality, clear communication, and a healthy open-source ecosystem:
-
- **Please disclose significant AI assistance.** If you used AI tools (e.g., Copilot, Claude, Cursor, ChatGPT) to generate a substantial portion of your code or text, let us know in your PR description. Transparency helps us review your changes more effectively.
- **Own your code (The Human-in-the-Loop).** You must fully understand all the changes you are proposing. If you cannot explain what your AI-assisted code does or how it interacts with LeRobot's broader architecture, please take the time to learn and test it before submitting.
- **Keep issues and discussions focused.** You are welcome to use AI to help draft issues or PR descriptions, but please review and edit them carefully before posting. AI can often be overly verbose; trimming the noise and getting straight to the point helps our maintainers address your needs faster.
-
-Our core maintainers also use AI tools to aid their workflows, but they do so while bringing deep contextual knowledge of the LeRobot codebase to validate the output. We ask all contributors to apply that same level of rigor.
-
-## Remember the Human Maintainers
-
-Please remember that LeRobot is maintained by a dedicated team of humans.
-
-Every discussion, issue, and pull request is read and reviewed by real people. While AI tools can generate thousands of lines of code in seconds, reviewing that code still takes human time and energy. Submitting unverified or low-effort AI output puts an unfair burden on our maintainers.
-
-Today, the quality of the AI output still heavily depends on the developer driving the tool. We ask that you respect our maintainers' time by thoroughly vetting, testing, and refining your submissions.
-
-## AI is Welcome Here
-
-LeRobot operates at the cutting edge of AI and robotics, and many of our maintainers actively embrace AI coding assistants as valuable productivity tools. We are a pro-AI project!
-
-Our reason for having an AI policy is not an anti-AI stance. Rather, it exists to ensure that AI is used to enhance human contributions, not replace them with unverified noise. It's about how the tools are used, not the tools themselves.
-
-We value the unique human insight you bring to the LeRobot community. Let AI empower your workflow, but always let your own judgment take the wheel.
@@ -2,7 +2,7 @@

 Everyone is welcome to contribute, and we value everybody's contribution. Code is not the only way to help the community. Answering questions, helping others, reaching out, and improving the documentation are immensely valuable.

-Whichever way you choose to contribute, please be mindful to respect our [code of conduct](https://github.com/huggingface/lerobot/blob/main/CODE_OF_CONDUCT.md) and our [AI policy](https://github.com/huggingface/lerobot/blob/main/AI_POLICY.md).
+Whichever way you choose to contribute, please be mindful to respect our [code of conduct](./CODE_OF_CONDUCT.md).

 ## Ways to Contribute

@@ -32,7 +32,7 @@ git remote add upstream https://github.com/huggingface/lerobot.git

 ### 2. Environment Installation

-Please follow our [Installation Guide](https://huggingface.co/docs/lerobot/installation) for the environment setup & installation from source.
+Please follow our [Installation Guide](./docs/source/installation.mdx) for the environment setup & installation from source.

 ## Running Tests & Quality Checks

@@ -75,8 +75,8 @@ pytest -sv tests/test_specific_feature.py

 Use the templates for required fields and examples.

- **Issues:** Follow the [ticket template](https://github.com/huggingface/lerobot/blob/main/.github/ISSUE_TEMPLATE/bug-report.yml).
- **Pull requests:** Rebase on `upstream/main`, use a descriptive branch (don't work on `main`), run `pre-commit` and tests locally, and follow the [PR template](https://github.com/huggingface/lerobot/blob/main/.github/PULL_REQUEST_TEMPLATE.md).
+- **Issues:** Follow the [ticket template](./.github/ISSUE_TEMPLATE/bug-report.yml).
+- **Pull requests:** Rebase on `upstream/main`, use a descriptive branch (don't work on `main`), run `pre-commit` and tests locally, and follow the [PR template](./.github/PULL_REQUEST_TEMPLATE.md).

 One member of the LeRobot team will then review your contribution.

@@ -1,3 +1,2 @@
 include src/lerobot/templates/lerobot_modelcard_template.md
 include src/lerobot/datasets/card_template.md
-include src/lerobot/envs/metaworld_config.json
@@ -135,7 +135,7 @@ Learn how to implement your own simulation environment or benchmark and distribu

 ## Citation

-If you use LeRobot in your project, please cite the GitHub repository to acknowledge the ongoing development and contributors:
+If you use LeRobot in your research, please cite:

 ```bibtex
@misc{cadene2024lerobot,
@@ -146,26 +146,9 @@ If you use LeRobot in your project, please cite the GitHub repository to acknowl
 }
 ```

-If you are referencing our research or the academic paper, please also cite our ICLR publication:
-
-<details>
-<summary><b>ICLR 2026 Paper</b></summary>
-
-```bibtex
-@inproceedings{cadenelerobot,
-  title={LeRobot: An Open-Source Library for End-to-End Robot Learning},
-  author={Cadene, Remi and Alibert, Simon and Capuano, Francesco and Aractingi, Michel and Zouitine, Adil and Kooijmans, Pepijn and Choghari, Jade and Russi, Martino and Pascal, Caroline and Palma, Steven and Shukor, Mustafa and Moss, Jess and Soare, Alexander and Aubakirova, Dana and Lhoest, Quentin and Gallou\'edec, Quentin and Wolf, Thomas},
-  booktitle={The Fourteenth International Conference on Learning Representations},
-  year={2026},
-  url={https://arxiv.org/abs/2602.22818}
-}
-```
-
-</details>
-
 ## Contribute

-We welcome contributions from everyone in the community! To get started, please read our [CONTRIBUTING.md](https://github.com/huggingface/lerobot/blob/main/CONTRIBUTING.md) guide. Whether you're adding a new feature, improving documentation, or fixing a bug, your help and feedback are invaluable. We're incredibly excited about the future of open-source robotics and can't wait to work with you on what's next—thank you for your support!
+We welcome contributions from everyone in the community! To get started, please read our [CONTRIBUTING.md](./CONTRIBUTING.md) guide. Whether you're adding a new feature, improving documentation, or fixing a bug, your help and feedback are invaluable. We're incredibly excited about the future of open-source robotics and can't wait to work with you on what's next—thank you for your support!

 <p align="center">
  <img alt="SO101 Video" src="./media/readme/so100_video.webp" width="640px">
@@ -0,0 +1,134 @@
+# Action tokenizer benchmark
+
+## Questions
+
+What is the trade-off between:
+
+- **Compression**: how many tokens are needed to represent an action chunk (e.g. horizon × action_dim floats)?
+- **Reconstruction quality**: how well does encode-then-decode preserve the original actions?
+- **Speed**: how long does encoding and decoding take per chunk?
+
+How to choose an action tokenizer?
+
+- Which tokenizer architecture (e.g. dct + BPE, DCT + BPE)?
+- Which **action horizon** and **encoded dimensions** to use?
+- Which **normalization** (QUANTILES, MEAN_STD, MIN_MAX) and **delta transform** (relative vs absolute actions)?
+- How do reconstruction error and compression ratio vary across datasets and tokenizer settings?
+
+This benchmark loads action chunks from a LeRobot dataset using the same pipeline as `lerobot-train-tokenizer`, runs a trained action tokenizer in encode/decode mode, and reports reconstruction error, compression stats, and timing. Results are saved as JSON under `outputs/` for comparison and analysis.
+
+## Variables
+
+**Dataset & chunking**
+
+- **repo_id**: LeRobot dataset (e.g. `lerobot/pusht`). Action statistics and normalization are taken from the dataset metadata when available.
+- **action_horizon**: Number of future steps per action chunk (must match the tokenizer’s training).
+- **encoded_dims**: Dimension ranges to encode (e.g. `0:6` or `0:6,7:14`). Must match the tokenizer.
+- **max_episodes**: Cap on episodes to load (default: all).
+- **sample_fraction**: Fraction of chunks to sample per episode (default `0.2`) to keep runtime manageable.
+
+**Transform & normalization**
+
+- **normalization_mode**: `IDENTITY`, `MEAN_STD`, `MIN_MAX`, `QUANTILES`, `QUANTILE10`. Should match the tokenizer’s training.
+- **delta_dims**: Comma-separated dimension indices for delta (relative) transform.
+- **use_delta_transform**: Whether to convert actions to relative to current state for those dimensions.
+- **state_key**: Dataset key for state (e.g. `observation.state`) used when applying delta transform.
+
+**Tokenizer & evaluation**
+
+- **action_tokenizer_path**: Path or HuggingFace repo id of the trained tokenizer (e.g. `outputs/wavetoken`).
+- **max_chunks_for_reconstruction**: Max number of chunks to use for reconstruction and timing (default `500`) to limit runtime.
+
+### Main parameters
+
+| parameter                        | default                      | description                                      |
+| -------------------------------- | ---------------------------- | ------------------------------------------------ |
+| **action_tokenizer_path**        | (required)                   | Path or Hub id of the trained action tokenizer.  |
+| **repo_id**                      | (required)                   | LeRobot dataset repo id.                         |
+| **action_horizon**               | `10`                         | Future steps per chunk.                          |
+| **encoded_dims**                 | `0:6`                        | Dimension ranges to encode (e.g. `0:6,7:14`).   |
+| **normalization_mode**           | `QUANTILES`                  | Normalization mode for actions.                  |
+| **max_episodes**                 | all                          | Max episodes to load.                            |
+| **sample_fraction**              | `0.2`                        | Fraction of chunks sampled per episode.          |
+| **max_chunks_for_reconstruction**| `500`                        | Chunks used for reconstruction and timing.       |
+| **output_dir**                   | `outputs/action_tokenizer_benchmark` | Directory for results JSON.              |
+
+## Metrics
+
+**Reconstruction (lower is better)**
+
+- **reconstruction_mae**: Mean absolute error between original and decoded action chunks.
+- **reconstruction_mse**: Mean squared error.
+- **reconstruction_rmse**: Root mean squared error.
+- **reconstruction_max_abs_error**: Maximum absolute error over all dimensions and samples.
+- **per_dimension_mae**: MAE per action dimension (list of length `action_dim`).
+
+**Compression**
+
+- **compression_ratio**: Ratio (action_horizon × action_dim) / mean number of tokens. Higher means more compression.
+- **mean_token_length**, **std_token_length**: Mean and standard deviation of token count per chunk.
+- **min_token_length**, **max_token_length**: Min and max token count.
+- **p50_token_length**, **p99_token_length**: 50th and 99th percentile token counts.
+
+**Timing (seconds per chunk)**
+
+- **mean_encode_time_sec**: Mean time to encode one chunk.
+- **mean_decode_time_sec**: Mean time to decode one chunk.
+
+The JSON output also includes **num_chunks_evaluated** and **total_chunks_available** for context.
+
+## How the benchmark works
+
+1. **Load dataset**: LeRobot dataset is loaded for the given `repo_id` and `root`.
+2. **Build action chunks**: For each episode (up to `max_episodes`), action chunks are built with the same logic as `lerobot-train-tokenizer`: sliding window of length `action_horizon`, optional delta transform, and per-episode sampling with `sample_fraction`.
+3. **Extract and normalize**: Only `encoded_dims` are kept. Normalization is applied using the dataset’s action stats when available, according to `normalization_mode`.
+4. **Encode / decode**: A random sample of chunks (size `max_chunks_for_reconstruction`) is encoded and then decoded with the tokenizer. Encode and decode times are recorded per chunk.
+5. **Compute metrics**: Reconstruction metrics are computed between original and decoded chunks; compression and timing stats are aggregated.
+6. **Save results**: A JSON file is written to `output_dir` with name `{timestamp}_{repo_id}_action_tokenizer_results.json`, containing the full config and all metrics.
+
+The pipeline (chunking, dimensions, normalization, delta) must match how the tokenizer was trained; otherwise reconstruction error can be large or the tokenizer may raise.
+
+## Caveats
+
+- The tokenizer’s **action_horizon** and **action_dim** (and optionally DCT settings) are fixed at training time. The benchmark infers dimensions from the dataset and encoded dims; the tokenizer path must correspond to a model trained with the same horizon and encoded dimensions.
+- Reconstruction is evaluated in **normalized space** (the same space the tokenizer sees). For interpretation in raw action space, you would need to invert normalization outside this script.
+- Only one tokenizer and one dataset are evaluated per run. To compare tokenizers or datasets, run the script multiple times and compare the saved JSON files.
+
+## Example
+
+Quick run with a local tokenizer and a small number of episodes:
+
+```bash
+python benchmarks/tokens/run_action_tokenizer_benchmark.py \
+    --action-tokenizer-path=outputs/wavetoken \
+    --repo-id=lerobot/pusht \
+    --action-horizon=10 \
+    --max-episodes=50 \
+    --output-dir=outputs/action_tokenizer_benchmark
+```
+
+With delta transform and custom encoded dimensions:
+
+```bash
+python benchmarks/tokens/run_action_tokenizer_benchmark.py \
+    --action-tokenizer-path=outputs/wavetoken \
+    --repo-id=lerobot/pusht \
+    --action-horizon=10 \
+    --encoded-dims=0:6,7:14 \
+    --delta-dims=0,1,2,3,4,5 \
+    --use-delta-transform \
+    --normalization-mode=QUANTILES \
+    --max-chunks-for-reconstruction=500 \
+    --output-dir=outputs/action_tokenizer_benchmark
+```
+
+Results are written to e.g. `outputs/action_tokenizer_benchmark/2026-02-12_14-30-00_lerobot_pusht_action_tokenizer_results.json`.
+
+## Results
+
+Results are stored as JSON in the directory given by `--output-dir` (default: `outputs/action_tokenizer_benchmark`). Each file contains:
+
+- **config**: All script arguments (tokenizer path, repo_id, action_horizon, encoded_dims, normalization_mode, etc.) for reproducibility.
+- **metrics**: All reconstruction, compression, and timing metrics described above.
+
+To compare runs, load and diff or aggregate these JSON files with your own scripts or notebooks.
@@ -0,0 +1,442 @@
+#!/usr/bin/env python
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Benchmark action tokenization: reconstruction error, compression ratio, and timing.
+
+Loads action chunks from a LeRobot dataset, encodes/decodes them with a trained action
+tokenizer, and reports:
+- Reconstruction: MAE, MSE, RMSE, max absolute error, per-dimension MAE
+- Jerk: mean absolute jerk (original and reconstructed), jerk reconstruction MAE
+- Compression: ratio (input size / mean tokens), token length stats
+- Timing: mean encode/decode time per chunk
+
+Results are saved to outputs/action_tokenizer_benchmark/<timestamp>_results.json.
+
+Example:
+
+```bash
+python benchmarks/tokens/run_action_tokenizer_benchmark.py \
+    --action-tokenizer-path=outputs/wavetoken \
+    --repo-id=lerobot/pusht \
+    --action-horizon=10 \
+    --max-episodes=50 \
+    --output-dir=outputs/action_tokenizer_benchmark
+```
+"""
+
+import argparse
+import json
+import time
+from pathlib import Path
+
+import numpy as np
+
+from lerobot.configs.types import NormalizationMode
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.utils.constants import ACTION, OBS_STATE
+
+# Optional: use same helpers as train script if we want to avoid duplication
+from lerobot.scripts.lerobot_train_tokenizer import (
+    apply_normalization,
+    process_episode,
+)
+
+
+def load_action_chunks(
+    repo_id: str,
+    root: str | None,
+    action_horizon: int,
+    max_episodes: int | None,
+    sample_fraction: float,
+    encoded_dims: str,
+    delta_dims: str | None,
+    use_delta_transform: bool,
+    state_key: str,
+    normalization_mode: NormalizationMode,
+):
+    """Load and normalize action chunks from a LeRobot dataset (same pipeline as training)."""
+    dataset = LeRobotDataset(repo_id=repo_id, root=root)
+    num_episodes = dataset.num_episodes
+    if max_episodes is not None:
+        num_episodes = min(max_episodes, num_episodes)
+
+    # Parse encoded dims
+    encoded_dim_ranges = []
+    for range_str in encoded_dims.split(","):
+        start, end = map(int, range_str.strip().split(":"))
+        encoded_dim_ranges.append((start, end))
+    total_encoded_dims = sum(end - start for start, end in encoded_dim_ranges)
+
+    delta_dim_list = None
+    if delta_dims is not None and delta_dims.strip():
+        delta_dim_list = [int(d.strip()) for d in delta_dims.split(",")]
+
+    all_chunks = []
+    for ep_idx in range(num_episodes):
+        chunks = process_episode(
+            (
+                dataset,
+                ep_idx,
+                action_horizon,
+                delta_dim_list,
+                sample_fraction,
+                state_key,
+                use_delta_transform,
+            )
+        )
+        if chunks is not None:
+            all_chunks.append(chunks)
+
+    if not all_chunks:
+        raise ValueError("No action chunks collected. Check action_horizon and dataset.")
+
+    all_chunks = np.concatenate(all_chunks, axis=0)
+
+    # Extract encoded dimensions only
+    encoded_chunks = []
+    for start, end in encoded_dim_ranges:
+        encoded_chunks.append(all_chunks[:, :, start:end])
+    encoded_chunks = np.concatenate(encoded_chunks, axis=-1)
+
+    # Normalize
+    norm_stats = dataset.meta.stats
+    if norm_stats is not None and ACTION in norm_stats:
+        action_stats = norm_stats[ACTION]
+        encoded_dim_indices = []
+        for start, end in encoded_dim_ranges:
+            encoded_dim_indices.extend(range(start, end))
+        encoded_dim_indices = np.array(encoded_dim_indices)
+        encoded_stats = {}
+        for stat_name, stat_values in action_stats.items():
+            if isinstance(stat_values, (list, np.ndarray)):
+                stat_array = np.array(stat_values)
+                if len(stat_array) > max(encoded_dim_indices):
+                    encoded_stats[stat_name] = stat_array[encoded_dim_indices]
+        if encoded_stats:
+            try:
+                encoded_chunks = apply_normalization(
+                    encoded_chunks, encoded_stats, normalization_mode, eps=1e-8
+                )
+            except ValueError:
+                pass
+
+    return encoded_chunks, total_encoded_dims, action_horizon, dataset.repo_id
+
+
+def compute_reconstruction_metrics(original: np.ndarray, reconstructed: np.ndarray):
+    """Compute reconstruction error metrics (original and reconstructed same shape [N, T, D])."""
+    diff = reconstructed - original
+    mae = float(np.mean(np.abs(diff)))
+    mse = float(np.mean(diff**2))
+    rmse = float(np.sqrt(mse))
+    max_abs_err = float(np.max(np.abs(diff)))
+
+    # Per-dimension MAE (over N and T)
+    per_dim_mae = np.mean(np.abs(diff), axis=(0, 1))
+    per_dim_mae = per_dim_mae.tolist()
+
+    return {
+        "reconstruction_mae": mae,
+        "reconstruction_mse": mse,
+        "reconstruction_rmse": rmse,
+        "reconstruction_max_abs_error": max_abs_err,
+        "per_dimension_mae": per_dim_mae,
+    }
+
+
+def compute_jerk_metrics(original: np.ndarray, reconstructed: np.ndarray) -> dict:
+    """Compute jerk (3rd derivative of action w.r.t. time) metrics.
+
+    Args:
+        original: Action chunks [N, T, D].
+        reconstructed: Reconstructed action chunks [N, T, D].
+
+    Returns:
+        Dict with mean absolute jerk for original, reconstructed, and jerk reconstruction MAE.
+    """
+    # Jerk = 3rd discrete difference along time axis; need T >= 4
+    if original.shape[1] < 4:
+        return {}
+    jerk_orig = np.diff(original, n=3, axis=1)  # (N, T-3, D)
+    jerk_recon = np.diff(reconstructed, n=3, axis=1)
+    mae_jerk_orig = float(np.mean(np.abs(jerk_orig)))
+    mae_jerk_recon = float(np.mean(np.abs(jerk_recon)))
+    jerk_reconstruction_mae = float(np.mean(np.abs(jerk_recon - jerk_orig)))
+    return {
+        "jerk_mae_original": mae_jerk_orig,
+        "jerk_mae_reconstructed": mae_jerk_recon,
+        "jerk_reconstruction_mae": jerk_reconstruction_mae,
+    }
+
+
+def run_benchmark(
+    action_chunks: np.ndarray,
+    action_horizon: int,
+    action_dim: int,
+    tokenizer_path: str,
+    max_chunks_for_reconstruction: int | None = 500,
+):
+    """Encode/decode action chunks and compute metrics."""
+    from transformers import AutoProcessor
+
+    processor = AutoProcessor.from_pretrained(tokenizer_path, trust_remote_code=True)
+
+    n_chunks = len(action_chunks)
+    sample_size = n_chunks
+    if max_chunks_for_reconstruction is not None:
+        sample_size = min(max_chunks_for_reconstruction, n_chunks)
+    rng = np.random.RandomState(42)
+    indices = rng.choice(n_chunks, size=sample_size, replace=False)
+    sample_chunks = action_chunks[indices]
+
+    # Encode
+    token_lengths = []
+    encode_times = []
+    all_tokens = []
+    for i in range(len(sample_chunks)):
+        chunk = sample_chunks[i : i + 1]
+        t0 = time.perf_counter()
+        tokens = processor(chunk)[0]
+        encode_times.append(time.perf_counter() - t0)
+        if isinstance(tokens, list):
+            token_lengths.append(len(tokens))
+            all_tokens.append(tokens)
+        else:
+            n = tokens.shape[0] if hasattr(tokens, "shape") else len(tokens)
+            token_lengths.append(n)
+            all_tokens.append(tokens.tolist() if hasattr(tokens, "tolist") else list(tokens))
+
+    # Decode (processor keeps time_horizon/action_dim from encode)
+    decoded_list = []
+    decode_times = []
+    for i, tok_list in enumerate(all_tokens):
+        t0 = time.perf_counter()
+        recon = processor.decode(
+            [tok_list],
+            time_horizon=action_horizon,
+            action_dim=action_dim,
+        )
+        decode_times.append(time.perf_counter() - t0)
+        decoded_list.append(recon)
+    decoded = np.concatenate(decoded_list, axis=0)
+
+    # Reconstruction metrics
+    metrics = compute_reconstruction_metrics(sample_chunks, decoded)
+
+    # Jerk metrics (3rd derivative along time)
+    jerk_metrics = compute_jerk_metrics(sample_chunks, decoded)
+    metrics.update(jerk_metrics)
+
+    # Compression
+    token_lengths = np.array(token_lengths)
+    input_size = action_horizon * action_dim
+    compression_ratio = input_size / float(np.mean(token_lengths))
+    metrics["compression_ratio"] = compression_ratio
+    metrics["mean_token_length"] = float(np.mean(token_lengths))
+    metrics["std_token_length"] = float(np.std(token_lengths))
+    metrics["min_token_length"] = int(np.min(token_lengths))
+    metrics["max_token_length"] = int(np.max(token_lengths))
+    metrics["p50_token_length"] = float(np.percentile(token_lengths, 50))
+    metrics["p99_token_length"] = float(np.percentile(token_lengths, 99))
+
+    # Timing (seconds per chunk)
+    metrics["mean_encode_time_sec"] = float(np.mean(encode_times))
+    metrics["mean_decode_time_sec"] = float(np.mean(decode_times))
+    metrics["num_chunks_evaluated"] = sample_size
+    metrics["total_chunks_available"] = n_chunks
+
+    return metrics
+
+
+def main(
+    action_tokenizer_path: str,
+    repo_id: str,
+    root: str | None = None,
+    action_horizon: int = 10,
+    max_episodes: int | None = 100,
+    sample_fraction: float = 0.2,
+    encoded_dims: str = "0:6",
+    delta_dims: str | None = None,
+    use_delta_transform: bool = False,
+    state_key: str = OBS_STATE,
+    normalization_mode: str = "QUANTILES",
+    max_chunks_for_reconstruction: int | None = 500,
+    output_dir: str | None = None,
+):
+    if output_dir is None:
+        output_dir = "outputs/action_tokenizer_benchmark"
+    output_path = Path(output_dir)
+    output_path.mkdir(parents=True, exist_ok=True)
+
+    try:
+        norm_mode = NormalizationMode(normalization_mode)
+    except ValueError:
+        norm_mode = NormalizationMode.QUANTILES
+
+    print("Loading action chunks...")
+    encoded_chunks, action_dim, horizon, _ = load_action_chunks(
+        repo_id=repo_id,
+        root=root,
+        action_horizon=action_horizon,
+        max_episodes=max_episodes,
+        sample_fraction=sample_fraction,
+        encoded_dims=encoded_dims,
+        delta_dims=delta_dims,
+        use_delta_transform=use_delta_transform,
+        state_key=state_key,
+        normalization_mode=norm_mode,
+    )
+    print(f"Loaded {len(encoded_chunks)} chunks, shape {encoded_chunks.shape} (H={horizon}, D={action_dim})")
+
+    print("Running tokenizer benchmark...")
+    metrics = run_benchmark(
+        action_chunks=encoded_chunks,
+        action_horizon=horizon,
+        action_dim=action_dim,
+        tokenizer_path=action_tokenizer_path,
+        max_chunks_for_reconstruction=max_chunks_for_reconstruction,
+    )
+
+    # Attach config for reproducibility
+    results = {
+        "config": {
+            "action_tokenizer_path": action_tokenizer_path,
+            "repo_id": repo_id,
+            "action_horizon": action_horizon,
+            "max_episodes": max_episodes,
+            "sample_fraction": sample_fraction,
+            "encoded_dims": encoded_dims,
+            "delta_dims": delta_dims,
+            "use_delta_transform": use_delta_transform,
+            "state_key": state_key,
+            "normalization_mode": normalization_mode,
+        },
+        "metrics": metrics,
+    }
+
+    timestamp = time.strftime("%Y-%m-%d_%H-%M-%S")
+    safe_repo = repo_id.replace("/", "_")
+    out_file = output_path / f"{timestamp}_{safe_repo}_action_tokenizer_results.json"
+    with open(out_file, "w") as f:
+        json.dump(results, f, indent=2)
+
+    print(f"Results saved to {out_file}")
+    print("Metrics:")
+    for k, v in metrics.items():
+        if isinstance(v, list):
+            print(f"  {k}: (length {len(v)})")
+        else:
+            print(f"  {k}: {v}")
+
+    return results
+
+
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser(
+        description="Benchmark action tokenization (reconstruction error, compression, timing)."
+    )
+    parser.add_argument(
+        "--action-tokenizer-path",
+        type=str,
+        required=True,
+        help="Path or HuggingFace repo id of the trained action tokenizer (e.g. outputs/wavetoken).",
+    )
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        required=True,
+        help="LeRobot dataset repo id (e.g. lerobot/pusht).",
+    )
+    parser.add_argument(
+        "--root",
+        type=str,
+        default=None,
+        help="Root directory for LeRobot datasets.",
+    )
+    parser.add_argument(
+        "--action-horizon",
+        type=int,
+        default=10,
+        help="Number of future steps per action chunk.",
+    )
+    parser.add_argument(
+        "--max-episodes",
+        type=int,
+        default=None,
+        help="Max episodes to use (default: all).",
+    )
+    parser.add_argument(
+        "--sample-fraction",
+        type=float,
+        default=0.2,
+        help="Fraction of chunks to sample per episode.",
+    )
+    parser.add_argument(
+        "--encoded-dims",
+        type=str,
+        default="0:6",
+        help="Dimension ranges to encode (e.g. 0:6,7:14).",
+    )
+    parser.add_argument(
+        "--delta-dims",
+        type=str,
+        default=None,
+        help="Comma-separated dimensions for delta transform.",
+    )
+    parser.add_argument(
+        "--use-delta-transform",
+        action="store_true",
+        help="Apply delta (relative) transform to specified dimensions.",
+    )
+    parser.add_argument(
+        "--state-key",
+        type=str,
+        default=OBS_STATE,
+        help="Dataset key for state (for delta transform).",
+    )
+    parser.add_argument(
+        "--normalization-mode",
+        type=str,
+        default="QUANTILES",
+        choices=[m.value for m in NormalizationMode],
+        help="Normalization mode for actions.",
+    )
+    parser.add_argument(
+        "--max-chunks-for-reconstruction",
+        type=int,
+        default=500,
+        help="Max chunks to use for reconstruction metrics (default: 500).",
+    )
+    parser.add_argument(
+        "--output-dir",
+        type=str,
+        default="outputs/action_tokenizer_benchmark",
+        help="Directory to save results JSON (default: outputs/action_tokenizer_benchmark).",
+    )
+    args = parser.parse_args()
+    main(
+        action_tokenizer_path=args.action_tokenizer_path,
+        repo_id=args.repo_id,
+        root=args.root,
+        action_horizon=args.action_horizon,
+        max_episodes=args.max_episodes,
+        sample_fraction=args.sample_fraction,
+        encoded_dims=args.encoded_dims,
+        delta_dims=args.delta_dims,
+        use_delta_transform=args.use_delta_transform,
+        state_key=args.state_key,
+        normalization_mode=args.normalization_mode,
+        max_chunks_for_reconstruction=args.max_chunks_for_reconstruction,
+        output_dir=args.output_dir,
+    )
@@ -28,9 +28,9 @@ We don't expect the same optimal settings for a dataset of images from a simulat
 For these reasons, we run this benchmark on four representative datasets:

 - `lerobot/pusht_image`: (96 x 96 pixels) simulation with simple geometric shapes, fixed camera.
- `lerobot/aloha_mobile_shrimp_image`: (480 x 640 pixels) real-world indoor, moving camera.
- `lerobot/paris_street`: (720 x 1280 pixels) real-world outdoor, moving camera.
- `lerobot/kitchen`: (1080 x 1920 pixels) real-world indoor, fixed camera.
+- `aliberts/aloha_mobile_shrimp_image`: (480 x 640 pixels) real-world indoor, moving camera.
+- `aliberts/paris_street`: (720 x 1280 pixels) real-world outdoor, moving camera.
+- `aliberts/kitchen`: (1080 x 1920 pixels) real-world indoor, fixed camera.

 Note: The datasets used for this benchmark need to be image datasets, not video datasets.

@@ -179,7 +179,7 @@ python benchmark/video/run_video_benchmark.py \
    --output-dir outputs/video_benchmark \
    --repo-ids \
        lerobot/pusht_image \
-        lerobot/aloha_mobile_shrimp_image \
+        aliberts/aloha_mobile_shrimp_image \
    --vcodec libx264 libx265 \
    --pix-fmt yuv444p yuv420p \
    --g 2 20 None \
@@ -203,9 +203,9 @@ python benchmark/video/run_video_benchmark.py \
    --output-dir outputs/video_benchmark \
    --repo-ids \
        lerobot/pusht_image \
-        lerobot/aloha_mobile_shrimp_image \
-        lerobot/paris_street \
-        lerobot/kitchen \
+        aliberts/aloha_mobile_shrimp_image \
+        aliberts/paris_street \
+        aliberts/kitchen \
    --vcodec libx264 libx265 \
    --pix-fmt yuv444p yuv420p \
    --g 1 2 3 4 5 6 10 15 20 40 None \
@@ -221,9 +221,9 @@ python benchmark/video/run_video_benchmark.py \
    --output-dir outputs/video_benchmark \
    --repo-ids \
        lerobot/pusht_image \
-        lerobot/aloha_mobile_shrimp_image \
-        lerobot/paris_street \
-        lerobot/kitchen \
+        aliberts/aloha_mobile_shrimp_image \
+        aliberts/paris_street \
+        aliberts/kitchen \
    --vcodec libsvtav1 \
    --pix-fmt yuv420p \
    --g 1 2 3 4 5 6 10 15 20 40 None \
@@ -252,37 +252,37 @@ Since we're using av1 encoding, we're choosing the `pyav` decoder as `video_read

 These tables show the results for `g=2` and `crf=30`, using `timestamps-modes=6_frames` and `backend=pyav`

-| video_images_size_ratio           | vcodec     | pix_fmt |           |           |           |
-| --------------------------------- | ---------- | ------- | --------- | --------- | --------- |
-|                                   | libx264    |         | libx265   |           | libsvtav1 |
-| repo_id                           | yuv420p    | yuv444p | yuv420p   | yuv444p   | yuv420p   |
-| lerobot/pusht_image               | **16.97%** | 17.58%  | 18.57%    | 18.86%    | 22.06%    |
-| lerobot/aloha_mobile_shrimp_image | 2.14%      | 2.11%   | 1.38%     | **1.37%** | 5.59%     |
-| lerobot/paris_street              | 2.12%      | 2.13%   | **1.54%** | **1.54%** | 4.43%     |
-| lerobot/kitchen                   | 1.40%      | 1.39%   | **1.00%** | **1.00%** | 2.52%     |
+| video_images_size_ratio            | vcodec     | pix_fmt |           |           |           |
+| ---------------------------------- | ---------- | ------- | --------- | --------- | --------- |
+|                                    | libx264    |         | libx265   |           | libsvtav1 |
+| repo_id                            | yuv420p    | yuv444p | yuv420p   | yuv444p   | yuv420p   |
+| lerobot/pusht_image                | **16.97%** | 17.58%  | 18.57%    | 18.86%    | 22.06%    |
+| aliberts/aloha_mobile_shrimp_image | 2.14%      | 2.11%   | 1.38%     | **1.37%** | 5.59%     |
+| aliberts/paris_street              | 2.12%      | 2.13%   | **1.54%** | **1.54%** | 4.43%     |
+| aliberts/kitchen                   | 1.40%      | 1.39%   | **1.00%** | **1.00%** | 2.52%     |

-| video_images_load_time_ratio      | vcodec  | pix_fmt |          |         |           |
-| --------------------------------- | ------- | ------- | -------- | ------- | --------- |
-|                                   | libx264 |         | libx265  |         | libsvtav1 |
-| repo_id                           | yuv420p | yuv444p | yuv420p  | yuv444p | yuv420p   |
-| lerobot/pusht_image               | 6.45    | 5.19    | **1.90** | 2.12    | 2.47      |
-| lerobot/aloha_mobile_shrimp_image | 11.80   | 7.92    | 0.71     | 0.85    | **0.48**  |
-| lerobot/paris_street              | 2.21    | 2.05    | 0.36     | 0.49    | **0.30**  |
-| lerobot/kitchen                   | 1.46    | 1.46    | 0.28     | 0.51    | **0.26**  |
+| video_images_load_time_ratio       | vcodec  | pix_fmt |          |         |           |
+| ---------------------------------- | ------- | ------- | -------- | ------- | --------- |
+|                                    | libx264 |         | libx265  |         | libsvtav1 |
+| repo_id                            | yuv420p | yuv444p | yuv420p  | yuv444p | yuv420p   |
+| lerobot/pusht_image                | 6.45    | 5.19    | **1.90** | 2.12    | 2.47      |
+| aliberts/aloha_mobile_shrimp_image | 11.80   | 7.92    | 0.71     | 0.85    | **0.48**  |
+| aliberts/paris_street              | 2.21    | 2.05    | 0.36     | 0.49    | **0.30**  |
+| aliberts/kitchen                   | 1.46    | 1.46    | 0.28     | 0.51    | **0.26**  |

-|                                   |          | vcodec   | pix_fmt      |          |           |              |
-| --------------------------------- | -------- | -------- | ------------ | -------- | --------- | ------------ |
-|                                   |          | libx264  |              | libx265  |           | libsvtav1    |
-| repo_id                           | metric   | yuv420p  | yuv444p      | yuv420p  | yuv444p   | yuv420p      |
-| lerobot/pusht_image               | avg_mse  | 2.90E-04 | **2.03E-04** | 3.13E-04 | 2.29E-04  | 2.19E-04     |
-|                                   | avg_psnr | 35.44    | 37.07        | 35.49    | **37.30** | 37.20        |
-|                                   | avg_ssim | 98.28%   | **98.85%**   | 98.31%   | 98.84%    | 98.72%       |
-| lerobot/aloha_mobile_shrimp_image | avg_mse  | 2.76E-04 | 2.59E-04     | 3.17E-04 | 3.06E-04  | **1.30E-04** |
-|                                   | avg_psnr | 35.91    | 36.21        | 35.88    | 36.09     | **40.17**    |
-|                                   | avg_ssim | 95.19%   | 95.18%       | 95.00%   | 95.05%    | **97.73%**   |
-| lerobot/paris_street              | avg_mse  | 6.89E-04 | 6.70E-04     | 4.03E-03 | 4.02E-03  | **3.09E-04** |
-|                                   | avg_psnr | 33.48    | 33.68        | 32.05    | 32.15     | **35.40**    |
-|                                   | avg_ssim | 93.76%   | 93.75%       | 89.46%   | 89.46%    | **95.46%**   |
-| lerobot/kitchen                   | avg_mse  | 2.50E-04 | 2.24E-04     | 4.28E-04 | 4.18E-04  | **1.53E-04** |
-|                                   | avg_psnr | 36.73    | 37.33        | 36.56    | 36.75     | **39.12**    |
-|                                   | avg_ssim | 95.47%   | 95.58%       | 95.52%   | 95.53%    | **96.82%**   |
+|                                    |          | vcodec   | pix_fmt      |          |           |              |
+| ---------------------------------- | -------- | -------- | ------------ | -------- | --------- | ------------ |
+|                                    |          | libx264  |              | libx265  |           | libsvtav1    |
+| repo_id                            | metric   | yuv420p  | yuv444p      | yuv420p  | yuv444p   | yuv420p      |
+| lerobot/pusht_image                | avg_mse  | 2.90E-04 | **2.03E-04** | 3.13E-04 | 2.29E-04  | 2.19E-04     |
+|                                    | avg_psnr | 35.44    | 37.07        | 35.49    | **37.30** | 37.20        |
+|                                    | avg_ssim | 98.28%   | **98.85%**   | 98.31%   | 98.84%    | 98.72%       |
+| aliberts/aloha_mobile_shrimp_image | avg_mse  | 2.76E-04 | 2.59E-04     | 3.17E-04 | 3.06E-04  | **1.30E-04** |
+|                                    | avg_psnr | 35.91    | 36.21        | 35.88    | 36.09     | **40.17**    |
+|                                    | avg_ssim | 95.19%   | 95.18%       | 95.00%   | 95.05%    | **97.73%**   |
+| aliberts/paris_street              | avg_mse  | 6.89E-04 | 6.70E-04     | 4.03E-03 | 4.02E-03  | **3.09E-04** |
+|                                    | avg_psnr | 33.48    | 33.68        | 32.05    | 32.15     | **35.40**    |
+|                                    | avg_ssim | 93.76%   | 93.75%       | 89.46%   | 89.46%    | **95.46%**   |
+| aliberts/kitchen                   | avg_mse  | 2.50E-04 | 2.24E-04     | 4.28E-04 | 4.18E-04  | **1.53E-04** |
+|                                    | avg_psnr | 36.73    | 37.33        | 36.56    | 36.75     | **39.12**    |
+|                                    | avg_ssim | 95.47%   | 95.58%       | 95.52%   | 95.53%    | **96.82%**   |
@@ -24,7 +24,7 @@ ARG OS_VERSION=22.04
 FROM nvidia/cuda:${CUDA_VERSION}-base-ubuntu${OS_VERSION}

 # Define Python version argument
-ARG PYTHON_VERSION=3.12
+ARG PYTHON_VERSION=3.10

 # Configure environment variables
 ENV DEBIAN_FRONTEND=noninteractive \
@@ -85,8 +85,6 @@ RUN if [ "$UNBOUND_DEPS" = "true" ]; then \

 RUN uv pip install --no-cache ".[all]"

-RUN chmod +x /lerobot/.venv/lib/python${PYTHON_VERSION}/site-packages/triton/backends/nvidia/bin/ptxas
-
 # Copy the rest of the application source code
 # Make sure to have the git-LFS files for testing
 COPY --chown=user_lerobot:user_lerobot . .
@@ -18,10 +18,8 @@
 # docker build -f docker/Dockerfile.user -t lerobot-user .
 # docker run -it --rm lerobot-user

-# With USB physical access : docker run -it --device=/dev/ -v /dev/:/dev/ --rm lerobot-user
-
 # Configure the base image
-ARG PYTHON_VERSION=3.12
+ARG PYTHON_VERSION=3.10
 FROM python:${PYTHON_VERSION}-slim

 # Configure environment variables
@@ -29,8 +29,6 @@
    title: Using the Dataset Tools
  - local: dataset_subtask
    title: Using Subtasks in the Dataset
-  - local: streaming_video_encoding
-    title: Streaming Video Encoding
  title: "Datasets"
 - sections:
  - local: act
@@ -88,8 +88,5 @@ lerobot-record \
  --dataset.repo_id=${HF_USER}/eval_act_your_dataset \
  --dataset.num_episodes=10 \
  --dataset.single_task="Your task description" \
-  --dataset.streaming_encoding=true \
-  --dataset.encoder_threads=2 \
-  # --dataset.vcodec=auto \
  --policy.path=${HF_USER}/act_policy
 ```
@@ -48,7 +48,7 @@ python -m lerobot.async_inference.robot_client \
    --task="dummy" \ # POLICY: The task to run the policy on (`Fold my t-shirt`). Not necessarily defined for all policies, such as `act`
    --policy_type=your_policy_type \ # POLICY: the type of policy to run (smolvla, act, etc)
    --pretrained_name_or_path=user/model \ # POLICY: the model name/path on server to the checkpoint to run (e.g., lerobot/smolvla_base)
-    --policy_device=mps \ # POLICY: the device to run the policy on, on the server (cuda, mps, xpu, cpu)
+    --policy_device=mps \ # POLICY: the device to run the policy on, on the server
    --actions_per_chunk=50 \ # POLICY: the number of actions to output at once
    --chunk_size_threshold=0.5 \ # CLIENT: the threshold for the chunk size before sending a new observation to the server
    --aggregate_fn_name=weighted_average \ # CLIENT: the function to aggregate actions on overlapping portions
@@ -32,7 +32,7 @@ version = "0.1.0"
 dependencies = [
    # your policy-specific dependencies
 ]
-requires-python = ">= 3.12"
+requires-python = ">= 3.11"

 [build-system]
 build-backend = # your-build-backend
@@ -82,7 +82,7 @@ Create your policy implementation by inheriting from LeRobot's base `PreTrainedP
 # modeling_my_custom_policy.py
 import torch
 import torch.nn as nn
-from typing import Any
+from typing import Dict, Any

 from lerobot.policies.pretrained import PreTrainedPolicy
 from .configuration_my_custom_policy import MyCustomPolicyConfig
@@ -91,7 +91,7 @@ class MyCustomPolicy(PreTrainedPolicy):
    config_class = MyCustomPolicyConfig
    name = "my_custom_policy"

-    def __init__(self, config: MyCustomPolicyConfig, dataset_stats: dict[str, Any] = None):
+    def __init__(self, config: MyCustomPolicyConfig, dataset_stats: Dict[str, Any] = None):
        super().__init__(config, dataset_stats)
        ...
 ```
@@ -102,7 +102,7 @@ Create processor functions:

 ```python
 # processor_my_custom_policy.py
-from typing import Any
+from typing import Dict, Any
 import torch


@@ -13,7 +13,7 @@ The EarthRover Mini Plus is a fully open source mobile robot that connects throu
 ### Hardware

 - EarthRover Mini robot
- Computer with Python 3.12 or newer
+- Computer with Python 3.10 or newer
 - Internet connection

 ### Setting Up the Frodobots SDK
@@ -170,13 +170,13 @@ Once you can drive the robot well, you can start recording data to train AI mode
 We use Hugging Face to store your data online. First, log in with your token from [Hugging Face settings](https://huggingface.co/settings/tokens):

 ```bash
-hf auth login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
+huggingface-cli login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
 ```

 Store your Hugging Face username:

 ```bash
-HF_USER=$(hf auth whoami | awk -F': *' 'NR==1 {print $2}')
+HF_USER=$(huggingface-cli whoami | head -n 1)
 echo $HF_USER
 ```

@@ -185,16 +185,13 @@ echo $HF_USER
 Use the standard recording command:

 ```bash
-lerobot-record \
+python src/lerobot/scripts/lerobot_record.py \
    --robot.type=earthrover_mini_plus \
    --teleop.type=keyboard_rover \
    --dataset.repo_id=your_username/dataset_name \
    --dataset.num_episodes=2 \
    --dataset.fps=10 \
    --dataset.single_task="Navigate around obstacles" \
-    --dataset.streaming_encoding=true \
-    --dataset.encoder_threads=2 \
-    # --dataset.vcodec=auto \
    --display_data=true
 ```

@@ -155,10 +155,10 @@ Upload your repository to Hugging Face:
 pip install huggingface_hub

 # Login to Hugging Face
-hf auth login
+huggingface-cli login

 # Create a new repository
-hf repo create my-org/my-custom-env
+huggingface-cli repo create my-custom-env --type space --org my-org

 # Initialize git and push
 git init
@@ -120,12 +120,9 @@ lerobot-record \
  --display_data=true \
  --dataset.repo_id=<user>/eval_groot-bimanual  \
  --dataset.num_episodes=10 \
-  --dataset.single_task="Grab and handover the red cube to the other arm" \
-  --dataset.streaming_encoding=true \
-  --dataset.encoder_threads=2 \
-  # --dataset.vcodec=auto \
-  --policy.path=<user>/groot-bimanual \ # your trained model
-  --dataset.episode_time_s=30 \
+  --dataset.single_task="Grab and handover the red cube to the other arm"
+  --policy.path=<user>/groot-bimanual # your trained model
+  --dataset.episode_time_s=30
  --dataset.reset_time_s=10
 ```

@@ -224,15 +224,12 @@ lerobot-record \
    --teleop.port=/dev/tty.usbmodem1201 \
    --teleop.id=right \
    --teleop.side=right \
-    --dataset.repo_id=<USER>/hand_record_test_with_video_data \
+    --dataset.repo_id=nepyope/hand_record_test_with_video_data \
    --dataset.single_task="Hand recording test with video data" \
    --dataset.num_episodes=1 \
    --dataset.episode_time_s=5 \
    --dataset.push_to_hub=true \
    --dataset.private=true \
-    --dataset.streaming_encoding=true \
-    --dataset.encoder_threads=2 \
-    # --dataset.vcodec=auto \
    --display_data=true
 ```

@@ -244,7 +241,7 @@ lerobot-replay \
    --robot.port=/dev/tty.usbmodem58760432281 \
    --robot.id=right \
    --robot.side=right \
-    --dataset.repo_id=<USER>/hand_record_test_with_camera \
+    --dataset.repo_id=nepyope/hand_record_test_with_camera \
    --dataset.episode=0
 ```

@@ -252,13 +249,13 @@ lerobot-replay \

 ```bash
 lerobot-train \
-  --dataset.repo_id=<USER>/hand_record_test_with_video_data \
+  --dataset.repo_id=nepyope/hand_record_test_with_video_data \
  --policy.type=act \
  --output_dir=outputs/train/hopejr_hand \
  --job_name=hopejr \
  --policy.device=mps \
  --wandb.enable=true \
-  --policy.repo_id=<USER>/hand_test_policy
+  --policy.repo_id=nepyope/hand_test_policy
 ```

 ### Evaluate
@@ -273,11 +270,8 @@ lerobot-record \
  --robot.side=right \
  --robot.cameras='{"main": {"type": "opencv", "index_or_path": 0, "width": 640, "height": 480, "fps": 30}}' \
  --display_data=false \
-  --dataset.repo_id=<USER>/eval_hopejr \
+  --dataset.repo_id=nepyope/eval_hopejr \
  --dataset.single_task="Evaluate hopejr hand policy" \
  --dataset.num_episodes=10 \
-  --dataset.streaming_encoding=true \
-  --dataset.encoder_threads=2 \
-  # --dataset.vcodec=auto \
  --policy.path=outputs/train/hopejr_hand/checkpoints/last/pretrained_model
 ```
@@ -159,13 +159,13 @@ We use the Hugging Face hub features for uploading your dataset. If you haven't
 Add your token to the CLI by running this command:

 ```bash
-hf auth login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
+huggingface-cli login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
 ```

 Then store your Hugging Face repository name in a variable:

 ```bash
-HF_USER=$(NO_COLOR=1 hf auth whoami | awk -F': *' 'NR==1 {print $2}')
+HF_USER=$(hf auth whoami | head -n 1)
 echo $HF_USER
 ```

@@ -185,10 +185,7 @@ lerobot-record \
    --display_data=true \
    --dataset.repo_id=${HF_USER}/record-test \
    --dataset.num_episodes=5 \
-    --dataset.single_task="Grab the black cube" \
-    --dataset.streaming_encoding=true \
-    # --dataset.vcodec=auto \
-    --dataset.encoder_threads=2
+    --dataset.single_task="Grab the black cube"
 ```
 </hfoption>
 <hfoption id="API example">
@@ -327,7 +324,7 @@ You can look for other LeRobot datasets on the hub by searching for `LeRobot` [t
 You can also push your local dataset to the Hub manually, running:

 ```bash
-hf upload ${HF_USER}/record-test ~/.cache/huggingface/lerobot/{repo-id} --repo-type dataset
+huggingface-cli upload ${HF_USER}/record-test ~/.cache/huggingface/lerobot/{repo-id} --repo-type dataset
 ```

 #### Record function
@@ -491,7 +488,7 @@ If your local computer doesn't have a powerful GPU you could utilize Google Cola
 Once training is done, upload the latest checkpoint with:

 ```bash
-hf upload ${HF_USER}/act_so101_test \
+huggingface-cli upload ${HF_USER}/act_so101_test \
  outputs/train/act_so101_test/checkpoints/last/pretrained_model
 ```

@@ -499,7 +496,7 @@ You can also upload intermediate checkpoints with:

 ```bash
 CKPT=010000
-hf upload ${HF_USER}/act_so101_test${CKPT} \
+huggingface-cli upload ${HF_USER}/act_so101_test${CKPT} \
  outputs/train/act_so101_test/checkpoints/${CKPT}/pretrained_model
 ```

@@ -518,9 +515,6 @@ lerobot-record  \
  --display_data=false \
  --dataset.repo_id=${HF_USER}/eval_so100 \
  --dataset.single_task="Put lego brick into the transparent box" \
-  --dataset.streaming_encoding=true \
-  --dataset.encoder_threads=2 \
-  # --dataset.vcodec=auto \
  # <- Teleop optional if you want to teleoperate in between episodes \
  # --teleop.type=so100_leader \
  # --teleop.port=/dev/ttyACM0 \
@@ -1,8 +1,8 @@
 # Installation

-This guide uses `conda` (via miniforge) to manage environments (recommended). If you prefer another environment manager (e.g. `uv`, `venv`), ensure you have Python >=3.12 and `ffmpeg` installed with the `libsvtav1` encoder, then skip ahead to [Environment Setup](#step-2-environment-setup).
+This guide uses conda (via miniforge) to manage environments. If you prefer another environment manager (e.g. `uv`, `venv`), ensure you have Python >=3.10 and ffmpeg installed with the `libsvtav1` encoder, then skip ahead to [Install LeRobot](#step-3-install-lerobot-).

-## Step 1 (`conda` only): Install [`miniforge`](https://conda-forge.org/download/)
+## Step 1: Install [`miniforge`](https://conda-forge.org/download/)

 ```bash
 wget "https://github.com/conda-forge/miniforge/releases/latest/download/Miniforge3-$(uname)-$(uname -m).sh"
@@ -11,47 +11,22 @@ bash Miniforge3-$(uname)-$(uname -m).sh

 ## Step 2: Environment Setup

-Create a virtual environment with Python 3.12:
+Create a virtual environment with Python 3.10, using conda:

-<!-- prettier-ignore-start -->
-<hfoptions id="create_venv">
-<hfoption id="conda">
 ```bash
-conda create -y -n lerobot python=3.12
+conda create -y -n lerobot python=3.10
 ```
-</hfoption>
-<hfoption id="uv">
+
+Then activate your conda environment, you have to do this each time you open a shell to use lerobot:
+
 ```bash
-uv python install 3.12
-uv venv --python 3.12
-```
-</hfoption>
-</hfoptions>
-<!-- prettier-ignore-end -->
-
-Then activate your virtual environment, you have to do this each time you open a shell to use lerobot:
-
-<!-- prettier-ignore-start -->
-<hfoptions id="activate_venv">
-<hfoption id="conda">```bash
 conda activate lerobot
-```</hfoption>
-<hfoption id="uv">
-```bash
-# Linux/macOSsource
-source .venv/bin/activate
-# Windows PowerShell
-source .venv\Scripts\Activate.ps1
 ```
-</hfoption>
-</hfoptions>
-<!-- prettier-ignore-end -->

 When using `conda`, install `ffmpeg` in your environment:

 ```bash
 conda install ffmpeg -c conda-forge
-ffmpeg -version  # ffmpeg 8.X is not yet supported !
 ```

 > [!TIP]
@@ -65,16 +40,6 @@ ffmpeg -version  # ffmpeg 8.X is not yet supported !
 >
 > - _[On Linux only]_ If you want to bring your own ffmpeg: Install [ffmpeg build dependencies](https://trac.ffmpeg.org/wiki/CompilationGuide/Ubuntu#GettheDependencies) and [compile ffmpeg from source with libsvtav1](https://trac.ffmpeg.org/wiki/CompilationGuide/Ubuntu#libsvtav1), and make sure you use the corresponding ffmpeg binary to your install with `which ffmpeg`.

-> [!NOTE]
-> When installing LeRobot inside WSL (Windows Subsystem for Linux), make sure to install `evdev` with the following command:
->
-> ```bash
-> conda install evdev -c conda-forge
-> ```
-
-> [!IMPORTANT]
-> If you are using `uv` you will have to install `ffmpeg` system-wide (outside of the virtual environment). You rely on `uv` and `torchcodec` ability to dynamically link to the system `ffmpeg`.
-
 ## Step 3: Install LeRobot 🤗

 ### From Source
@@ -88,45 +53,23 @@ cd lerobot

 Then, install the library in editable mode. This is useful if you plan to contribute to the code.

-<!-- prettier-ignore-start -->
-<hfoptions id="install_lerobot_src">
-<hfoption id="conda">
 ```bash
 pip install -e .
 ```
-</hfoption>
-<hfoption id="uv">
-```bash
-uv pip install -e .
-```
-</hfoption>
-</hfoptions>
-<!-- prettier-ignore-end -->

 ### Installation from PyPI

 **Core Library:**
 Install the base package with:

-<!-- prettier-ignore-start -->
-<hfoptions id="install_lerobot_pypi">
-<hfoption id="conda">
 ```bash
 pip install lerobot
 ```
-</hfoption>
-<hfoption id="uv">
-```bash
-uv pip install lerobot
-```
-</hfoption>
-</hfoptions>
-<!-- prettier-ignore-end -->

 _This installs only the default dependencies._

 **Extra Features:**
-To install additional functionality, use one of the following (If you are using `uv`, replace `pip install` with `uv pip install` in the commands below.):
+To install additional functionality, use one of the following:

 ```bash
 pip install 'lerobot[all]'          # All available features
@@ -140,10 +83,13 @@ _Replace `[...]` with your desired features._
 For a full list of optional dependencies, see:
 https://pypi.org/project/lerobot/

+> [!NOTE]
+> For lerobot 0.4.0, if you want to install pi, you will have to do: `pip install "lerobot[pi]@git+https://github.com/huggingface/lerobot.git"`
+
 ### Troubleshooting

 If you encounter build errors, you may need to install additional dependencies: `cmake`, `build-essential`, and `ffmpeg libs`.
-To install these for Linux run:
+To install these for linux run:

 ```bash
 sudo apt-get install cmake build-essential python3-dev pkg-config libavformat-dev libavcodec-dev libavdevice-dev libavutil-dev libswscale-dev libswresample-dev libavfilter-dev
@@ -153,7 +99,7 @@ For other systems, see: [Compiling PyAV](https://pyav.org/docs/develop/overview/

 ## Optional dependencies

-LeRobot provides optional extras for specific functionalities. Multiple extras can be combined (e.g., `.[aloha,feetech]`). For all available extras, refer to `pyproject.toml`. If you are using `uv`, replace `pip install` with `uv pip install` in the commands below.
+LeRobot provides optional extras for specific functionalities. Multiple extras can be combined (e.g., `.[aloha,feetech]`). For all available extras, refer to `pyproject.toml`.

 ### Simulations

@@ -279,13 +279,13 @@ We use the Hugging Face hub features for uploading your dataset. If you haven't
 Add your token to the CLI by running this command:

 ```bash
-hf auth login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
+huggingface-cli login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
 ```

 Then store your Hugging Face repository name in a variable:

 ```bash
-HF_USER=$(hf auth whoami | awk -F': *' 'NR==1 {print $2}')
+HF_USER=$(huggingface-cli whoami | head -n 1)
 echo $HF_USER
 ```

@@ -41,10 +41,7 @@ lerobot-record \
  --display_data=true \
  --dataset.repo_id=${HF_USER}/record-test \
  --dataset.num_episodes=5 \
-  --dataset.single_task="Grab the black cube" \
-  --dataset.streaming_encoding=true \
-  # --dataset.vcodec=auto \
-  --dataset.encoder_threads=2
+  --dataset.single_task="Grab the black cube"
 ```

 See the [recording guide](./il_robots#record-a-dataset) for more details.
@@ -66,13 +66,12 @@ Run on of the examples scripts to teleoperate, record a dataset, replay a datase

 All scripts assume you configured your robot (e.g., SO-100 follower) and set the correct serial port.

-Additionally you need to **copy the URDF of the robot into the examples folder**. For the examples in this tutorial (using SO100/SO101), copy the `SO101` folder from the [SO-ARM100 repo](https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101) into the `examples/phone_to_so100/` directory, so that the URDF file path becomes `examples/phone_to_so100/SO101/so101_new_calib.urdf`.
+Additionally you need to **copy the urdf of the robot to the examples folder**. For the examples in this tutorial (Using SO100/SO101) it is highly recommended to use the urdf in the [SO-ARM100 repo](https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf)

 - Run this example to teleoperate:

  ```bash
-  cd examples/phone_to_so100
-  python teleoperate.py
+  python examples/phone_to_so100/teleoperate.py
  ```

 After running the example:
@@ -85,22 +84,19 @@ Additionally you can customize mapping or safety limits by editing the processor
 - Run this example to record a dataset, which saves absolute end effector observations and actions:

  ```bash
-  cd examples/phone_to_so100
-  python record.py
+  python examples/phone_to_so100/record.py
  ```

 - Run this example to replay recorded episodes:

  ```bash
-  cd examples/phone_to_so100
-  python replay.py
+  python examples/phone_to_so100/replay.py
  ```

 - Run this example to evaluate a pretrained policy:

  ```bash
-  cd examples/phone_to_so100
-  python evaluate.py
+  python examples/phone_to_so100/evaluate.py
  ```

 ### Important pipeline steps and options
@@ -34,6 +34,11 @@ As described by Physical Intelligence, while AI has achieved remarkable success
   pip install -e ".[pi]"
   ```

+   > [!NOTE]
+   > For lerobot 0.4.0, if you want to install pi tag, you will have to do: `pip install "lerobot[pi]@git+https://github.com/huggingface/lerobot.git"`.
+   >
+   > This will be solved in the next patch release
+
 ## Training Data and Capabilities

 π₀ is trained on the largest robot interaction dataset to date, combining three key data sources:
@@ -55,7 +60,7 @@ policy.type=pi0
 For training π₀, you can use the standard LeRobot training script with the appropriate configuration:

 ```bash
-lerobot-train \
+python src/lerobot/scripts/lerobot_train.py \
    --dataset.repo_id=your_dataset \
    --policy.type=pi0 \
    --output_dir=./outputs/pi0_training \
@@ -36,6 +36,11 @@ This diverse training mixture creates a "curriculum" that enables generalization
   pip install -e ".[pi]"
   ```

+   > [!NOTE]
+   > For lerobot 0.4.0, if you want to install pi tag, you will have to do: `pip install "lerobot[pi]@git+https://github.com/huggingface/lerobot.git"`.
+   >
+   > This will be solved in the next patch release
+
 ## Usage

 To use π₀.₅ in your LeRobot configuration, specify the policy type as:
@@ -51,7 +56,7 @@ policy.type=pi05
 Here's a complete training command for finetuning the base π₀.₅ model on your own dataset:

 ```bash
-lerobot-train \
+python src/lerobot/scripts/lerobot_train.py\
    --dataset.repo_id=your_dataset \
    --policy.type=pi05 \
    --output_dir=./outputs/pi05_training \
@@ -43,11 +43,16 @@ This approach can transform **any existing VLM** into a VLA by training it to pr
   pip install -e ".[pi]"
   ```

+   > [!NOTE]
+   > For lerobot 0.4.0, if you want to install the pi tag, you will have to do: `pip install "lerobot[pi]@git+https://github.com/huggingface/lerobot.git"`.
+   >
+   > This will be solved in the next patch release
+
 ## Training a Custom FAST Tokenizer

 You have two options for the FAST tokenizer:

-1. **Use the pre-trained tokenizer**: The `lerobot/fast-action-tokenizer` tokenizer was trained on 1M+ real robot action sequences and works as a general-purpose tokenizer.
+1. **Use the pre-trained tokenizer**: The `physical-intelligence/fast` tokenizer was trained on 1M+ real robot action sequences and works as a general-purpose tokenizer.

 2. **Train your own tokenizer**: For maximum performance on your specific dataset, you can finetune the tokenizer on your own data.

@@ -109,15 +114,15 @@ lerobot-train \

 ### Key Training Parameters

-| Parameter                              | Description                                        | Default                         |
-| -------------------------------------- | -------------------------------------------------- | ------------------------------- |
-| `--policy.gradient_checkpointing=true` | Reduces memory usage significantly during training | `false`                         |
-| `--policy.dtype=bfloat16`              | Use mixed precision training for efficiency        | `float32`                       |
-| `--policy.chunk_size`                  | Number of action steps to predict (action horizon) | `50`                            |
-| `--policy.n_action_steps`              | Number of action steps to execute                  | `50`                            |
-| `--policy.max_action_tokens`           | Maximum number of FAST tokens per action chunk     | `256`                           |
-| `--policy.action_tokenizer_name`       | FAST tokenizer to use                              | `lerobot/fast-action-tokenizer` |
-| `--policy.compile_model=true`          | Enable torch.compile for faster training           | `false`                         |
+| Parameter                              | Description                                        | Default                      |
+| -------------------------------------- | -------------------------------------------------- | ---------------------------- |
+| `--policy.gradient_checkpointing=true` | Reduces memory usage significantly during training | `false`                      |
+| `--policy.dtype=bfloat16`              | Use mixed precision training for efficiency        | `float32`                    |
+| `--policy.chunk_size`                  | Number of action steps to predict (action horizon) | `50`                         |
+| `--policy.n_action_steps`              | Number of action steps to execute                  | `50`                         |
+| `--policy.max_action_tokens`           | Maximum number of FAST tokens per action chunk     | `256`                        |
+| `--policy.action_tokenizer_name`       | FAST tokenizer to use                              | `physical-intelligence/fast` |
+| `--policy.compile_model=true`          | Enable torch.compile for faster training           | `false`                      |

 ## Inference

@@ -159,9 +159,6 @@ lerobot-record \
    --dataset.fps=15 \
    --dataset.push_to_hub=true \
    --dataset.private=true \
-    --dataset.streaming_encoding=true \
-    --dataset.encoder_threads=2 \
-    # --dataset.vcodec=auto \
    --display_data=true
 ```

@@ -201,9 +198,6 @@ lerobot-record \
    --dataset.fps=15 \
    --dataset.push_to_hub=true \
    --dataset.private=true \
-    --dataset.streaming_encoding=true \
-    --dataset.encoder_threads=2 \
-    # --dataset.vcodec=auto \
    --display_data=true
 ```

@@ -269,7 +269,7 @@ This generates visualizations showing video frames with subtask boundaries overl
 Train with **no annotations** - uses linear progress from 0 to 1:

 ```bash
-lerobot-train \
+python src/lerobot/scripts/lerobot_train.py \
  --dataset.repo_id=your-username/your-dataset \
  --policy.type=sarm \
  --policy.annotation_mode=single_stage \
@@ -288,7 +288,7 @@ lerobot-train \
 Train with **dense annotations only** (sparse auto-generated):

 ```bash
-lerobot-train \
+python src/lerobot/scripts/lerobot_train.py \
  --dataset.repo_id=your-username/your-dataset \
  --policy.type=sarm \
  --policy.annotation_mode=dense_only \
@@ -307,7 +307,7 @@ lerobot-train \
 Train with **both sparse and dense annotations**:

 ```bash
-lerobot-train \
+python src/lerobot/scripts/lerobot_train.py \
  --dataset.repo_id=your-username/your-dataset \
  --policy.type=sarm \
  --policy.annotation_mode=dual \
@@ -468,7 +468,7 @@ This script:
 Once you have the progress file, train your policy with RA-BC weighting. The progress file is auto-detected from the dataset path (`sarm_progress.parquet`). Currently PI0, PI0.5 and SmolVLA are supported with RA-BC:

 ```bash
-lerobot-train \
+python src/lerobot/scripts/lerobot_train.py \
  --dataset.repo_id=your-username/your-dataset \
  --policy.type=pi0 \
  --use_rabc=true \
@@ -106,9 +106,6 @@ lerobot-record \
  --dataset.repo_id=${HF_USER}/eval_DATASET_NAME_test \  # <- This will be the dataset name on HF Hub
  --dataset.episode_time_s=50 \
  --dataset.num_episodes=10 \
-  --dataset.streaming_encoding=true \
-  --dataset.encoder_threads=2 \
-  # --dataset.vcodec=auto \
  # <- Teleop optional if you want to teleoperate in between episodes \
  # --teleop.type=so100_leader \
  # --teleop.port=/dev/ttyACM0 \
@@ -1,155 +0,0 @@
-# Streaming Video Encoding Guide
-
-## 1. Overview
-
-Streaming video encoding eliminates the traditional PNG round-trip during video dataset recording. Instead of:
-
-1. Capture frame -> write PNG to disk -> (at episode end) read PNG's -> encode to MP4 -> delete PNG's
-
-Frames can be encoded in real-time during capture:
-
-1. Capture frame -> queue to encoder thread -> encode to MP4 directly
-
-This makes `save_episode()` near-instant (the video is already encoded by the time the episode ends) and removes the blocking wait that previously occurred between episodes, especially with multiple cameras in long episodes.
-
-## 2. Tuning Parameters
-
-| Parameter               | CLI Flag                          | Type          | Default       | Description                                                       |
-| ----------------------- | --------------------------------- | ------------- | ------------- | ----------------------------------------------------------------- |
-| `streaming_encoding`    | `--dataset.streaming_encoding`    | `bool`        | `True`        | Enable real-time encoding during capture                          |
-| `vcodec`                | `--dataset.vcodec`                | `str`         | `"libsvtav1"` | Video codec. `"auto"` detects best HW encoder                     |
-| `encoder_threads`       | `--dataset.encoder_threads`       | `int \| None` | `None` (auto) | Threads per encoder instance. `None` will leave the vcoded decide |
-| `encoder_queue_maxsize` | `--dataset.encoder_queue_maxsize` | `int`         | `60`          | Max buffered frames per camera (~2s at 30fps). Consumes RAM       |
-
-## 3. Performance Considerations
-
-Streaming encoding means the CPU is encoding video **during** the capture loop, not after. This creates a CPU budget that must be shared between:
-
- **Control loop** (reading cameras, control the robot, writing non-video data)
- **Encoder threads** (one pool per camera)
- **Rerun visualization** (if enabled)
- **OS and other processes**
-
-### Resolution & Number of Cameras Impact
-
-| Setup                     | Throughput (px/sec) | CPU Encoding Load | Notes                          |
-| ------------------------- | ------------------- | ----------------- | ------------------------------ |
-| 2camsx 640x480x3 @30fps   | 55M                 | Low               | Works on most systems          |
-| 2camsx 1280x720x3 @30fps  | 165M                | Moderate          | Comfortable on modern systems  |
-| 2camsx 1920x1080x3 @30fps | 373M                | High              | Requires powerful high-end CPU |
-
-### `encoder_threads` Tuning
-
-This parameter controls how many threads each encoder instance uses internally:
-
- **Higher values** (e.g., 4-5): Faster encoding, but uses more CPU cores per camera. Good for high-end systems with many cores.
- **Lower values** (e.g., 1-2): Less CPU per camera, freeing cores for capture and visualization. Good for low-res images and capable CPUs.
- **`None` (default)**: Lets the codec decide. Information available in the codec logs.
-
-### Backpressure and Frame Dropping
-
-Each camera has a bounded queue (`encoder_queue_maxsize`, default 60 frames). When the encoder can't keep up:
-
-1. The queue fills up (consuming RAM)
-2. New frames are **dropped** (not blocked) — the capture loop continues uninterrupted
-3. A warning is logged: `"Encoder queue full for {camera}, dropped N frame(s)"`
-4. At episode end, total dropped frames per camera are reported
-
-### Symptoms of Encoder Falling Behind
-
- **System feels laggy and freezes**: all CPUs are at 100%
- **Dropped frame warnings** in the log or lower frames/FPS than expected in the recorded dataset
- **Choppy robot movement**: If CPU is severely overloaded, even the capture loop may be affected
- **Accumulated rerun lag**: Visualization falls behind real-time
-
-## 4. Hardware-Accelerated Encoding
-
-### When to Use
-
-Use HW encoding when:
-
- CPU is the bottleneck (dropped frames, choppy robot, rerun lag)
- You have compatible hardware (GPU or dedicated encoder)
- You're recording at high throughput (high resolution or with many cameras)
-
-### Choosing a Codec
-
-| Codec                 | CPU Usage | File Size      | Quality | Notes                                                            |
-| --------------------- | --------- | -------------- | ------- | ---------------------------------------------------------------- |
-| `libsvtav1` (default) | High      | Smallest       | Best    | Default. Best compression but most CPU-intensive                 |
-| `h264`                | Medium    | ~30-50% larger | Good    | Software H.264. Lower CPU                                        |
-| HW encoders           | Very Low  | Largest        | Good    | Offloads to dedicated hardware. Best for CPU-constrained systems |
-
-### Available HW Encoders
-
-| Encoder             | Platform      | Hardware                                                                                         | CLI Value                            |
-| ------------------- | ------------- | ------------------------------------------------------------------------------------------------ | ------------------------------------ |
-| `h264_videotoolbox` | macOS         | Apple Silicon / Intel                                                                            | `--dataset.vcodec=h264_videotoolbox` |
-| `hevc_videotoolbox` | macOS         | Apple Silicon / Intel                                                                            | `--dataset.vcodec=hevc_videotoolbox` |
-| `h264_nvenc`        | Linux/Windows | NVIDIA GPU                                                                                       | `--dataset.vcodec=h264_nvenc`        |
-| `hevc_nvenc`        | Linux/Windows | NVIDIA GPU                                                                                       | `--dataset.vcodec=hevc_nvenc`        |
-| `h264_vaapi`        | Linux         | Intel/AMD GPU                                                                                    | `--dataset.vcodec=h264_vaapi`        |
-| `h264_qsv`          | Linux/Windows | Intel Quick Sync                                                                                 | `--dataset.vcodec=h264_qsv`          |
-| `auto`              | Any           | Probes the system for available HW encoders. Falls back to `libsvtav1` if no HW encoder is found | `--dataset.vcodec=auto`              |
-
-> [!NOTE]
-> In order to use the HW accelerated encoders you might need to upgrade your GPU drivers.
-
-> [!NOTE]
-> `libsvtav1` is the default because it provides the best training performance; other vcodecs can reduce CPU usage and be faster, but they typically produce larger files and may affect training time.
-
-## 5. Troubleshooting
-
-| Symptom                                                            | Likely Cause                                 | Fix                                                                                                                                                                                                                                                                                  |
-| ------------------------------------------------------------------ | -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| System freezes or choppy robot movement or Rerun visualization lag | CPU starved (100% load usage)                | Close other apps, reduce encoding throughput, lower `encoder_threads`, use `h264`, use `display_data=False`. If the CPU continues to be at 100% then it might be insufficient for your setup, consider `--dataset.streaming_encoding=false` or HW encoding (`--dataset.vcodec=auto`) |
-| "Encoder queue full" warnings or dropped frames in dataset         | Encoder can't keep up (Queue overflow)       | If CPU is not at 100%: Increase `encoder_threads`, increase `encoder_queue_maxsize` or use HW encoding (`--dataset.vcodec=auto`).                                                                                                                                                    |
-| High RAM usage                                                     | Queue filling faster than encoding           | `encoder_threads` too low or CPU insufficient. Reduce `encoder_queue_maxsize` or use HW encoding                                                                                                                                                                                     |
-| Large video files                                                  | Using HW encoder or H.264                    | Expected trade-off. Switch to `libsvtav1` if CPU allows                                                                                                                                                                                                                              |
-| `save_episode()` still slow                                        | `streaming_encoding` is `False`              | Set `--dataset.streaming_encoding=true`                                                                                                                                                                                                                                              |
-| Encoder thread crash                                               | Codec not available or invalid settings      | Check `vcodec` is installed, try `--dataset.vcodec=auto`                                                                                                                                                                                                                             |
-| Recorded dataset is missing frames                                 | CPU/GPU starvation or occasional load spikes | If ~5% of frames are missing, your system is likely overloaded — follow the recommendations above. If fewer frames are missing (~2%), they are probably due to occasional transient load spikes (often at startup) and can be considered expected.                                   |
-
-## 6. Recommended Configurations
-
-These estimates are conservative; we recommend testing them on your setup—start with a low load and increase it gradually.
-
-### High-End Systems: modern 12+ cores (24+ threads)
-
-A throughput between ~250-500M px/sec should be comfortable in CPU. For even better results try HW encoding if available.
-
-```bash
-# 3camsx 1280x720x3 @30fps: Defaults work well. Optionally increase encoder parallelism.
-# 2camsx 1920x1080x3 @30fps: Defaults work well. Optionally increase encoder parallelism.
-lerobot-record --dataset.encoder_threads=5 ...
-
-# 3camsx 1920x1080x3 @30fps: Might require some tuning.
-```
-
-### Mid-Range Systems: modern 8+ cores (16+ threads) or Apple Silicon
-
-A throughput between ~80-300M px/sec should be possible in CPU.
-
-```bash
-# 3camsx 640x480x3 @30fps: Defaults work well. Optionally decrease encoder parallelism.
-# 2camsx 1280x720x3 @30fps: Defaults work well. Optionally decrease encoder parallelism.
-lerobot-record --dataset.encoder_threads=2 ...
-
-# 2camsx 1920x1080x3 @30fps: Might require some tuning.
-```
-
-### Low-Resource Systems: modern 4+ cores (8+ threads) or Raspberry Pi 5
-
-On very constrained systems, streaming encoding may compete too heavily with the capture loop. Disabling it falls back to the PNG-based approach where encoding happens between episodes (blocking, but doesn't interfere with capture). Alternatively, record at a lower throughput to reduce both capture and encoding load. Consider also changing codec to `h264` and using batch encoding.
-
-```bash
-# 2camsx 640x480x3 @30fps: Requires some tuning.
-
-# Use H.264, disable streaming, consider batching encoding
-lerobot-record --dataset.vcodec=h264 --dataset.streaming_encoding=false ...
-```
-
-## 7. Closing note
-
-Performance ultimately depends on your exact setup — frames-per-second, resolution, CPU cores and load, available memory, episode length, and the encoder you choose. Always test with your target workload, be mindful about your CPU & system capabilities and tune `encoder_threads`, `encoder_queue_maxsize`, and
-`vcodec` reasonably. That said, a common practical configuration (for many applications) is three cameras at 640×480x3 @30fps; this usually runs fine with the default streaming video encoding settings in modern systems. Always verify your recorded dataset is healthy by comparing the video duration to the CLI episode duration and confirming the row count equals FPS × CLI duration.
@@ -1,72 +1,23 @@
 # Unitree G1

-<img
-  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/unitree_thumbnail.jpg"
-  alt="Unitree G1 locomanipulation demo"
-  style={{ width: "100%" }}
-/>
+This guide covers the complete setup process for the Unitree G1 humanoid, from initial connection to running gr00t_wbc locomotion.

-The Unitree G1 humanoid is now supported in LeRobot! You can teleoperate, train locomanipulation policies, test in sim, and more. Both 29 and 23 DoF variants are supported.
+## About
+
+We support both 29 and 23 DOF G1 EDU version. We introduce:
+
+- **`unitree g1` robot class, handling low level read/write from/to the humanoid**
+- **ZMQ socket bridge** for remote communication and camera streaming, allowing for remote policy deployment over wlan, eth or directly on the robot
+- **Locomotion policies** from NVIDIA gr00t and Amazon FAR Holosoma
+- **Simulation mode** for testing policies without the physical robot in mujoco

 ---

-## Part 1: Getting Started
+## Connection guide

-### Install the Unitree SDK
+### Step 1: Configure Ethernet Interface

-Follow the [unitree_sdk2_python installation guide](https://github.com/unitreerobotics/unitree_sdk2_python#installation). Tested with `unitree_sdk2py==1.0.1` and `cyclonedds==0.10.2`:
-
-```bash
-conda create -y -n lerobot python=3.12
-conda activate lerobot
-git clone https://github.com/unitreerobotics/unitree_sdk2_python.git
-cd unitree_sdk2_python
-pip install -e .
-cd ..
-```
-
-### Install LeRobot
-
-```bash
-conda install ffmpeg -c conda-forge
-conda install -c conda-forge "pinocchio>=3.0.0,<4.0.0"
-git clone https://github.com/huggingface/lerobot.git
-cd lerobot
-pip install -e '.[unitree_g1]'
-```
-
-<Tip>
-  For now, pinocchio must be installed from conda-forge (not pip) to include the
-  CasADi bindings needed for arm IK.
-</Tip>
-
-### Test the Installation (Simulation)
-
-The simulation environment has its own dependencies. Check the Simulation environment dependencies: [Unitree G1 Mujoco EnvHub](https://huggingface.co/lerobot/unitree-g1-mujoco/tree/main).
-
-```bash
-pip install mujoco loguru msgpack msgpack-numpy
-```
-
-```bash
-lerobot-teleoperate \
-  --robot.type=unitree_g1 \
-  --robot.is_simulation=true \
-  --teleop.type=unitree_g1 \
-  --teleop.id=wbc_unitree \
-  --robot.cameras='{"global_view": {"type": "zmq", "server_address": "localhost", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30, "warmup_s": 5}}' \
-  --display_data=true \
-  --robot.controller=GrootLocomotionController
-```
-
-This will launch a [MuJoCo sim instance](https://huggingface.co/lerobot/unitree-g1-mujoco/tree/main) for the G1. You can connect a gamepad to your machine before launching in order to control the robot's locomotion in sim. We support both [HolosomaLocomotionController](https://github.com/amazon-far/holosoma) and [GrootLocomotionController](https://github.com/NVlabs/GR00T-WholeBodyControl) via `--robot.controller`.
-
- Press `9` to release the robot
- Press `7` / `8` to increase / decrease waist height
-
-### Connect to the Physical Robot
-
-The G1's Ethernet IP is fixed at `192.168.123.164`. Your machine must have a static IP on the same subnet: `192.168.123.x` where `x ≠ 164`.
+Set a static IP on the same subnet as the robot:

 ```bash
 # Replace 'enp131s0' with your ethernet interface name (check with `ip a`)
@@ -75,23 +26,47 @@ sudo ip addr add 192.168.123.200/24 dev enp131s0
 sudo ip link set enp131s0 up
 ```

-### SSH into the Robot
+**Note**: The G1's Ethernet IP is fixed at `192.168.123.164`. Your computer must use `192.168.123.x` with x ≠ 164.
+
+### Step 2: SSH into the Robot

 ```bash
 ssh unitree@192.168.123.164
 # Password: 123
 ```

-### Share Internet via Ethernet
+You should now be connected to the G1's Orin.

-The G1 needs internet access to clone repos and install packages. Share your laptop's connection over Ethernet:
+---
+
+## Part 2: Enable WiFi on the Robot
+
+Wlan0 is disabled by default on the G1. To enable it:
+
+### Step 1: Enable WiFi Hardware
+
+```bash
+sudo rfkill unblock wifi
+sudo rfkill unblock all
+
+# Bring up wlan0
+sudo ip link set wlan0 up
+
+# Enable NetworkManager control of wlan0
+sudo nmcli radio wifi on
+sudo nmcli device set wlan0 managed yes
+sudo systemctl restart NetworkManager
+```
+
+### Step 2: Enable Internet Forwarding

 **On your laptop:**

 ```bash
+# Enable IP forwarding
 sudo sysctl -w net.ipv4.ip_forward=1

-# Replace wlp132s0f0 with your WiFi interface name
+# Set up NAT (replace wlp132s0f0 with your WiFi interface)
 sudo iptables -t nat -A POSTROUTING -o wlp132s0f0 -s 192.168.123.0/24 -j MASQUERADE
 sudo iptables -A FORWARD -i wlp132s0f0 -o enp131s0 -m state --state RELATED,ESTABLISHED -j ACCEPT
 sudo iptables -A FORWARD -i enp131s0 -o wlp132s0f0 -j ACCEPT
@@ -100,193 +75,217 @@ sudo iptables -A FORWARD -i enp131s0 -o wlp132s0f0 -j ACCEPT
 **On the G1:**

 ```bash
+# Add laptop as default gateway
 sudo ip route del default 2>/dev/null || true
 sudo ip route add default via 192.168.123.200 dev eth0
 echo "nameserver 8.8.8.8" | sudo tee /etc/resolv.conf

-# Verify
+# Test connection
 ping -c 3 8.8.8.8
 ```

-### Install the Unitree SDK on the G1
-
-Follow the [unitree_sdk2_python installation guide](https://github.com/unitreerobotics/unitree_sdk2_python#installation):
-
-```bash
-conda create -y -n lerobot python=3.12
-conda activate lerobot
-git clone https://github.com/unitreerobotics/unitree_sdk2_python.git
-cd unitree_sdk2_python
-python -m pip install -e .
-cd ..
-```
-
-### Install LeRobot on the G1
-
-```bash
-git clone https://github.com/huggingface/lerobot.git
-cd lerobot
-conda install -c conda-forge "pinocchio>=3.0.0,<4.0.0"
-python -m pip install -e '.[unitree_g1]'
-```
-
-<Tip>
-  For now, pinocchio must be installed from conda-forge (not pip) to include the
-  CasADi bindings needed for arm IK.
-</Tip>
-
-### (Optional) Enable WiFi on the Robot
-
-For wireless SSH access, you can enable WiFi on the G1 (it's blocked by default):
-
-```bash
-sudo rfkill unblock all
-sudo ip link set wlan0 up
-sudo nmcli radio wifi on
-sudo nmcli device set wlan0 managed yes
-sudo systemctl restart NetworkManager
-```
-
-**Connect to a WiFi network:**
+### Step 3: Connect to WiFi Network

 ```bash
+# List available networks
 nmcli device wifi list

+# Connect to your WiFi (example)
 sudo nmcli connection add type wifi ifname wlan0 con-name "YourNetwork" ssid "YourNetwork"
 sudo nmcli connection modify "YourNetwork" wifi-sec.key-mgmt wpa-psk
 sudo nmcli connection modify "YourNetwork" wifi-sec.psk "YourPassword"
 sudo nmcli connection modify "YourNetwork" connection.autoconnect yes
 sudo nmcli connection up "YourNetwork"

+# Check WiFi IP address
 ip a show wlan0
 ```

-You can then SSH over WiFi instead of Ethernet:
+### Step 4: SSH Over WiFi
+
+Once connected to WiFi, note the robot's IP address and disconnect the Ethernet cable. You can now SSH over WiFi:

 ```bash
-ssh unitree@<ROBOT_WIFI_IP>
+ssh unitree@<YOUR_ROBOT_IP>
 # Password: 123
 ```

---
-
-## Part 2: Teleoperation & Locomotion
-
-### Run the Robot Server
-
-On the robot (from `~/lerobot`):
-
-```bash
-cd ~/lerobot
-python src/lerobot/robots/unitree_g1/run_g1_server.py --camera
-```
-
-### Run the Locomotion Policy
-
-You can run the teleoperation client from your laptop over Ethernet, over WiFi (experimental), or directly on the robot itself. Mind potential latency introduced by your network.
-
-**From your laptop:**
-
-```bash
-lerobot-teleoperate \
-  --robot.type=unitree_g1 \
-  --robot.is_simulation=false \
-  --robot.robot_ip=<ROBOT_IP> \
-  --teleop.type=unitree_g1 \
-  --teleop.id=wbc_unitree \
-  --robot.cameras='{"global_view": {"type": "zmq", "server_address": "<ROBOT_IP>", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30}}' \
-  --display_data=true \
-  --robot.controller=HolosomaLocomotionController
-```
-
-We support both [GrootLocomotionController](https://github.com/NVlabs/GR00T-WholeBodyControl) and [HolosomaLocomotionController](https://github.com/amazon-far/holosoma) via `--robot.controller`.
+Replace `<YOUR_ROBOT_IP>` with your robot's actual WiFi IP address.

 ---

-## Part 3: Loco-Manipulation with the Homunculus Exoskeleton
+## Part 3: Robot Server Setup

-We provide a loco-manipulation solution via the Homunculus Exoskeleton — an open-source 7 DoF exoskeleton for whole-body control. Check it out [here](https://github.com/nepyope/hmc_exo).
+### Step 1: Install LeRobot on the Orin

-### Calibrate
+SSH into the robot and install LeRobot:
+
+```bash
+ssh unitree@<YOUR_ROBOT_IP>
+
+conda create -y -n lerobot python=3.10
+conda activate lerobot
+git clone https://github.com/huggingface/lerobot.git
+cd lerobot
+pip install -e '.[unitree_g1]'
+git clone https://github.com/unitreerobotics/unitree_sdk2_python.git
+cd unitree_sdk2_python  && pip install -e .
+```
+
+**Note**: The Unitree SDK requires CycloneDDS v0.10.2 to be installed. See the [Unitree SDK documentation](https://github.com/unitreerobotics/unitree_sdk2_python) for details.
+
+### Step 2: Run the Robot Server
+
+On the robot:
+
+```bash
+python src/lerobot/robots/unitree_g1/run_g1_server.py
+```
+
+**Important**: Keep this terminal running. The server must be active for remote control.
+
+---
+
+## Part 4: Controlling the robot
+
+With the robot server running, you can now control the robot remotely. Let's launch a locomotion policy
+
+### Step 1: Install LeRobot on your machine
+
+```bash
+conda create -y -n lerobot python=3.10
+conda activate lerobot
+git clone https://github.com/huggingface/lerobot.git
+cd lerobot
+pip install -e '.[unitree_g1]'
+git clone https://github.com/unitreerobotics/unitree_sdk2_python.git
+cd unitree_sdk2_python  && pip install -e .
+```
+
+### Step 2: Update Robot IP in Config
+
+Edit the config file to match your robot's WiFi IP:
+
+```python
+# In src/lerobot/robots/unitree_g1/config_unitree_g1.py
+robot_ip: str = "<YOUR_ROBOT_IP>"  # Replace with your robot's WiFi IP.
+```
+
+### Step 3: Run the Locomotion Policy
+
+```bash
+# Run GR00T locomotion controller
+python examples/unitree_g1/gr00t_locomotion.py --repo-id "nepyope/GR00T-WholeBodyControl_g1"
+
+# Run Holosoma locomotion controller
+python examples/unitree_g1/holosoma_locomotion.py
+
+```
+
+Press `Ctrl+C` to stop the policy.
+
+---
+
+## Running in Simulation Mode (MuJoCo)
+
+You can test policies before deploying on the physical robot using MuJoCo simulation. Set `is_simulation=True` in config or pass `--robot.is_simulation=true` via CLI.
+
+### Calibrate Exoskeleton Teleoperator

 ```bash
 lerobot-calibrate \
-  --teleop.type=unitree_g1 \
-  --teleop.left_arm_config.port=/dev/ttyACM1 \
-  --teleop.right_arm_config.port=/dev/ttyACM0 \
-  --teleop.id=exo
+    --teleop.type=unitree_g1 \
+    --teleop.left_arm_config.port=/dev/ttyACM1 \
+    --teleop.right_arm_config.port=/dev/ttyACM0 \
+    --teleop.id=exo
 ```

-During calibration move each joint through its entire range. After fitting, move the joint in a neutral position and press `n` to advance.
-
-### Record a Dataset
+### Teleoperate in Simulation

 ```bash
-lerobot-record \
-  --robot.type=unitree_g1 \
-  --robot.is_simulation=true \
-  --robot.cameras='{"global_view": {"type": "zmq", "server_address": "localhost", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30}}' \
-  --teleop.type=unitree_g1 \
-  --teleop.left_arm_config.port=/dev/ttyACM1 \
-  --teleop.right_arm_config.port=/dev/ttyACM0 \
-  --teleop.id=exo \
-  --dataset.repo_id=your-username/dataset-name \
-  --dataset.single_task="Test" \
-  --dataset.num_episodes=2 \
-  --dataset.episode_time_s=5 \
-  --dataset.reset_time_s=5 \
-  --dataset.push_to_hub=true \
-  --dataset.streaming_encoding=true \
-  --dataset.encoder_threads=2
+lerobot-teleoperate \
+    --robot.type=unitree_g1 \
+    --robot.is_simulation=true \
+    --teleop.type=unitree_g1 \
+    --teleop.left_arm_config.port=/dev/ttyACM1 \
+    --teleop.right_arm_config.port=/dev/ttyACM0 \
+    --teleop.id=exo \
+    --fps=100
 ```

-> **Note:** Omit `--teleop.left_arm_config.port` and `--teleop.right_arm_config.port` if you're only using the joystick.
+### Record Dataset in Simulation

-Example dataset: [nepyope/unitree_box_move_blue_full](https://huggingface.co/datasets/nepyope/unitree_box_move_blue_full)
+```bash
+python -m lerobot.scripts.lerobot_record \
+    --robot.type=unitree_g1 \
+    --robot.is_simulation=true \
+    --robot.cameras='{"global_view": {"type": "zmq", "server_address": "localhost", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30}}' \
+    --teleop.type=unitree_g1 \
+    --teleop.left_arm_config.port=/dev/ttyACM1 \
+    --teleop.right_arm_config.port=/dev/ttyACM0 \
+    --teleop.id=exo \
+    --dataset.repo_id=your-username/dataset-name \
+    --dataset.single_task="Test" \
+    --dataset.num_episodes=2 \
+    --dataset.episode_time_s=5 \
+    --dataset.reset_time_s=5 \
+    --dataset.push_to_hub=true
+```
+
+Example simulation dataset: [nepyope/teleop_test_sim](https://huggingface.co/datasets/nepyope/teleop_test_sim)

 ---

-## Part 4: Training & Inference
+## Running on Real Robot

-### Train
+Once the robot server is running on the G1 (see Part 3), you can teleoperate and record on the real robot.
+
+### Start the Camera Server
+
+On the robot, start the ZMQ image server:

 ```bash
-python src/lerobot/scripts/lerobot_train.py \
-  --dataset.repo_id=your-username/dataset-name  \
-  --policy.type=pi05 \
-  --output_dir=./outputs/pi05_training \
-  --job_name=pi05_training \
-  --policy.repo_id=your-username/your-repo-id \
-  --policy.pretrained_path=lerobot/pi05_base \
-  --policy.compile_model=true \
-  --policy.gradient_checkpointing=true \
-  --wandb.enable=true \
-  --policy.dtype=bfloat16 \
-  --policy.freeze_vision_encoder=false \
-  --policy.train_expert_only=false \
-  --steps=3000 \
-  --policy.device=cuda \
-  --batch_size=32
+python src/lerobot/cameras/zmq/image_server.py
 ```

-### Inference with RTC
+Keep this running in a separate terminal for camera streaming during recording.

-Once trained, we recommend deploying policies using inference-time RTC:
+### Teleoperate Real Robot

 ```bash
-python examples/rtc/eval_with_real_robot.py \
-  --policy.path=your-username/your-repo-id \
-  --policy.device=cuda \
-  --robot.type=unitree_g1 \
-  --robot.is_simulation=false \
-  --robot.controller=HolosomaLocomotionController \
-  --robot.cameras='{"global_view": {"type": "zmq", "server_address": "<ROBOT_IP>", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30}}' \
-  --task="task_description" \
-  --duration=1000 \
-  --fps=30 \
-  --rtc.enabled=true
+lerobot-teleoperate \
+    --robot.type=unitree_g1 \
+    --robot.is_simulation=false \
+    --teleop.type=unitree_g1 \
+    --teleop.left_arm_config.port=/dev/ttyACM1 \
+    --teleop.right_arm_config.port=/dev/ttyACM0 \
+    --teleop.id=exo \
+    --fps=100
 ```

+### Record Dataset on Real Robot
+
+```bash
+python -m lerobot.scripts.lerobot_record \
+    --robot.type=unitree_g1 \
+    --robot.is_simulation=false \
+    --robot.cameras='{"global_view": {"type": "zmq", "server_address": "172.18.129.215", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30}}' \
+    --teleop.type=unitree_g1 \
+    --teleop.left_arm_config.port=/dev/ttyACM1 \
+    --teleop.right_arm_config.port=/dev/ttyACM0 \
+    --teleop.id=exo \
+    --dataset.repo_id=your-username/dataset-name \
+    --dataset.single_task="Test" \
+    --dataset.num_episodes=2 \
+    --dataset.episode_time_s=5 \
+    --dataset.reset_time_s=5 \
+    --dataset.push_to_hub=true
+```
+
+**Note**: Update `server_address` to match your robot's camera server IP.
+
+Example real robot dataset: [nepyope/teleop_test_real](https://huggingface.co/datasets/nepyope/teleop_test_real)
+
 ---

 ## Additional Resources
@@ -295,8 +294,8 @@ python examples/rtc/eval_with_real_robot.py \
 - [GR00T-WholeBodyControl](https://github.com/NVlabs/GR00T-WholeBodyControl)
 - [Holosoma](https://github.com/amazon-far/holosoma)
 - [LeRobot Documentation](https://github.com/huggingface/lerobot)
- [Unitree IL LeRobot](https://github.com/unitreerobotics/unitree_IL_lerobot)
+- [Unitree_IL_Lerobot](https://github.com/unitreerobotics/unitree_IL_lerobot)

 ---

-_Last updated: March 2026_
+_Last updated: December 2025_
@@ -12,7 +12,6 @@ LeRobot provides several utilities for manipulating datasets:
 4. **Add Features** - Add new features to a dataset
 5. **Remove Features** - Remove features from a dataset
 6. **Convert to Video** - Convert image-based datasets to video format for efficient storage
-7. **Show the Info of Datasets** - Show the summary of datasets information such as number of episode etc.

 The core implementation is in `lerobot.datasets.dataset_tools`.
 An example script detailing how to use the tools API is available in `examples/dataset/use_dataset_tools.py`.
@@ -157,30 +156,6 @@ lerobot-edit-dataset \

 **Note:** The resulting dataset will be a proper LeRobotDataset with all cameras encoded as videos in the `videos/` directory, with parquet files containing only metadata (no raw image data). All episodes, stats, and tasks are preserved.

-### Show the information of datasets
-
-Show the information of datasets such as number of episode, number of frame, File size and so on.
-No change will be made to the dataset
-
-```bash
-
-# Show dataset information without feature details
-lerobot-edit-dataset \
-    --repo_id lerobot/pusht_image \
-    --operation.type info \
-
-# Show dataset information with feature details
-lerobot-edit-dataset \
-    --repo_id lerobot/pusht_image \
-    --operation.type info \
-    --operation.show_features true
-
-```
-
-**Parameters:**
-
- `parameters`: The flag to control show or no show dataset information with feature details.(default=false)
-
 ### Push to Hub

 Add the `--push_to_hub true` flag to any command to automatically upload the resulting dataset to the Hugging Face Hub:
@@ -45,7 +45,7 @@ policy.type=wall_x
 For training WallX, you can use the standard LeRobot training script with the appropriate configuration:

 ```bash
-lerobot-train \
+python src/lerobot/scripts/lerobot_train.py \
    --dataset.repo_id=your_dataset \
    --policy.type=wall_x \
    --output_dir=./outputs/wallx_training \
@@ -154,7 +154,7 @@ lerobot-train \

 ```bash
 lerobot-train \
-  --dataset.repo_id=<USER>/bimanual-so100-handover-cube \
+  --dataset.repo_id=pepijn223/bimanual-so100-handover-cube \
  --output_dir=./outputs/xvla_bimanual \
  --job_name=xvla_so101_training \
  --policy.path="lerobot/xvla-base" \
@@ -22,7 +22,7 @@ lerobot-replay \
    --robot.type=so100_follower \
    --robot.port=/dev/tty.usbmodem58760431541 \
    --robot.id=black \
-    --dataset.repo_id=<USER>/record-test \
+    --dataset.repo_id=aliberts/record-test \
    --dataset.episode=2
 ```
 """
@@ -57,7 +57,7 @@ class DatasetReplayConfig:
    repo_id: str
    # Episode to replay.
    episode: int
-    # Root directory where the dataset will be stored (e.g. 'dataset/path'). If None, defaults to $HF_LEROBOT_HOME/repo_id.
+    # Root directory where the dataset will be stored (e.g. 'dataset/path').
    root: str | Path | None = None
    # Limit the frames per second. By default, uses the policy fps.
    fps: int = 30
@@ -32,8 +32,7 @@ import torch
 from huggingface_hub import HfApi

 import lerobot
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.lerobot_dataset import LeRobotDataset, LeRobotDatasetMetadata


 def main():
@@ -1,490 +0,0 @@
-#!/usr/bin/env python
-
-# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-#     http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-
-"""
-SLURM-distributed SARM RA-BC annotation pipeline.
-
-Computes SARM progress values for all frames in a dataset, distributed across
-SLURM workers, then merges the shards into a single sarm_progress.parquet.
-
-Two subcommands, each a separate SLURM submission:
-
-  compute    – N workers, each computes progress for a subset of episodes
-  aggregate  – 1 worker, merges N shards into sarm_progress.parquet, pushes to hub
-
-Usage:
-    python slurm_compute_rabc.py compute \\
-        --repo-id user/dataset --reward-model-path user/sarm_model \\
-        --stride 10 --device cpu --workers 50 --partition cpu
-
-    python slurm_compute_rabc.py aggregate \\
-        --repo-id user/dataset --reward-model-path user/sarm_model \\
-        --partition cpu --push-to-hub
-"""
-
-import argparse
-from pathlib import Path
-
-from datatrove.executor import LocalPipelineExecutor
-from datatrove.executor.slurm import SlurmPipelineExecutor
-from datatrove.pipeline.base import PipelineStep
-
-
-class ComputeProgressShards(PipelineStep):
-    """Each worker computes SARM progress for its assigned episodes."""
-
-    def __init__(
-        self, repo_id, reward_model_path, stride=1, head_mode="sparse", device="cpu", shard_dir="rabc_shards"
-    ):
-        super().__init__()
-        if stride < 1:
-            raise ValueError(f"stride must be >= 1, got {stride}")
-        self.repo_id = repo_id
-        self.reward_model_path = reward_model_path
-        self.stride = stride
-        self.head_mode = head_mode
-        self.device = device
-        self.shard_dir = shard_dir
-
-    def run(self, data=None, rank: int = 0, world_size: int = 1):
-        import logging
-        from pathlib import Path
-
-        import numpy as np
-        import pyarrow as pa
-        import pyarrow.parquet as pq
-        import torch
-        from tqdm import tqdm
-
-        from lerobot.policies.sarm.compute_rabc_weights import (
-            generate_all_frame_indices,
-            interpolate_progress,
-            load_sarm_resources,
-        )
-        from lerobot.utils.utils import init_logging
-
-        init_logging()
-
-        dataset, reward_model, preprocess = load_sarm_resources(
-            self.repo_id,
-            self.reward_model_path,
-            self.device,
-        )
-
-        if hasattr(preprocess, "eval"):
-            preprocess.eval()
-        for step in preprocess.steps:
-            if hasattr(step, "eval"):
-                step.eval()
-
-        image_key = reward_model.config.image_key
-        state_key = reward_model.config.state_key
-        frame_gap = reward_model.config.frame_gap
-        center_idx = reward_model.config.n_obs_steps // 2
-
-        dual_mode = reward_model.config.uses_dual_heads
-        compute_sparse = self.head_mode in ("sparse", "both") or not dual_mode
-        compute_dense = self.head_mode in ("dense", "both") and dual_mode
-
-        my_episodes = list(range(dataset.num_episodes))[rank::world_size]
-        if not my_episodes:
-            logging.info(f"Rank {rank}: no episodes assigned")
-            return
-        logging.info(f"Rank {rank}: {len(my_episodes)} / {dataset.num_episodes} episodes")
-
-        all_rows = []
-
-        for ep_idx in tqdm(my_episodes, desc=f"Rank {rank}"):
-            ep = dataset.meta.episodes[ep_idx]
-            ep_start, ep_end = ep["dataset_from_index"], ep["dataset_to_index"]
-            task = dataset[ep_start].get("task", "perform the task")
-
-            all_ep_indices = generate_all_frame_indices(ep_start, ep_end, frame_gap)
-            if self.stride > 1:
-                compute_indices = [i for i in all_ep_indices if (i - ep_start) % self.stride == 0]
-                if (ep_end - 1) not in compute_indices:
-                    compute_indices.append(ep_end - 1)
-                compute_indices = sorted(set(compute_indices))
-            else:
-                compute_indices = all_ep_indices
-
-            frame_results = {}
-            for qi in tqdm(compute_indices, desc=f"  Ep {ep_idx}", leave=False):
-                try:
-                    sample = dataset[qi]
-                    batch = {
-                        image_key: sample[image_key],
-                        "task": task,
-                        "index": qi,
-                        "episode_index": ep_idx,
-                    }
-                    if state_key in sample:
-                        batch[state_key] = sample[state_key]
-
-                    with torch.no_grad():
-                        processed = preprocess(batch)
-                        vf = processed["video_features"].to(self.device)
-                        tf = processed["text_features"].to(self.device)
-                        sf = processed.get("state_features")
-                        if sf is not None:
-                            sf = sf.to(self.device)
-                        lengths = processed.get("lengths")
-
-                        sparse_val = dense_val = np.nan
-                        if compute_sparse:
-                            r = reward_model.calculate_rewards(
-                                text_embeddings=tf,
-                                video_embeddings=vf,
-                                state_features=sf,
-                                lengths=lengths,
-                                return_all_frames=True,
-                                head_mode="sparse",
-                            )
-                            sparse_val = float(r[0, center_idx] if r.ndim == 2 else r[center_idx])
-                        if compute_dense:
-                            r = reward_model.calculate_rewards(
-                                text_embeddings=tf,
-                                video_embeddings=vf,
-                                state_features=sf,
-                                lengths=lengths,
-                                return_all_frames=True,
-                                head_mode="dense",
-                            )
-                            dense_val = float(r[0, center_idx] if r.ndim == 2 else r[center_idx])
-
-                        frame_results[qi] = (sparse_val, dense_val)
-                except Exception as e:
-                    logging.warning(f"Failed frame {qi}: {e}")
-
-            if not frame_results:
-                logging.warning(f"Episode {ep_idx}: all frames failed, skipping")
-                continue
-
-            # Interpolate to all frames in this episode
-            computed_idx = np.array(sorted(frame_results.keys()))
-            all_frame_arr = np.arange(ep_start, ep_end)
-
-            sparse_vals = np.array([frame_results[i][0] for i in computed_idx]) if compute_sparse else None
-            dense_vals = np.array([frame_results[i][1] for i in computed_idx]) if compute_dense else None
-
-            if self.stride > 1 and len(computed_idx) > 1:
-                if compute_sparse:
-                    sparse_vals = interpolate_progress(computed_idx, sparse_vals, all_frame_arr)
-                if compute_dense:
-                    dense_vals = interpolate_progress(computed_idx, dense_vals, all_frame_arr)
-                output_frames = all_frame_arr
-            else:
-                # Use only successfully computed frames to avoid indexing mismatch on failures
-                output_frames = computed_idx
-
-            for i, fi in enumerate(output_frames):
-                row = {"index": int(fi), "episode_index": ep_idx, "frame_index": int(fi - ep_start)}
-                if compute_sparse:
-                    row["progress_sparse"] = float(sparse_vals[i])
-                if compute_dense:
-                    row["progress_dense"] = float(dense_vals[i])
-                all_rows.append(row)
-
-        if all_rows:
-            import pandas as pd
-
-            df = pd.DataFrame(all_rows).sort_values("index").reset_index(drop=True)
-            table = pa.Table.from_pandas(df, preserve_index=False)
-            table = table.replace_schema_metadata({b"reward_model_path": self.reward_model_path.encode()})
-            shard_dir = Path(self.shard_dir)
-            shard_dir.mkdir(parents=True, exist_ok=True)
-            out = shard_dir / f"shard_{rank:05d}.parquet"
-            pq.write_table(table, out)
-            logging.info(f"Rank {rank}: saved {len(df)} rows to {out}")
-
-
-class AggregateProgress(PipelineStep):
-    """Merge all shard parquets into final sarm_progress.parquet."""
-
-    def __init__(self, repo_id, reward_model_path, shard_dir="rabc_shards", push_to_hub=False):
-        super().__init__()
-        self.repo_id = repo_id
-        self.reward_model_path = reward_model_path
-        self.shard_dir = shard_dir
-        self.push_to_hub = push_to_hub
-
-    def run(self, data=None, rank: int = 0, world_size: int = 1):
-        import datetime
-        import logging
-        import os
-        from pathlib import Path
-
-        import pandas as pd
-        import pyarrow as pa
-        import pyarrow.parquet as pq
-
-        from lerobot.datasets.lerobot_dataset import LeRobotDataset
-        from lerobot.utils.utils import init_logging
-
-        init_logging()
-        if rank != 0:
-            return
-
-        shard_dir = Path(self.shard_dir)
-        shards = sorted(shard_dir.glob("shard_*.parquet"))
-        if not shards:
-            raise FileNotFoundError(f"No shards found in {shard_dir}")
-
-        # Log shard modification time range to help detect stale files
-        mtimes = [os.path.getmtime(s) for s in shards]
-        oldest = datetime.datetime.fromtimestamp(min(mtimes)).isoformat(timespec="seconds")
-        newest = datetime.datetime.fromtimestamp(max(mtimes)).isoformat(timespec="seconds")
-        logging.info(f"Aggregating {len(shards)} shards (oldest: {oldest}, newest: {newest})")
-
-        df = pd.concat([pd.read_parquet(s) for s in shards], ignore_index=True)
-        df = df.sort_values("index").reset_index(drop=True)
-
-        table = pa.Table.from_pandas(df, preserve_index=False)
-        table = table.replace_schema_metadata({b"reward_model_path": self.reward_model_path.encode()})
-
-        temp_ds = LeRobotDataset(self.repo_id, download_videos=False)
-        out_path = Path(temp_ds.root) / "sarm_progress.parquet"
-        out_path.parent.mkdir(parents=True, exist_ok=True)
-        pq.write_table(table, out_path)
-        logging.info(f"Saved {len(df)} rows to {out_path}")
-
-        for col in ["progress_sparse", "progress_dense"]:
-            if col in df.columns:
-                v = df[col].dropna()
-                logging.info(
-                    f"{col}: mean={v.mean():.4f} std={v.std():.4f} min={v.min():.4f} max={v.max():.4f}"
-                )
-
-        if self.push_to_hub:
-            from huggingface_hub import HfApi
-
-            api = HfApi()
-            hub_path = "sarm_progress.parquet"
-            logging.info(f"Uploading to {self.repo_id}/{hub_path}")
-            api.upload_file(
-                path_or_fileobj=str(out_path),
-                path_in_repo=hub_path,
-                repo_id=self.repo_id,
-                repo_type="dataset",
-            )
-            logging.info(f"Uploaded: https://huggingface.co/datasets/{self.repo_id}/blob/main/{hub_path}")
-
-
-def make_compute_executor(
-    repo_id,
-    reward_model_path,
-    stride,
-    head_mode,
-    device,
-    shard_dir,
-    logs_dir,
-    job_name,
-    slurm,
-    workers,
-    partition,
-    cpus_per_task,
-    mem_per_cpu,
-):
-    kwargs = {
-        "pipeline": [
-            ComputeProgressShards(repo_id, reward_model_path, stride, head_mode, device, str(shard_dir)),
-        ],
-        "logging_dir": str(logs_dir / job_name),
-    }
-
-    if slurm:
-        kwargs.update(
-            {
-                "job_name": job_name,
-                "tasks": workers,
-                "workers": workers,
-                "time": "24:00:00",
-                "partition": partition,
-                "cpus_per_task": cpus_per_task,
-                "sbatch_args": {"mem-per-cpu": mem_per_cpu},
-            }
-        )
-        return SlurmPipelineExecutor(**kwargs)
-
-    kwargs.update({"tasks": workers, "workers": 1})
-    return LocalPipelineExecutor(**kwargs)
-
-
-def make_aggregate_executor(
-    repo_id,
-    reward_model_path,
-    shard_dir,
-    logs_dir,
-    job_name,
-    slurm,
-    partition,
-    cpus_per_task,
-    mem_per_cpu,
-    push_to_hub,
-):
-    kwargs = {
-        "pipeline": [
-            AggregateProgress(repo_id, reward_model_path, str(shard_dir), push_to_hub),
-        ],
-        "logging_dir": str(logs_dir / job_name),
-    }
-
-    if slurm:
-        kwargs.update(
-            {
-                "job_name": job_name,
-                "tasks": 1,
-                "workers": 1,
-                "time": "02:00:00",
-                "partition": partition,
-                "cpus_per_task": cpus_per_task,
-                "sbatch_args": {"mem-per-cpu": mem_per_cpu},
-            }
-        )
-        return SlurmPipelineExecutor(**kwargs)
-
-    kwargs.update({"tasks": 1, "workers": 1})
-    return LocalPipelineExecutor(**kwargs)
-
-
-def _add_shared_args(p):
-    p.add_argument(
-        "--repo-id",
-        type=str,
-        required=True,
-        help="Hugging Face repository identifier, e.g. 'user/dataset'.",
-    )
-    p.add_argument(
-        "--shard-dir",
-        type=Path,
-        default=Path("rabc_shards"),
-        help="Directory to read/write per-rank parquet shards.",
-    )
-    p.add_argument(
-        "--logs-dir",
-        type=Path,
-        default=Path("logs"),
-        help="Directory for datatrove logs.",
-    )
-    p.add_argument(
-        "--job-name",
-        type=str,
-        default=None,
-        help="SLURM job name (defaults to rabc_<subcommand>).",
-    )
-    p.add_argument(
-        "--slurm",
-        type=int,
-        default=1,
-        help="1 = submit via SLURM; 0 = run locally (useful for debugging).",
-    )
-    p.add_argument(
-        "--partition",
-        type=str,
-        default=None,
-        help="SLURM partition to submit to.",
-    )
-    p.add_argument(
-        "--cpus-per-task",
-        type=int,
-        default=4,
-        help="Number of CPUs per SLURM task.",
-    )
-    p.add_argument(
-        "--mem-per-cpu",
-        type=str,
-        default="4G",
-        help="Memory per CPU, e.g. '4G' or '1950M'.",
-    )
-
-
-def main():
-    parser = argparse.ArgumentParser(
-        description="SLURM-distributed SARM RA-BC annotation pipeline",
-        formatter_class=argparse.RawDescriptionHelpFormatter,
-    )
-    sub = parser.add_subparsers(dest="command", required=True)
-
-    # compute subcommand
-    cp = sub.add_parser(
-        "compute",
-        help="Distribute progress computation across SLURM workers.",
-    )
-    _add_shared_args(cp)
-    cp.add_argument(
-        "--reward-model-path",
-        type=str,
-        required=True,
-        help="Path or HF repo id of the SARM reward model.",
-    )
-    cp.add_argument(
-        "--stride",
-        type=int,
-        default=1,
-        help="Compute every Nth frame; intermediate frames are interpolated (must be >= 1).",
-    )
-    cp.add_argument(
-        "--head-mode",
-        type=str,
-        default="sparse",
-        choices=["sparse", "dense", "both"],
-        help="Which reward head(s) to compute.",
-    )
-    cp.add_argument(
-        "--device",
-        type=str,
-        default="cpu",
-        help="Device for reward model inference, e.g. 'cpu' or 'cuda'.",
-    )
-    cp.add_argument(
-        "--workers",
-        type=int,
-        default=50,
-        help="Number of parallel SLURM tasks (one shard per worker).",
-    )
-
-    # aggregate subcommand
-    ap = sub.add_parser(
-        "aggregate",
-        help="Merge per-rank shards into a single sarm_progress.parquet.",
-    )
-    _add_shared_args(ap)
-    ap.add_argument(
-        "--reward-model-path",
-        type=str,
-        required=True,
-        help="Path or HF repo id of the SARM reward model (stored in parquet metadata).",
-    )
-    ap.add_argument(
-        "--push-to-hub",
-        action="store_true",
-        help="Upload sarm_progress.parquet to the Hugging Face Hub after aggregation.",
-    )
-
-    args = parser.parse_args()
-    job_name = args.job_name or f"rabc_{args.command}"
-    kwargs = vars(args)
-    kwargs["slurm"] = kwargs.pop("slurm") == 1
-    kwargs["job_name"] = job_name
-    command = kwargs.pop("command")
-
-    executor = make_compute_executor(**kwargs) if command == "compute" else make_aggregate_executor(**kwargs)
-
-    executor.run()
-
-
-if __name__ == "__main__":
-    main()
@@ -14,8 +14,8 @@
 # See the License for the specific language governing permissions and
 # limitations under the License.

-from lerobot.datasets.feature_utils import hw_to_dataset_features
 from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import hw_to_dataset_features
 from lerobot.policies.act.modeling_act import ACTPolicy
 from lerobot.policies.factory import make_pre_post_processors
 from lerobot.processor import make_default_processors
@@ -14,8 +14,8 @@
 # See the License for the specific language governing permissions and
 # limitations under the License.

-from lerobot.datasets.feature_utils import hw_to_dataset_features
 from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import hw_to_dataset_features
 from lerobot.processor import make_default_processors
 from lerobot.robots.lekiwi.config_lekiwi import LeKiwiClientConfig
 from lerobot.robots.lekiwi.lekiwi_client import LeKiwiClient
@@ -16,13 +16,15 @@

 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
 from lerobot.configs.types import FeatureType, PolicyFeature
-from lerobot.datasets.feature_utils import combine_feature_dicts
 from lerobot.datasets.lerobot_dataset import LeRobotDataset
 from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.datasets.utils import combine_feature_dicts
 from lerobot.model.kinematics import RobotKinematics
 from lerobot.policies.act.modeling_act import ACTPolicy
 from lerobot.policies.factory import make_pre_post_processors
 from lerobot.processor import (
+    RobotAction,
+    RobotObservation,
    RobotProcessorPipeline,
    make_default_teleop_action_processor,
 )
@@ -38,7 +40,6 @@ from lerobot.robots.so_follower.robot_kinematic_processor import (
    InverseKinematicsEEToJoints,
 )
 from lerobot.scripts.lerobot_record import record_loop
-from lerobot.types import RobotAction, RobotObservation
 from lerobot.utils.control_utils import init_keyboard_listener
 from lerobot.utils.utils import log_say
 from lerobot.utils.visualization_utils import init_rerun
@@ -15,11 +15,11 @@
 # limitations under the License.

 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
-from lerobot.datasets.feature_utils import combine_feature_dicts
 from lerobot.datasets.lerobot_dataset import LeRobotDataset
 from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.datasets.utils import combine_feature_dicts
 from lerobot.model.kinematics import RobotKinematics
-from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor import RobotAction, RobotObservation, RobotProcessorPipeline
 from lerobot.processor.converters import (
    observation_to_transition,
    robot_action_observation_to_transition,
@@ -38,7 +38,6 @@ from lerobot.scripts.lerobot_record import record_loop
 from lerobot.teleoperators.phone.config_phone import PhoneConfig, PhoneOS
 from lerobot.teleoperators.phone.phone_processor import MapPhoneActionToRobotAction
 from lerobot.teleoperators.phone.teleop_phone import Phone
-from lerobot.types import RobotAction, RobotObservation
 from lerobot.utils.control_utils import init_keyboard_listener
 from lerobot.utils.utils import log_say
 from lerobot.utils.visualization_utils import init_rerun
@@ -18,7 +18,7 @@ import time

 from lerobot.datasets.lerobot_dataset import LeRobotDataset
 from lerobot.model.kinematics import RobotKinematics
-from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor import RobotAction, RobotObservation, RobotProcessorPipeline
 from lerobot.processor.converters import (
    robot_action_observation_to_transition,
    transition_to_robot_action,
@@ -27,7 +27,6 @@ from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
 from lerobot.robots.so_follower.robot_kinematic_processor import (
    InverseKinematicsEEToJoints,
 )
-from lerobot.types import RobotAction, RobotObservation
 from lerobot.utils.constants import ACTION
 from lerobot.utils.robot_utils import precise_sleep
 from lerobot.utils.utils import log_say
@@ -16,7 +16,7 @@
 import time

 from lerobot.model.kinematics import RobotKinematics
-from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor import RobotAction, RobotObservation, RobotProcessorPipeline
 from lerobot.processor.converters import (
    robot_action_observation_to_transition,
    transition_to_robot_action,
@@ -31,7 +31,6 @@ from lerobot.robots.so_follower.robot_kinematic_processor import (
 from lerobot.teleoperators.phone.config_phone import PhoneConfig, PhoneOS
 from lerobot.teleoperators.phone.phone_processor import MapPhoneActionToRobotAction
 from lerobot.teleoperators.phone.teleop_phone import Phone
-from lerobot.types import RobotAction, RobotObservation
 from lerobot.utils.robot_utils import precise_sleep
 from lerobot.utils.visualization_utils import init_rerun, log_rerun_data

@@ -22,8 +22,7 @@ from pathlib import Path
 import numpy as np
 import tensorflow_datasets as tfds

-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.lerobot_dataset import LeRobotDataset, LeRobotDatasetMetadata
 from lerobot.utils.utils import get_elapsed_time_in_days_hours_minutes_seconds

 DROID_SHARDS = 2048
@@ -26,7 +26,7 @@ from huggingface_hub import HfApi
 from huggingface_hub.constants import REPOCARD_NAME
 from port_droid import DROID_SHARDS

-from lerobot.datasets.dataset_metadata import CODEBASE_VERSION, LeRobotDatasetMetadata
+from lerobot.datasets.lerobot_dataset import CODEBASE_VERSION, LeRobotDatasetMetadata
 from lerobot.datasets.utils import create_lerobot_dataset_card
 from lerobot.utils.utils import init_logging

@@ -155,7 +155,7 @@ class UploadDataset(PipelineStep):
        from datasets.utils.tqdm import disable_progress_bars
        from huggingface_hub import CommitOperationAdd, preupload_lfs_files

-        from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+        from lerobot.datasets.lerobot_dataset import LeRobotDatasetMetadata
        from lerobot.utils.utils import init_logging

        init_logging()
@@ -27,8 +27,8 @@ measuring consistency and ground truth alignment.
 Usage:
    # Basic usage with smolvla policy
    uv run python examples/rtc/eval_dataset.py \
-        --policy.path=<USER>/smolvla_check_rtc_last3 \
-        --dataset.repo_id=<USER>/check_rtc \
+        --policy.path=helper2424/smolvla_check_rtc_last3 \
+        --dataset.repo_id=helper2424/check_rtc \
        --rtc.execution_horizon=8 \
        --device=mps \
        --rtc.max_guidance_weight=10.0 \
@@ -58,16 +58,16 @@ Usage:
        --device=cuda

    uv run python examples/rtc/eval_dataset.py \
-        --policy.path=<USER>/reuben_pi0 \
-        --dataset.repo_id=<USER>/so101_cube_in_cup \
+        --policy.path=lipsop/reuben_pi0 \
+        --dataset.repo_id=ReubenLim/so101_cube_in_cup \
        --rtc.execution_horizon=8 \
        --device=cuda

    # With torch.compile for faster inference (PyTorch 2.0+)
    # Note: CUDA graphs disabled by default due to in-place ops in denoising loop
    uv run python examples/rtc/eval_dataset.py \
-        --policy.path=<USER>/smolvla_check_rtc_last3 \
-        --dataset.repo_id=<USER>/check_rtc \
+        --policy.path=helper2424/smolvla_check_rtc_last3 \
+        --dataset.repo_id=helper2424/check_rtc \
        --rtc.execution_horizon=8 \
        --device=mps \
        --use_torch_compile=true \
@@ -75,8 +75,8 @@ Usage:

    # With torch.compile on CUDA (CUDA graphs disabled by default)
    uv run python examples/rtc/eval_dataset.py \
-        --policy.path=<USER>/smolvla_check_rtc_last3 \
-        --dataset.repo_id=<USER>/check_rtc \
+        --policy.path=helper2424/smolvla_check_rtc_last3 \
+        --dataset.repo_id=helper2424/check_rtc \
        --rtc.execution_horizon=8 \
        --device=cuda \
        --use_torch_compile=true \
@@ -84,8 +84,8 @@ Usage:

    # Enable CUDA graphs (advanced - may cause tensor aliasing errors)
    uv run python examples/rtc/eval_dataset.py \
-        --policy.path=<USER>/smolvla_check_rtc_last3 \
-        --dataset.repo_id=<USER>/check_rtc \
+        --policy.path=helper2424/smolvla_check_rtc_last3 \
+        --dataset.repo_id=helper2424/check_rtc \
        --use_torch_compile=true \
        --torch_compile_backend=inductor \
        --torch_compile_mode=max-autotune \
@@ -113,9 +113,8 @@ from lerobot.configs import parser
 from lerobot.configs.default import DatasetConfig
 from lerobot.configs.policies import PreTrainedConfig
 from lerobot.configs.types import RTCAttentionSchedule
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
 from lerobot.datasets.factory import resolve_delta_timestamps
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.lerobot_dataset import LeRobotDataset, LeRobotDatasetMetadata
 from lerobot.policies.factory import get_policy_class, make_pre_post_processors
 from lerobot.policies.rtc.configuration_rtc import RTCConfig
 from lerobot.policies.rtc.debug_visualizer import RTCDebugVisualizer
@@ -28,7 +28,7 @@ For simulation environments, see eval_with_simulation.py
 Usage:
    # Run RTC with Real robot with RTC
    uv run examples/rtc/eval_with_real_robot.py \
-        --policy.path=<USER>/smolvla_check_rtc_last3 \
+        --policy.path=helper2424/smolvla_check_rtc_last3 \
        --policy.device=mps \
        --rtc.enabled=true \
        --rtc.execution_horizon=20 \
@@ -41,7 +41,7 @@ Usage:

    # Run RTC with Real robot without RTC
    uv run examples/rtc/eval_with_real_robot.py \
-        --policy.path=<USER>/smolvla_check_rtc_last3 \
+        --policy.path=helper2424/smolvla_check_rtc_last3 \
        --policy.device=mps \
        --rtc.enabled=false \
        --robot.type=so100_follower \
@@ -53,7 +53,7 @@ Usage:

    # Run RTC with Real robot with pi0.5 policy
    uv run examples/rtc/eval_with_real_robot.py \
-        --policy.path=<USER>/pi05_check_rtc \
+        --policy.path=helper2424/pi05_check_rtc \
        --policy.device=mps \
        --rtc.enabled=true \
        --rtc.execution_horizon=20 \
@@ -78,11 +78,10 @@ from torch import Tensor

 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig  # noqa: F401
 from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig  # noqa: F401
-from lerobot.cameras.zmq.configuration_zmq import ZMQCameraConfig  # noqa: F401
 from lerobot.configs import parser
 from lerobot.configs.policies import PreTrainedConfig
 from lerobot.configs.types import RTCAttentionSchedule
-from lerobot.datasets.feature_utils import build_dataset_frame, hw_to_dataset_features
+from lerobot.datasets.utils import build_dataset_frame, hw_to_dataset_features
 from lerobot.policies.factory import get_policy_class, make_pre_post_processors
 from lerobot.policies.rtc.action_queue import ActionQueue
 from lerobot.policies.rtc.configuration_rtc import RTCConfig
@@ -98,7 +97,6 @@ from lerobot.robots import (  # noqa: F401
    bi_so_follower,
    koch_follower,
    so_follower,
-    unitree_g1,
 )
 from lerobot.robots.utils import make_robot_from_config
 from lerobot.utils.constants import OBS_IMAGES
@@ -16,13 +16,15 @@

 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
 from lerobot.configs.types import FeatureType, PolicyFeature
-from lerobot.datasets.feature_utils import combine_feature_dicts
 from lerobot.datasets.lerobot_dataset import LeRobotDataset
 from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.datasets.utils import combine_feature_dicts
 from lerobot.model.kinematics import RobotKinematics
 from lerobot.policies.act.modeling_act import ACTPolicy
 from lerobot.policies.factory import make_pre_post_processors
 from lerobot.processor import (
+    RobotAction,
+    RobotObservation,
    RobotProcessorPipeline,
    make_default_teleop_action_processor,
 )
@@ -38,7 +40,6 @@ from lerobot.robots.so_follower.robot_kinematic_processor import (
    InverseKinematicsEEToJoints,
 )
 from lerobot.scripts.lerobot_record import record_loop
-from lerobot.types import RobotAction, RobotObservation
 from lerobot.utils.control_utils import init_keyboard_listener
 from lerobot.utils.utils import log_say
 from lerobot.utils.visualization_utils import init_rerun
@@ -16,11 +16,11 @@


 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
-from lerobot.datasets.feature_utils import combine_feature_dicts
 from lerobot.datasets.lerobot_dataset import LeRobotDataset
 from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.datasets.utils import combine_feature_dicts
 from lerobot.model.kinematics import RobotKinematics
-from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor import RobotAction, RobotObservation, RobotProcessorPipeline
 from lerobot.processor.converters import (
    observation_to_transition,
    robot_action_observation_to_transition,
@@ -35,7 +35,6 @@ from lerobot.robots.so_follower.robot_kinematic_processor import (
 )
 from lerobot.scripts.lerobot_record import record_loop
 from lerobot.teleoperators.so_leader import SO100Leader, SO100LeaderConfig
-from lerobot.types import RobotAction, RobotObservation
 from lerobot.utils.control_utils import init_keyboard_listener
 from lerobot.utils.utils import log_say
 from lerobot.utils.visualization_utils import init_rerun
@@ -19,7 +19,7 @@ import time

 from lerobot.datasets.lerobot_dataset import LeRobotDataset
 from lerobot.model.kinematics import RobotKinematics
-from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor import RobotAction, RobotObservation, RobotProcessorPipeline
 from lerobot.processor.converters import (
    robot_action_observation_to_transition,
    transition_to_robot_action,
@@ -28,7 +28,6 @@ from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
 from lerobot.robots.so_follower.robot_kinematic_processor import (
    InverseKinematicsEEToJoints,
 )
-from lerobot.types import RobotAction, RobotObservation
 from lerobot.utils.constants import ACTION
 from lerobot.utils.robot_utils import precise_sleep
 from lerobot.utils.utils import log_say
@@ -17,7 +17,7 @@
 import time

 from lerobot.model.kinematics import RobotKinematics
-from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor import RobotAction, RobotObservation, RobotProcessorPipeline
 from lerobot.processor.converters import (
    robot_action_observation_to_transition,
    robot_action_to_transition,
@@ -30,7 +30,6 @@ from lerobot.robots.so_follower.robot_kinematic_processor import (
    InverseKinematicsEEToJoints,
 )
 from lerobot.teleoperators.so_leader import SO100Leader, SO100LeaderConfig
-from lerobot.types import RobotAction, RobotObservation
 from lerobot.utils.robot_utils import precise_sleep
 from lerobot.utils.visualization_utils import init_rerun, log_rerun_data

@@ -19,9 +19,8 @@ from pathlib import Path
 import torch

 from lerobot.configs.types import FeatureType
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.feature_utils import dataset_to_policy_features
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.lerobot_dataset import LeRobotDataset, LeRobotDatasetMetadata
+from lerobot.datasets.utils import dataset_to_policy_features
 from lerobot.policies.diffusion.configuration_diffusion import DiffusionConfig
 from lerobot.policies.diffusion.modeling_diffusion import DiffusionPolicy
 from lerobot.policies.factory import make_pre_post_processors
@@ -20,9 +20,9 @@ from pathlib import Path
 import torch

 from lerobot.configs.types import FeatureType
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.feature_utils import dataset_to_policy_features
+from lerobot.datasets.lerobot_dataset import LeRobotDatasetMetadata
 from lerobot.datasets.streaming_dataset import StreamingLeRobotDataset
+from lerobot.datasets.utils import dataset_to_policy_features
 from lerobot.policies.act.configuration_act import ACTConfig
 from lerobot.policies.act.modeling_act import ACTPolicy
 from lerobot.policies.factory import make_pre_post_processors
@@ -5,9 +5,8 @@ from pathlib import Path
 import torch

 from lerobot.configs.types import FeatureType
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.feature_utils import dataset_to_policy_features
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.lerobot_dataset import LeRobotDataset, LeRobotDatasetMetadata
+from lerobot.datasets.utils import dataset_to_policy_features
 from lerobot.policies.act.configuration_act import ACTConfig
 from lerobot.policies.act.modeling_act import ACTPolicy
 from lerobot.policies.factory import make_pre_post_processors
@@ -1,7 +1,7 @@
 import torch

 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.lerobot_dataset import LeRobotDatasetMetadata
 from lerobot.policies.act.modeling_act import ACTPolicy
 from lerobot.policies.factory import make_pre_post_processors
 from lerobot.policies.utils import build_inference_frame, make_robot_action
@@ -5,9 +5,8 @@ from pathlib import Path
 import torch

 from lerobot.configs.types import FeatureType
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.feature_utils import dataset_to_policy_features
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.lerobot_dataset import LeRobotDataset, LeRobotDatasetMetadata
+from lerobot.datasets.utils import dataset_to_policy_features
 from lerobot.policies.diffusion.configuration_diffusion import DiffusionConfig
 from lerobot.policies.diffusion.modeling_diffusion import DiffusionPolicy
 from lerobot.policies.factory import make_pre_post_processors
@@ -1,7 +1,7 @@
 import torch

 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.lerobot_dataset import LeRobotDatasetMetadata
 from lerobot.policies.diffusion.modeling_diffusion import DiffusionPolicy
 from lerobot.policies.factory import make_pre_post_processors
 from lerobot.policies.utils import build_inference_frame, make_robot_action
@@ -1,7 +1,7 @@
 import torch

 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
-from lerobot.datasets.feature_utils import hw_to_dataset_features
+from lerobot.datasets.utils import hw_to_dataset_features
 from lerobot.policies.factory import make_pre_post_processors
 from lerobot.policies.pi0.modeling_pi0 import PI0Policy
 from lerobot.policies.utils import build_inference_frame, make_robot_action
@@ -6,8 +6,8 @@ from queue import Empty, Full
 import torch
 import torch.optim as optim

-from lerobot.datasets.feature_utils import hw_to_dataset_features
 from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import hw_to_dataset_features
 from lerobot.envs.configs import HILSerlProcessorConfig, HILSerlRobotEnvConfig
 from lerobot.policies.sac.configuration_sac import SACConfig
 from lerobot.policies.sac.modeling_sac import SACPolicy
@@ -1,7 +1,7 @@
 import torch

 from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
-from lerobot.datasets.feature_utils import hw_to_dataset_features
+from lerobot.datasets.utils import hw_to_dataset_features
 from lerobot.policies.factory import make_pre_post_processors
 from lerobot.policies.smolvla.modeling_smolvla import SmolVLAPolicy
 from lerobot.policies.utils import build_inference_frame, make_robot_action
@@ -14,20 +14,20 @@
 # See the License for the specific language governing permissions and
 # limitations under the License.

+import argparse
 import logging
+import time
 from collections import deque

 import numpy as np
 import onnxruntime as ort
 from huggingface_hub import hf_hub_download

-from lerobot.robots.unitree_g1.g1_utils import (
-    REMOTE_AXES,
-    REMOTE_BUTTONS,
-    G1_29_JointIndex,
-    get_gravity_orientation,
-)
+from lerobot.robots.unitree_g1.config_unitree_g1 import UnitreeG1Config
+from lerobot.robots.unitree_g1.g1_utils import G1_29_JointIndex
+from lerobot.robots.unitree_g1.unitree_g1 import UnitreeG1

+logging.basicConfig(level=logging.INFO)
 logger = logging.getLogger(__name__)


@@ -36,13 +36,18 @@ GROOT_DEFAULT_ANGLES[[0, 6]] = -0.1  # Hip pitch
 GROOT_DEFAULT_ANGLES[[3, 9]] = 0.3  # Knee
 GROOT_DEFAULT_ANGLES[[4, 10]] = -0.2  # Ankle pitch

+MISSING_JOINTS = []
+G1_MODEL = "g1_23"  # Or "g1_29"
+if G1_MODEL == "g1_23":
+    MISSING_JOINTS = [12, 14, 20, 21, 27, 28]  # Waist yaw/pitch, wrist pitch/yaw
+
 # Control parameters
 ACTION_SCALE = 0.25
 CONTROL_DT = 0.02  # 50Hz
 ANG_VEL_SCALE: float = 0.25
 DOF_POS_SCALE: float = 1.0
 DOF_VEL_SCALE: float = 0.05
-CMD_SCALE: list[float] = [2.0, 2.0, 0.25]
+CMD_SCALE: list = [2.0, 2.0, 0.25]


 DEFAULT_GROOT_REPO_ID = "nepyope/GR00T-WholeBodyControl_g1"
@@ -80,11 +85,11 @@ def load_groot_policies(
 class GrootLocomotionController:
    """GR00T lower-body locomotion controller for the Unitree G1."""

-    control_dt = CONTROL_DT  # Expose for unitree_g1.py
-
-    def __init__(self):
-        # Load policies
-        self.policy_balance, self.policy_walk = load_groot_policies()
+    def __init__(self, policy_balance, policy_walk, robot, config):
+        self.policy_balance = policy_balance
+        self.policy_walk = policy_walk
+        self.robot = robot
+        self.config = config

        self.cmd = np.array([0.0, 0.0, 0.0], dtype=np.float32)  # vx, vy, theta_dot

@@ -104,60 +109,45 @@ class GrootLocomotionController:

        logger.info("GrootLocomotionController initialized")

-    def reset(self) -> None:
-        """Reset internal state for a new episode."""
-        self.cmd[:] = 0.0
-        self.groot_qj_all[:] = 0.0
-        self.groot_dqj_all[:] = 0.0
-        self.groot_action[:] = 0.0
-        self.groot_obs_single[:] = 0.0
-        self.groot_obs_stacked[:] = 0.0
-        self.groot_height_cmd = 0.74
-        self.groot_orientation_cmd[:] = 0.0
-        self.groot_obs_history.clear()
-        for _ in range(6):
-            self.groot_obs_history.append(np.zeros(86, dtype=np.float32))
+    def run_step(self):
+        # Get current observation
+        obs = self.robot.get_observation()

-    def run_step(self, action: dict, lowstate) -> dict:
-        """Run one step of the locomotion controller.
+        if not obs:
+            return

-        Args:
-            action: Action dict containing remote.lx/ly/rx/ry and buttons
-            lowstate: Robot lowstate containing motor positions/velocities and IMU
-
-        Returns:
-            Action dict for lower body joints (0-14)
-        """
-        if lowstate is None:
-            return {}
-
-        buttons = [int(action.get(k, 0)) for k in REMOTE_BUTTONS]
-        if buttons[0]:  # R1 - raise waist
+        # Get command from remote controller
+        if obs["remote.buttons"][0]:  # R1 - raise waist
            self.groot_height_cmd += 0.001
            self.groot_height_cmd = np.clip(self.groot_height_cmd, 0.50, 1.00)
-        if buttons[4]:  # R2 - lower waist
+        if obs["remote.buttons"][4]:  # R2 - lower waist
            self.groot_height_cmd -= 0.001
            self.groot_height_cmd = np.clip(self.groot_height_cmd, 0.50, 1.00)

-        lx, ly, rx, _ry = (action.get(k, 0.0) for k in REMOTE_AXES)
-        self.cmd[0] = ly  # Forward/backward
-        self.cmd[1] = -lx  # Left/right (negated)
-        self.cmd[2] = -rx  # Rotation rate (negated)
+        self.cmd[0] = obs["remote.ly"]  # Forward/backward
+        self.cmd[1] = obs["remote.lx"] * -1  # Left/right
+        self.cmd[2] = obs["remote.rx"] * -1  # Rotation rate

-        # Get joint positions and velocities from lowstate
+        # Get joint positions and velocities from flat dict
        for motor in G1_29_JointIndex:
+            name = motor.name
            idx = motor.value
-            self.groot_qj_all[idx] = lowstate.motor_state[idx].q
-            self.groot_dqj_all[idx] = lowstate.motor_state[idx].dq
+            self.groot_qj_all[idx] = obs[f"{name}.q"]
+            self.groot_dqj_all[idx] = obs[f"{name}.dq"]
+
+        # Adapt observation for g1_23dof
+        for idx in MISSING_JOINTS:
+            self.groot_qj_all[idx] = 0.0
+            self.groot_dqj_all[idx] = 0.0

        # Scale joint positions and velocities
        qj_obs = self.groot_qj_all.copy()
        dqj_obs = self.groot_dqj_all.copy()

        # Express IMU data in gravity frame of reference
-        quat = lowstate.imu_state.quaternion
-        ang_vel = np.array(lowstate.imu_state.gyroscope, dtype=np.float32)
-        gravity_orientation = get_gravity_orientation(quat)
+        quat = [obs["imu.quat.w"], obs["imu.quat.x"], obs["imu.quat.y"], obs["imu.quat.z"]]
+        ang_vel = np.array([obs["imu.gyro.x"], obs["imu.gyro.y"], obs["imu.gyro.z"]], dtype=np.float32)
+        gravity_orientation = self.robot.get_gravity_orientation(quat)

        # Scale joint positions and velocities before policy inference
        qj_obs = (qj_obs - GROOT_DEFAULT_ANGLES) * DOF_POS_SCALE
@@ -196,10 +186,73 @@ class GrootLocomotionController:
        # Transform action back to target joint positions
        target_dof_pos_15 = GROOT_DEFAULT_ANGLES[:15] + self.groot_action * ACTION_SCALE

-        # Build action dict
+        # Build action dict (only first 15 joints for GR00T)
        action_dict = {}
        for i in range(15):
            motor_name = G1_29_JointIndex(i).name
            action_dict[f"{motor_name}.q"] = float(target_dof_pos_15[i])

-        return action_dict
+        # Zero out missing joints for g1_23dof
+        for joint_idx in MISSING_JOINTS:
+            motor_name = G1_29_JointIndex(joint_idx).name
+            action_dict[f"{motor_name}.q"] = 0.0
+
+        # Send action to robot
+        self.robot.send_action(action_dict)
+
+
+def run(repo_id: str = DEFAULT_GROOT_REPO_ID) -> None:
+    """Main function to run the GR00T locomotion controller.
+
+    Args:
+        repo_id: Hugging Face Hub repository ID for GR00T policies.
+    """
+    # Load policies
+    policy_balance, policy_walk = load_groot_policies(repo_id=repo_id)
+
+    # Initialize robot
+    config = UnitreeG1Config()
+    robot = UnitreeG1(config)
+
+    robot.connect()
+
+    # Initialize gr00T locomotion controller
+    groot_controller = GrootLocomotionController(
+        policy_balance=policy_balance,
+        policy_walk=policy_walk,
+        robot=robot,
+        config=config,
+    )
+
+    try:
+        robot.reset(CONTROL_DT, GROOT_DEFAULT_ANGLES)
+
+        logger.info("Use joystick: LY=fwd/back, LX=left/right, RX=rotate, R1=raise waist, R2=lower waist")
+        logger.info("Press Ctrl+C to stop")
+
+        # Run step
+        while not robot._shutdown_event.is_set():
+            start_time = time.time()
+            groot_controller.run_step()
+            elapsed = time.time() - start_time
+            sleep_time = max(0, CONTROL_DT - elapsed)
+            time.sleep(sleep_time)
+    except KeyboardInterrupt:
+        logger.info("Stopping locomotion...")
+    finally:
+        if robot.is_connected:
+            robot.disconnect()
+        logger.info("Done!")
+
+
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser(description="GR00T Locomotion Controller for Unitree G1")
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        default=DEFAULT_GROOT_REPO_ID,
+        help=f"Hugging Face Hub repo ID for GR00T policies (default: {DEFAULT_GROOT_REPO_ID})",
+    )
+    args = parser.parse_args()
+
+    run(repo_id=args.repo_id)
@@ -14,21 +14,21 @@
 # See the License for the specific language governing permissions and
 # limitations under the License.

+import argparse
 import json
 import logging
+import time

 import numpy as np
 import onnx
 import onnxruntime as ort
 from huggingface_hub import hf_hub_download

-from lerobot.robots.unitree_g1.g1_utils import (
-    REMOTE_AXES,
-    G1_29_JointArmIndex,
-    G1_29_JointIndex,
-    get_gravity_orientation,
-)
+from lerobot.robots.unitree_g1.config_unitree_g1 import UnitreeG1Config
+from lerobot.robots.unitree_g1.g1_utils import G1_29_JointIndex
+from lerobot.robots.unitree_g1.unitree_g1 import UnitreeG1

+logging.basicConfig(level=logging.INFO)
 logger = logging.getLogger(__name__)

 DEFAULT_ANGLES = np.zeros(29, dtype=np.float32)
@@ -40,13 +40,18 @@ DEFAULT_ANGLES[16] = 0.2  # Left shoulder roll
 DEFAULT_ANGLES[23] = -0.2  # Right shoulder roll
 DEFAULT_ANGLES[[18, 25]] = 0.6  # Elbow

+MISSING_JOINTS = []
+G1_MODEL = "g1_23"  # Or "g1_29"
+if G1_MODEL == "g1_23":
+    MISSING_JOINTS = [12, 14, 20, 21, 27, 28]  # Waist yaw/pitch, wrist pitch/yaw
+
 # Control parameters
 ACTION_SCALE = 0.25
-CONTROL_DT = 0.005  # 200Hz
+CONTROL_DT = 0.02  # 50Hz
 ANG_VEL_SCALE = 0.25
 DOF_POS_SCALE = 1.0
 DOF_VEL_SCALE = 0.05
-GAIT_PERIOD = 0.5
+GAIT_PERIOD = 1.0


 DEFAULT_HOLOSOMA_REPO_ID = "nepyope/holosoma_locomotion"
@@ -82,7 +87,7 @@ def load_policy(
    logger.info(f"Policy loaded: {policy.get_inputs()[0].shape} → {policy.get_outputs()[0].shape}")

    # Extract KP/KD from ONNX metadata
-    model = onnx.load(policy_path, load_external_data=False)
+    model = onnx.load(policy_path)
    metadata = {prop.key: prop.value for prop in model.metadata_props}

    if "kp" not in metadata or "kd" not in metadata:
@@ -96,13 +101,15 @@ def load_policy(


 class HolosomaLocomotionController:
-    """Holosoma lower-body locomotion controller for Unitree G1."""
+    """Holosoma whole-body locomotion controller for Unitree G1."""

-    control_dt = CONTROL_DT  # Expose for unitree_g1.py
+    def __init__(self, policy, robot, kp: np.ndarray, kd: np.ndarray):
+        self.policy = policy
+        self.robot = robot

-    def __init__(self):
-        # Load policy and gains
-        self.policy, self.kp, self.kd = load_policy()
+        # Override robot's PD gains with policy gains
+        self.robot.kp = kp
+        self.robot.kd = kd

        self.cmd = np.zeros(3, dtype=np.float32)

@@ -117,55 +124,35 @@ class HolosomaLocomotionController:
        self.phase_dt = 2 * np.pi / ((1.0 / CONTROL_DT) * GAIT_PERIOD)
        self.is_standing = True

-        logger.info("HolosomaLocomotionController initialized")
+    def run_step(self):
+        # Get current observation
+        obs = self.robot.get_observation()

-    def reset(self) -> None:
-        """Reset internal state for a new episode."""
-        self.cmd[:] = 0.0
-        self.qj[:] = 0.0
-        self.dqj[:] = 0.0
-        self.obs[:] = 0.0
-        self.last_action[:] = 0.0
-        self.phase = np.array([[0.0, np.pi]], dtype=np.float32)
-        self.is_standing = True
+        if not obs:
+            return

-    def run_step(self, action: dict, lowstate) -> dict:
-        """Run one step of the locomotion controller.
-
-        Args:
-            action: Action dict containing remote.lx/ly/rx/ry
-            lowstate: Robot lowstate containing motor positions/velocities and IMU
-
-        Returns:
-            Action dict for lower body joints (0-14)
-        """
-        if lowstate is None:
-            return {}
-
-        lx, ly, rx, _ry = (action.get(k, 0.0) for k in REMOTE_AXES)
-        ly = ly if abs(ly) > 0.1 else 0.0
-        lx = lx if abs(lx) > 0.1 else 0.0
-        rx = rx if abs(rx) > 0.1 else 0.0
-        ly = np.clip(ly, -0.3, 0.3)
-        lx = np.clip(lx, -0.3, 0.3)
+        # Get command from remote controller
+        ly = obs["remote.ly"] if abs(obs["remote.ly"]) > 0.1 else 0.0
+        lx = obs["remote.lx"] if abs(obs["remote.lx"]) > 0.1 else 0.0
+        rx = obs["remote.rx"] if abs(obs["remote.rx"]) > 0.1 else 0.0
        self.cmd[:] = [ly, -lx, -rx]

-        # Get joint positions and velocities from lowstate
+        # Get joint positions and velocities
        for motor in G1_29_JointIndex:
+            name = motor.name
            idx = motor.value
-            self.qj[idx] = lowstate.motor_state[idx].q
-            self.dqj[idx] = lowstate.motor_state[idx].dq
+            self.qj[idx] = obs[f"{name}.q"]
+            self.dqj[idx] = obs[f"{name}.dq"]

-        # Hide arm positions from policy (show DEFAULT_ANGLES instead)
-        # This prevents policy from reacting to teleop arm movements
-        for arm_joint in G1_29_JointArmIndex:
-            self.qj[arm_joint.value] = DEFAULT_ANGLES[arm_joint.value]
-            self.dqj[arm_joint.value] = 0.0
+        # Adapt observation for g1_23dof
+        for idx in MISSING_JOINTS:
+            self.qj[idx] = 0.0
+            self.dqj[idx] = 0.0

        # Express IMU data in gravity frame of reference
-        quat = lowstate.imu_state.quaternion
-        ang_vel = np.array(lowstate.imu_state.gyroscope, dtype=np.float32)
-        gravity = get_gravity_orientation(quat)
+        quat = [obs["imu.quat.w"], obs["imu.quat.x"], obs["imu.quat.y"], obs["imu.quat.z"]]
+        ang_vel = np.array([obs["imu.gyro.x"], obs["imu.gyro.y"], obs["imu.gyro.z"]], dtype=np.float32)
+        gravity = self.robot.get_gravity_orientation(quat)

        # Scale joint positions and velocities before policy inference
        qj_obs = (self.qj - DEFAULT_ANGLES) * DOF_POS_SCALE
@@ -199,16 +186,79 @@ class HolosomaLocomotionController:
        # Run policy inference
        ort_in = {self.policy.get_inputs()[0].name: self.obs.reshape(1, -1).astype(np.float32)}
        raw_action = self.policy.run(None, ort_in)[0].squeeze()
-        policy_action = np.clip(raw_action, -100.0, 100.0)
-        self.last_action = policy_action.copy()
+        action = np.clip(raw_action, -100.0, 100.0)
+        self.last_action = action.copy()

        # Transform action back to target joint positions
-        target = DEFAULT_ANGLES + policy_action * ACTION_SCALE
+        target = DEFAULT_ANGLES + action * ACTION_SCALE

-        # Build action dict (first 15 joints only)
+        # Build action dict
        action_dict = {}
-        for i in range(15):
-            motor_name = G1_29_JointIndex(i).name
-            action_dict[f"{motor_name}.q"] = float(target[i])
+        for motor in G1_29_JointIndex:
+            action_dict[f"{motor.name}.q"] = float(target[motor.value])

-        return action_dict
+        # Zero out missing joints for g1_23dof
+        for joint_idx in MISSING_JOINTS:
+            motor_name = G1_29_JointIndex(joint_idx).name
+            action_dict[f"{motor_name}.q"] = 0.0
+
+        # Send action to robot
+        self.robot.send_action(action_dict)
+
+
+def run(repo_id: str = DEFAULT_HOLOSOMA_REPO_ID, policy_type: str = "fastsac") -> None:
+    """Main function to run the Holosoma locomotion controller.
+
+    Args:
+        repo_id: Hugging Face Hub repository ID for Holosoma policies.
+        policy_type: Policy type to use ('fastsac' or 'ppo').
+    """
+    # Load policy and gains
+    policy, kp, kd = load_policy(repo_id=repo_id, policy_type=policy_type)
+
+    # Initialize robot
+    config = UnitreeG1Config()
+    robot = UnitreeG1(config)
+    robot.connect()
+
+    holosoma_controller = HolosomaLocomotionController(policy, robot, kp, kd)
+
+    try:
+        robot.reset(CONTROL_DT, DEFAULT_ANGLES)
+
+        logger.info("Use joystick: LY=fwd/back, LX=left/right, RX=rotate")
+        logger.info("Press Ctrl+C to stop")
+
+        # Run step
+        while not robot._shutdown_event.is_set():
+            start_time = time.time()
+            holosoma_controller.run_step()
+            elapsed = time.time() - start_time
+            sleep_time = max(0, CONTROL_DT - elapsed)
+            time.sleep(sleep_time)
+    except KeyboardInterrupt:
+        logger.info("Stopping locomotion...")
+    finally:
+        if robot.is_connected:
+            robot.disconnect()
+        logger.info("Done!")
+
+
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser(description="Holosoma Locomotion Controller for Unitree G1")
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        default=DEFAULT_HOLOSOMA_REPO_ID,
+        help=f"Hugging Face Hub repo ID for Holosoma policies (default: {DEFAULT_HOLOSOMA_REPO_ID})",
+    )
+    parser.add_argument(
+        "--policy",
+        type=str,
+        choices=["fastsac", "ppo"],
+        default="fastsac",
+        help="Policy type to use: 'fastsac' (default) or 'ppo'",
+    )
+    args = parser.parse_args()
+
+    run(repo_id=args.repo_id, policy_type=args.policy)
@@ -25,11 +25,11 @@ discord = "https://discord.gg/s3KuuzsPFb"

 [project]
 name = "lerobot"
-version = "0.5.1"
+version = "0.4.4"
 description = "🤗 LeRobot: State-of-the-art Machine Learning for Real-World Robotics in Pytorch"
 dynamic = ["readme"]
 license = { text = "Apache-2.0" }
-requires-python = ">=3.12"
+requires-python = ">=3.10"
 authors = [
    { name = "Rémi Cadène", email = "re.cadene@gmail.com" },
    { name = "Simon Alibert", email = "alibert.sim@gmail.com" },
@@ -50,8 +50,7 @@ classifiers = [
    "Intended Audience :: Education",
    "Intended Audience :: Science/Research",
    "License :: OSI Approved :: Apache Software License",
-    "Programming Language :: Python :: 3.12",
-    "Programming Language :: Python :: 3.13",
+    "Programming Language :: Python :: 3.10",
    "Topic :: Software Development :: Build Tools",
    "Topic :: Scientific/Engineering :: Artificial Intelligence",
 ]
@@ -60,30 +59,28 @@ keywords = ["lerobot", "huggingface", "robotics",  "machine learning", "artifici
 dependencies = [

    # Hugging Face dependencies
-    "datasets>=4.0.0,<5.0.0",
+    "datasets>=4.0.0,<4.2.0",
    "diffusers>=0.27.2,<0.36.0",
-    "huggingface-hub>=1.0.0,<2.0.0",
+    "huggingface-hub[hf-transfer,cli]>=0.34.2,<0.36.0",
    "accelerate>=1.10.0,<2.0.0",

    # Core dependencies
-    "numpy>=2.0.0,<2.3.0", # NOTE: Explicitly listing numpy helps the resolver converge faster. Upper bound imposed by opencv-python-headless.
    "setuptools>=71.0.0,<81.0.0",
    "cmake>=3.29.0.1,<4.2.0",
-    "packaging>=24.2,<26.0",
-
-    "torch>=2.2.1,<2.11.0",
-    "torchcodec>=0.2.1,<0.11.0; sys_platform != 'win32' and (sys_platform != 'linux' or (platform_machine != 'aarch64' and platform_machine != 'arm64' and platform_machine != 'armv7l')) and (sys_platform != 'darwin' or platform_machine != 'x86_64')",
-    "torchvision>=0.21.0,<0.26.0",
-
    "einops>=0.8.0,<0.9.0",
-    "opencv-python-headless>=4.9.0,<4.14.0",
+    "opencv-python-headless>=4.9.0,<4.13.0",
    "av>=15.0.0,<16.0.0",
    "jsonlines>=4.0.0,<5.0.0",
-    "pynput>=1.7.8,<1.9.0",
+    "packaging>=24.2,<26.0",
+    "pynput>=1.7.7,<1.9.0",
    "pyserial>=3.5,<4.0",
-
    "wandb>=0.24.0,<0.25.0",
-    "draccus==0.10.0", # TODO: Relax version constraint
+
+    "torch>=2.2.1,<2.8.0", # TODO: Bumb dependency
+    "torchcodec>=0.2.1,<0.6.0; sys_platform != 'win32' and (sys_platform != 'linux' or (platform_machine != 'aarch64' and platform_machine != 'arm64' and platform_machine != 'armv7l')) and (sys_platform != 'darwin' or platform_machine != 'x86_64')", # TODO: Bumb dependency
+    "torchvision>=0.21.0,<0.23.0", # TODO: Bumb dependency
+
+    "draccus==0.10.0", # TODO: Remove ==
    "gymnasium>=1.1.1,<2.0.0",
    "rerun-sdk>=0.24.0,<0.27.0",

@@ -98,20 +95,14 @@ dependencies = [

 # Common
 pygame-dep = ["pygame>=2.5.1,<2.7.0"]
-placo-dep = ["placo>=0.9.6,<0.9.17"]
-transformers-dep = ["transformers>=5.3.0,<6.0.0"]
+placo-dep = ["placo>=0.9.6,<0.10.0"]
+transformers-dep = ["transformers>=4.57.1,<5.0.0"]
 grpcio-dep = ["grpcio==1.73.1", "protobuf>=6.31.1,<6.32.0"]
-can-dep = ["python-can>=4.2.0,<5.0.0"]
-peft-dep = ["peft>=0.18.0,<1.0.0"]
-scipy-dep = ["scipy>=1.14.0,<2.0.0"]
-qwen-vl-utils-dep = ["qwen-vl-utils>=0.0.11,<0.1.0"]
-matplotlib-dep = ["matplotlib>=3.10.3,<4.0.0", "contourpy>=1.3.0,<2.0.0"] # NOTE: Explicitly listing contourpy helps the resolver converge faster.

 # Motors
 feetech = ["feetech-servo-sdk>=1.0.0,<2.0.0"]
 dynamixel = ["dynamixel-sdk>=3.7.31,<3.9.0"]
-damiao = ["lerobot[can-dep]"]
-robstride = ["lerobot[can-dep]"]
+damiao = ["python-can>=4.2.0,<5.0.0"]

 # Robots
 openarms = ["lerobot[damiao]"]
@@ -119,35 +110,34 @@ gamepad = ["lerobot[pygame-dep]", "hidapi>=0.14.0,<0.15.0"]
 hopejr = ["lerobot[feetech]", "lerobot[pygame-dep]"]
 lekiwi = ["lerobot[feetech]", "pyzmq>=26.2.1,<28.0.0"]
 unitree_g1 = [
-    # "unitree-sdk2==1.0.1",
    "pyzmq>=26.2.1,<28.0.0",
    "onnxruntime>=1.16.0,<2.0.0",
-    "onnx>=1.16.0,<2.0.0",
+    "pin>=3.0.0,<4.0.0",
    "meshcat>=0.3.0,<0.4.0",
-    "lerobot[matplotlib-dep]",
-    "lerobot[pygame-dep]",
+    "matplotlib>=3.9.0,<4.0.0",
+    "casadi>=3.6.0,<4.0.0",
 ]
 reachy2 = ["reachy2_sdk>=1.0.15,<1.1.0"]
 kinematics = ["lerobot[placo-dep]"]
 intelrealsense = [
    "pyrealsense2>=2.55.1.6486,<2.57.0 ; sys_platform != 'darwin'",
-    "pyrealsense2-macosx>=2.54,<2.57.0 ; sys_platform == 'darwin'",
+    "pyrealsense2-macosx>=2.54,<2.55.0 ; sys_platform == 'darwin'",
 ]
-phone = ["hebi-py>=2.8.0,<2.12.0", "teleop>=0.1.0,<0.2.0", "fastapi<1.0", "lerobot[scipy-dep]"]
+phone = ["hebi-py>=2.8.0,<2.12.0", "teleop>=0.1.0,<0.2.0", "fastapi<1.0"]

 # Policies
 wallx = [
-    "lerobot[transformers-dep]",
-    "lerobot[peft]",
-    "lerobot[scipy-dep]",
-    "torchdiffeq>=0.2.4,<0.3.0",
-    "lerobot[qwen-vl-utils-dep]",
+    "transformers==4.49.0",
+    "peft==0.17.1",
+    "scipy==1.15.3",
+    "torchdiffeq==0.2.5",
+    "qwen_vl_utils==0.0.11"
 ]
-pi = ["lerobot[transformers-dep]", "lerobot[scipy-dep]"]
+pi = ["transformers @ git+https://github.com/huggingface/transformers.git@fix/lerobot_openpi", "scipy>=1.10.1,<1.15"]
 smolvla = ["lerobot[transformers-dep]", "num2words>=0.5.14,<0.6.0", "accelerate>=1.7.0,<2.0.0", "safetensors>=0.4.3,<1.0.0"]
 groot = [
    "lerobot[transformers-dep]",
-    "lerobot[peft]",
+    "peft>=0.13.0,<1.0.0",
    "dm-tree>=0.1.8,<1.0.0",
    "timm>=1.0.0,<1.1.0",
    "safetensors>=0.4.3,<1.0.0",
@@ -156,13 +146,13 @@ groot = [
    "ninja>=1.11.1,<2.0.0",
    "flash-attn>=2.5.9,<3.0.0 ; sys_platform != 'darwin'"
 ]
-sarm = ["lerobot[transformers-dep]", "faker>=33.0.0,<35.0.0", "lerobot[matplotlib-dep]", "lerobot[qwen-vl-utils-dep]"]
+sarm = ["lerobot[transformers-dep]", "faker>=33.0.0,<35.0.0", "matplotlib>=3.10.3,<4.0.0", "qwen-vl-utils>=0.0.14,<0.1.0"]
 xvla = ["lerobot[transformers-dep]"]
 hilserl = ["lerobot[transformers-dep]", "gym-hil>=0.1.13,<0.2.0", "lerobot[grpcio-dep]", "lerobot[placo-dep]"]

 # Features
-async = ["lerobot[grpcio-dep]", "lerobot[matplotlib-dep]"]
-peft = ["lerobot[transformers-dep]", "lerobot[peft-dep]"]
+async = ["lerobot[grpcio-dep]", "matplotlib>=3.10.3,<4.0.0"]
+peft = ["lerobot[transformers-dep]", "peft>=0.18.0,<1.0.0"]

 # Development
 dev = ["pre-commit>=3.7.0,<5.0.0", "debugpy>=1.8.1,<1.9.0", "lerobot[grpcio-dep]", "grpcio-tools==1.73.1", "mypy>=1.19.1"]
@@ -170,19 +160,13 @@ test = ["pytest>=8.1.0,<9.0.0", "pytest-timeout>=2.4.0,<3.0.0", "pytest-cov>=5.0
 video_benchmark = ["scikit-image>=0.23.2,<0.26.0", "pandas>=2.2.2,<2.4.0"]

 # Simulation
-# NOTE: Explicitly listing scipy helps flatten the dependecy tree.
-aloha = ["gym-aloha>=0.1.2,<0.2.0", "lerobot[scipy-dep]"]
+aloha = ["gym-aloha>=0.1.2,<0.2.0"]
 pusht = ["gym-pusht>=0.1.5,<0.2.0", "pymunk>=6.6.0,<7.0.0"] # TODO: Fix pymunk version in gym-pusht instead
-libero = ["lerobot[transformers-dep]", "hf-libero>=0.1.3,<0.2.0; sys_platform == 'linux'", "lerobot[scipy-dep]"]
-metaworld = ["metaworld==3.0.0", "lerobot[scipy-dep]"]
+libero = ["lerobot[transformers-dep]", "hf-libero>=0.1.3,<0.2.0"]
+metaworld = ["metaworld==3.0.0"]

 # All
 all = [
-    # NOTE(resolver hint): scipy is pulled in transitively via lerobot[scipy-dep] through
-    # multiple extras (aloha, metaworld, pi, wallx, phone). Listing it explicitly
-    # helps pip's resolver converge by constraining scipy early, before it encounters
-    # the loose scipy requirements from transitive deps like dm-control and metaworld.
-    "scipy>=1.14.0,<2.0.0",
    "lerobot[dynamixel]",
    "lerobot[gamepad]",
    "lerobot[hopejr]",
@@ -190,8 +174,8 @@ all = [
    "lerobot[reachy2]",
    "lerobot[kinematics]",
    "lerobot[intelrealsense]",
-    "lerobot[wallx]",
-    "lerobot[pi]",
+    # "lerobot[wallx]",
+    # "lerobot[pi]", TODO(Pepijn): Update pi to transformers v5
    "lerobot[smolvla]",
    # "lerobot[groot]", TODO(Steven): Gr00t requires specific installation instructions for flash-attn
    "lerobot[xvla]",
@@ -203,11 +187,10 @@ all = [
    "lerobot[aloha]",
    "lerobot[pusht]",
    "lerobot[phone]",
-    "lerobot[libero]; sys_platform == 'linux'",
+    "lerobot[libero]",
    "lerobot[metaworld]",
    "lerobot[sarm]",
    "lerobot[peft]",
-    # "lerobot[unitree_g1]", TODO: Unitree requires specific installation instructions for unitree_sdk2
 ]

 [project.scripts]
@@ -229,14 +212,11 @@ lerobot-edit-dataset="lerobot.scripts.lerobot_edit_dataset:main"
 lerobot-setup-can="lerobot.scripts.lerobot_setup_can:main"

 # ---------------- Tool Configurations ----------------
-[tool.setuptools.package-data]
-lerobot = ["envs/*.json"]
-
 [tool.setuptools.packages.find]
 where = ["src"]

 [tool.ruff]
-target-version = "py312"
+target-version = "py310"
 line-length = 110
 exclude = ["tests/artifacts/**/*.safetensors", "*_pb2.py", "*_pb2_grpc.py"]

@@ -328,7 +308,7 @@ default.extend-ignore-identifiers-re = [
 # Uncomment [tool.mypy] first, then uncomment individual module overrides as they get proper type annotations

 [tool.mypy]
-python_version = "3.12"
+python_version = "3.10"
 ignore_missing_imports = true
 follow_imports = "skip"
 # warn_return_any = true
@@ -412,3 +392,85 @@ ignore_errors = false
 # [[tool.mypy.overrides]]
 # module = "lerobot.scripts.*"
 # ignore_errors = false
+
+[tool.uv]
+# wallx requires transformers==4.49.0 which conflicts with other extras that need >=4.53.0
+conflicts = [
+    [
+        { extra = "wallx" },
+        { extra = "transformers-dep" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "pi" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "smolvla" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "groot" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "xvla" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "sarm" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "hilserl" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "libero" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "peft" },
+    ],
+    [
+        { extra = "wallx" },
+        { extra = "all" },
+    ],
+    # pi uses custom branch which conflicts with transformers-dep
+    [
+        { extra = "pi" },
+        { extra = "transformers-dep" },
+    ],
+    [
+        { extra = "pi" },
+        { extra = "smolvla" },
+    ],
+    [
+        { extra = "pi" },
+        { extra = "groot" },
+    ],
+    [
+        { extra = "pi" },
+        { extra = "xvla" },
+    ],
+    [
+        { extra = "pi" },
+        { extra = "sarm" },
+    ],
+    [
+        { extra = "pi" },
+        { extra = "hilserl" },
+    ],
+    [
+        { extra = "pi" },
+        { extra = "libero" },
+    ],
+    [
+        { extra = "pi" },
+        { extra = "peft" },
+    ],
+    [
+        { extra = "pi" },
+        { extra = "all" },
+    ],
+]
@@ -1,73 +1,76 @@
 #
-# This file is autogenerated by pip-compile with Python 3.12
+# This file is autogenerated by pip-compile with Python 3.10
 # by the following command:
 #
 #    pip-compile --output-file=requirements-macos.txt requirements.in
 #
 -e .[all]
    # via -[all]
-absl-py==2.4.0
+absl-py==2.3.1
    # via
    #   dm-control
    #   dm-env
    #   dm-tree
    #   labmaze
    #   mujoco
-accelerate==1.13.0
+    #   tensorboard
+accelerate==1.11.0
    # via
    #   lerobot
    #   peft
 aiohappyeyeballs==2.6.1
    # via aiohttp
-aiohttp==3.13.3
+aiohttp==3.13.1
    # via fsspec
 aiosignal==1.4.0
    # via aiohttp
-annotated-doc==0.0.4
-    # via
-    #   fastapi
-    #   typer
 annotated-types==0.7.0
    # via pydantic
-anyio==4.12.1
+antlr4-python3-runtime==4.9.3
+    # via
+    #   hydra-core
+    #   omegaconf
+anyio==4.11.0
    # via
-    #   httpx
    #   starlette
    #   watchfiles
-asttokens==3.0.1
+asttokens==3.0.0
    # via stack-data
+async-timeout==5.0.1
+    # via aiohttp
 attrs==25.4.0
    # via
    #   aiohttp
    #   dm-tree
    #   jsonlines
+    #   jsonschema
+    #   referencing
    #   rerun-sdk
 av==15.1.0
+    # via lerobot
+bddl==1.0.1
+    # via libero
+certifi==2025.10.5
    # via
-    #   lerobot
-    #   qwen-vl-utils
-certifi==2026.2.25
-    # via
-    #   httpcore
-    #   httpx
    #   requests
    #   sentry-sdk
 cffi==2.0.0
    # via pymunk
-cfgv==3.5.0
+cfgv==3.4.0
    # via pre-commit
-charset-normalizer==3.4.5
+charset-normalizer==3.4.4
    # via requests
-click==8.3.1
+click==8.3.0
    # via
-    #   typer
    #   uvicorn
    #   wandb
-cloudpickle==3.1.2
-    # via gymnasium
-cmake==4.1.3
+cloudpickle==3.1.1
+    # via
+    #   gymnasium
+    #   libero
+cmake==4.1.0
    # via lerobot
-cmeel==0.59.0
+cmeel==0.57.3
    # via
    #   cmeel-assimp
    #   cmeel-boost
@@ -105,17 +108,15 @@ cmeel-zlib==1.3.1
    # via cmeel-assimp
 coal-library==3.0.1
    # via pin
-contourpy==1.3.3
-    # via
-    #   lerobot
-    #   matplotlib
-coverage[toml]==7.13.4
+contourpy==1.3.2
+    # via matplotlib
+coverage[toml]==7.11.0
    # via pytest-cov
 cycler==0.12.1
    # via matplotlib
-datasets==4.6.1
+datasets==4.1.1
    # via lerobot
-debugpy==1.8.20
+debugpy==1.8.17
    # via lerobot
 decorator==5.2.1
    # via ipython
@@ -129,7 +130,7 @@ dill==0.4.0
    #   multiprocess
 distlib==0.4.0
    # via virtualenv
-dm-control==1.0.37
+dm-control==1.0.34
    # via gym-aloha
 dm-env==1.6
    # via dm-control
@@ -137,55 +138,69 @@ dm-tree==0.1.9
    # via
    #   dm-control
    #   dm-env
+    #   lerobot
 docopt==0.6.2
    # via num2words
 draccus==0.10.0
    # via lerobot
 dynamixel-sdk==3.8.4
    # via lerobot
+easydict==1.13
+    # via libero
+egl-probe @ git+https://github.com/huggingface/egl_probe.git
+    # via
+    #   libero
+    #   robomimic
 eigenpy==3.10.3
    # via coal-library
-einops==0.8.2
-    # via lerobot
-eiquadprog==1.2.9
-    # via placo
-etils[epath,epy]==1.14.0
-    # via mujoco
-executing==2.2.1
-    # via stack-data
-faker==34.0.2
-    # via lerobot
-farama-notifications==0.0.4
-    # via gymnasium
-fastapi==0.135.1
+einops==0.8.1
    # via
    #   lerobot
-    #   teleop
+    #   libero
+eiquadprog==1.2.9
+    # via placo
+etils[epath,epy]==1.13.0
+    # via mujoco
+exceptiongroup==1.3.0
+    # via
+    #   anyio
+    #   ipython
+    #   pytest
+executing==2.2.1
+    # via stack-data
+farama-notifications==0.0.4
+    # via gymnasium
+fastapi==0.119.1
+    # via teleop
+fastjsonschema==2.21.2
+    # via nbformat
 feetech-servo-sdk==1.0.0
    # via lerobot
-filelock==3.25.0
+filelock==3.20.0
    # via
    #   datasets
    #   diffusers
    #   huggingface-hub
-    #   python-discovery
    #   torch
+    #   transformers
    #   virtualenv
-fonttools==4.61.1
+fonttools==4.60.1
    # via matplotlib
 frozenlist==1.8.0
    # via
    #   aiohttp
    #   aiosignal
-fsspec[http]==2026.2.0
+fsspec[http]==2025.9.0
    # via
    #   datasets
    #   etils
    #   huggingface-hub
    #   torch
+future==1.0.0
+    # via libero
 gitdb==4.0.12
    # via gitpython
-gitpython==3.1.46
+gitpython==3.1.45
    # via wandb
 glfw==2.10.0
    # via
@@ -197,6 +212,7 @@ grpcio==1.73.1
    #   lerobot
    #   reachy2-sdk
    #   reachy2-sdk-api
+    #   tensorboard
 grpcio-tools==1.73.1
    # via
    #   lerobot
@@ -207,67 +223,71 @@ gym-hil==0.1.13
    # via lerobot
 gym-pusht==0.1.6
    # via lerobot
-gymnasium==1.2.3
+gymnasium==1.2.1
    # via
    #   gym-aloha
    #   gym-hil
    #   gym-pusht
    #   lerobot
+    #   libero
    #   metaworld
 h11==0.16.0
-    # via
-    #   httpcore
-    #   uvicorn
+    # via uvicorn
+h5py==3.15.1
+    # via robomimic
 hebi-py==2.11.0
    # via lerobot
-hf-xet==1.3.2
+hf-transfer==0.1.9
+    # via huggingface-hub
+hf-xet==1.1.10
    # via huggingface-hub
 hidapi==0.14.0.post4
    # via
    #   gym-hil
    #   lerobot
-httpcore==1.0.9
-    # via httpx
 httptools==0.7.1
    # via uvicorn
-httpx==0.28.1
-    # via
-    #   datasets
-    #   huggingface-hub
-huggingface-hub==1.6.0
+huggingface-hub[cli,hf-transfer]==0.35.3
    # via
    #   accelerate
    #   datasets
    #   diffusers
    #   lerobot
    #   peft
+    #   timm
    #   tokenizers
    #   transformers
-identify==2.6.17
+hydra-core==1.3.2
+    # via libero
+identify==2.6.15
    # via pre-commit
 idna==3.11
    # via
    #   anyio
-    #   httpx
    #   requests
    #   yarl
-imageio[ffmpeg]==2.37.2
+imageio[ffmpeg]==2.37.0
    # via
    #   gym-aloha
    #   gym-hil
    #   lerobot
    #   metaworld
+    #   robomimic
    #   scikit-image
 imageio-ffmpeg==0.6.0
-    # via imageio
-importlib-metadata==8.7.1
+    # via
+    #   imageio
+    #   robomimic
+importlib-metadata==8.7.0
    # via diffusers
+importlib-resources==6.5.2
+    # via etils
 iniconfig==2.3.0
    # via pytest
-ipython==9.11.0
+inquirerpy==0.3.4
+    # via huggingface-hub
+ipython==8.37.0
    # via meshcat
-ipython-pygments-lexers==1.1.1
-    # via ipython
 ischedule==1.2.7
    # via placo
 jedi==0.19.2
@@ -276,24 +296,44 @@ jinja2==3.1.6
    # via torch
 jsonlines==4.0.0
    # via lerobot
+jsonschema==4.25.1
+    # via nbformat
+jsonschema-specifications==2025.9.1
+    # via jsonschema
+jupyter-core==5.9.1
+    # via nbformat
+jupytext==1.18.1
+    # via bddl
 kiwisolver==1.4.9
    # via matplotlib
 labmaze==1.0.6
    # via dm-control
-lazy-loader==0.5
+lazy-loader==0.4
    # via scikit-image
-librt==0.8.1
-    # via mypy
+libero @ git+https://github.com/huggingface/lerobot-libero.git@main
+    # via lerobot
+llvmlite==0.45.1
+    # via numba
 lxml==6.0.2
    # via dm-control
+markdown==3.9
+    # via tensorboard
 markdown-it-py==4.0.0
-    # via rich
+    # via
+    #   jupytext
+    #   mdit-py-plugins
 markupsafe==3.0.3
-    # via jinja2
-matplotlib==3.10.8
-    # via lerobot
+    # via
+    #   jinja2
+    #   werkzeug
+matplotlib==3.10.7
+    # via
+    #   lerobot
+    #   libero
 matplotlib-inline==0.2.1
    # via ipython
+mdit-py-plugins==0.5.0
+    # via jupytext
 mdurl==0.1.2
    # via markdown-it-py
 mergedeep==1.3.4
@@ -306,35 +346,41 @@ mock-serial==0.0.1
    # via lerobot
 mpmath==1.3.0
    # via sympy
-mujoco==3.5.0
+mujoco==3.3.7
    # via
    #   dm-control
    #   gym-aloha
    #   gym-hil
+    #   libero
    #   metaworld
-multidict==6.7.1
+    #   robosuite
+multidict==6.7.0
    # via
    #   aiohttp
    #   yarl
-multiprocess==0.70.18
+multiprocess==0.70.16
    # via datasets
-mypy==1.19.1
-    # via lerobot
 mypy-extensions==1.1.0
+    # via typing-inspect
+nbformat==5.10.4
+    # via jupytext
+networkx==3.4.2
    # via
-    #   mypy
-    #   typing-inspect
-networkx==3.6.1
-    # via
+    #   bddl
    #   scikit-image
    #   torch
-nodeenv==1.10.0
+ninja==1.13.0
+    # via lerobot
+nodeenv==1.9.1
    # via pre-commit
 num2words==0.5.14
    # via lerobot
+numba==0.62.1
+    # via robosuite
 numpy==2.2.6
    # via
    #   accelerate
+    #   bddl
    #   cmeel-boost
    #   contourpy
    #   datasets
@@ -343,14 +389,16 @@ numpy==2.2.6
    #   dm-env
    #   dm-tree
    #   gymnasium
+    #   h5py
    #   hebi-py
    #   imageio
    #   labmaze
-    #   lerobot
+    #   libero
    #   matplotlib
    #   meshcat
    #   metaworld
    #   mujoco
+    #   numba
    #   opencv-python
    #   opencv-python-headless
    #   pandas
@@ -358,18 +406,26 @@ numpy==2.2.6
    #   pyquaternion
    #   reachy2-sdk
    #   rerun-sdk
+    #   robomimic
+    #   robosuite
    #   scikit-image
    #   scipy
    #   shapely
    #   teleop
+    #   tensorboard
+    #   tensorboardx
    #   tifffile
    #   torchvision
    #   transformers
    #   transforms3d
-opencv-python==4.13.0.92
+omegaconf==2.3.0
+    # via hydra-core
+opencv-python==4.12.0.88
    # via
    #   gym-pusht
+    #   libero
    #   reachy2-sdk
+    #   robosuite
 opencv-python-headless==4.12.0.88
    # via lerobot
 orderly-set==5.5.0
@@ -379,87 +435,97 @@ packaging==25.0
    #   accelerate
    #   datasets
    #   huggingface-hub
+    #   hydra-core
+    #   jupytext
    #   lazy-loader
    #   lerobot
    #   matplotlib
    #   peft
    #   pytest
-    #   qwen-vl-utils
    #   reachy2-sdk
    #   scikit-image
+    #   tensorboard
+    #   tensorboardx
    #   transformers
    #   wandb
 pandas==2.3.3
    # via
    #   datasets
    #   lerobot
-parso==0.8.6
+parso==0.8.5
    # via jedi
-pathspec==1.0.4
-    # via mypy
-peft==0.18.1
+peft==0.17.1
    # via lerobot
 pexpect==4.9.0
    # via ipython
-pillow==12.1.1
+pfzy==0.3.4
+    # via inquirerpy
+pillow==12.0.0
    # via
    #   diffusers
    #   imageio
+    #   lerobot
    #   matplotlib
    #   meshcat
-    #   qwen-vl-utils
    #   rerun-sdk
+    #   robosuite
    #   scikit-image
+    #   tensorboard
    #   torchvision
 pin==3.4.0
    # via placo
-placo==0.9.16
+placo==0.9.14
    # via lerobot
-platformdirs==4.9.4
+platformdirs==4.5.0
    # via
-    #   python-discovery
+    #   jupyter-core
    #   virtualenv
    #   wandb
 pluggy==1.6.0
    # via
    #   pytest
    #   pytest-cov
-pre-commit==4.5.1
+pre-commit==4.3.0
    # via lerobot
 prompt-toolkit==3.0.52
-    # via ipython
+    # via
+    #   inquirerpy
+    #   ipython
 propcache==0.4.1
    # via
    #   aiohttp
    #   yarl
-protobuf==6.31.1
+protobuf==6.31.0
    # via
    #   dm-control
    #   grpcio-tools
    #   lerobot
    #   reachy2-sdk
    #   reachy2-sdk-api
+    #   tensorboard
+    #   tensorboardx
    #   wandb
-psutil==7.2.2
+psutil==7.1.1
    # via
    #   accelerate
    #   imageio
    #   peft
+    #   robomimic
 ptyprocess==0.7.0
    # via pexpect
 pure-eval==0.2.3
    # via stack-data
-pyarrow==23.0.1
+pyarrow==21.0.0
    # via
    #   datasets
    #   rerun-sdk
-pycparser==3.0
+pycparser==2.23
    # via cffi
-pydantic==2.12.5
+pydantic==2.12.3
    # via
    #   fastapi
    #   wandb
-pydantic-core==2.41.5
+pydantic-core==2.41.4
    # via pydantic
 pygame==2.6.1
    # via
@@ -469,35 +535,33 @@ pygame==2.6.1
 pygments==2.19.2
    # via
    #   ipython
-    #   ipython-pygments-lexers
    #   pytest
-    #   rich
 pymunk==6.11.1
    # via
    #   gym-pusht
    #   lerobot
-pyngrok==7.5.1
+pyngrok==7.4.1
    # via meshcat
 pynput==1.8.1
    # via
    #   gym-hil
    #   lerobot
-pyobjc-core==12.1
+pyobjc-core==12.0
    # via
    #   pyobjc-framework-applicationservices
    #   pyobjc-framework-cocoa
    #   pyobjc-framework-coretext
    #   pyobjc-framework-quartz
-pyobjc-framework-applicationservices==12.1
+pyobjc-framework-applicationservices==12.0
    # via pynput
-pyobjc-framework-cocoa==12.1
+pyobjc-framework-cocoa==12.0
    # via
    #   pyobjc-framework-applicationservices
    #   pyobjc-framework-coretext
    #   pyobjc-framework-quartz
-pyobjc-framework-coretext==12.1
+pyobjc-framework-coretext==12.0
    # via pyobjc-framework-applicationservices
-pyobjc-framework-quartz==12.1
+pyobjc-framework-quartz==12.0
    # via
    #   pynput
    #   pyobjc-framework-applicationservices
@@ -506,13 +570,13 @@ pyopengl==3.1.10
    # via
    #   dm-control
    #   mujoco
-pyparsing==3.3.2
+pyparsing==3.2.5
    # via
    #   dm-control
    #   matplotlib
 pyquaternion==0.9.9
    # via reachy2-sdk
-pyrealsense2-macosx==2.56.5
+pyrealsense2-macosx==2.54.2
    # via lerobot
 pyserial==3.5
    # via
@@ -521,6 +585,7 @@ pyserial==3.5
    #   lerobot
 pytest==8.4.2
    # via
+    #   bddl
    #   lerobot
    #   pytest-cov
    #   pytest-timeout
@@ -531,14 +596,11 @@ pytest-timeout==2.4.0
    # via lerobot
 python-dateutil==2.9.0.post0
    # via
-    #   faker
    #   matplotlib
    #   pandas
-python-discovery==1.1.1
-    # via virtualenv
-python-dotenv==1.2.2
+python-dotenv==1.1.1
    # via uvicorn
-pytz==2026.1.post1
+pytz==2025.2
    # via pandas
 pyyaml==6.0.3
    # via
@@ -547,10 +609,13 @@ pyyaml==6.0.3
    #   draccus
    #   hebi-py
    #   huggingface-hub
+    #   jupytext
+    #   omegaconf
    #   peft
    #   pre-commit
    #   pyngrok
    #   pyyaml-include
+    #   timm
    #   transformers
    #   uvicorn
    #   wandb
@@ -560,13 +625,15 @@ pyzmq==27.1.0
    # via
    #   lerobot
    #   meshcat
-qwen-vl-utils==0.0.14
-    # via lerobot
-reachy2-sdk==1.0.15
+reachy2-sdk==1.0.14
    # via lerobot
 reachy2-sdk-api==1.0.21
    # via reachy2-sdk
-regex==2026.2.28
+referencing==0.37.0
+    # via
+    #   jsonschema
+    #   jsonschema-specifications
+regex==2025.10.23
    # via
    #   diffusers
    #   transformers
@@ -575,150 +642,184 @@ requests==2.32.5
    #   datasets
    #   diffusers
    #   dm-control
-    #   qwen-vl-utils
+    #   huggingface-hub
    #   teleop
+    #   transformers
    #   wandb
-rerun-sdk==0.26.2
+rerun-sdk==0.26.1
    # via lerobot
 rhoban-cmeel-jsoncpp==1.9.4.9
    # via placo
-rich==14.3.3
-    # via typer
-safetensors==0.7.0
+robomimic==0.2.0
+    # via libero
+robosuite==1.4.0
+    # via libero
+rpds-py==0.28.0
+    # via
+    #   jsonschema
+    #   referencing
+safetensors==0.6.2
    # via
    #   accelerate
    #   diffusers
    #   lerobot
    #   peft
+    #   timm
    #   transformers
 scikit-image==0.25.2
    # via
    #   gym-pusht
    #   lerobot
-scipy==1.17.1
+scipy==1.15.3
    # via
    #   dm-control
-    #   lerobot
    #   metaworld
+    #   robosuite
    #   scikit-image
-    #   torchdiffeq
-sentry-sdk==2.54.0
+sentry-sdk==2.42.1
    # via wandb
 shapely==2.1.2
    # via gym-pusht
-shellingham==1.5.4
-    # via typer
 six==1.17.0
    # via
    #   pynput
    #   python-dateutil
-smmap==5.0.3
+smmap==5.0.2
    # via gitdb
+sniffio==1.3.1
+    # via anyio
 stack-data==0.6.3
    # via ipython
-starlette==0.52.1
+starlette==0.48.0
    # via fastapi
 sympy==1.14.0
    # via torch
-teleop==0.1.4
+teleop==0.1.2
    # via lerobot
-termcolor==3.3.0
-    # via lerobot
-tifffile==2026.3.3
+tensorboard==2.20.0
+    # via robomimic
+tensorboard-data-server==0.7.2
+    # via tensorboard
+tensorboardx==2.6.4
+    # via robomimic
+termcolor==3.1.0
+    # via
+    #   lerobot
+    #   robomimic
+thop==0.1.1.post2209072238
+    # via libero
+tifffile==2025.5.10
    # via scikit-image
-tokenizers==0.22.2
+timm==1.0.20
+    # via lerobot
+tokenizers==0.22.1
    # via transformers
 toml==0.10.2
    # via draccus
-torch==2.10.0
+tomli==2.3.0
+    # via
+    #   cmeel
+    #   coverage
+    #   jupytext
+    #   pytest
+torch==2.7.1
    # via
    #   accelerate
    #   lerobot
    #   peft
-    #   torchdiffeq
+    #   robomimic
+    #   thop
+    #   timm
    #   torchvision
-torchcodec==0.10.0
+torchcodec==0.5
    # via lerobot
-torchdiffeq==0.2.5
-    # via lerobot
-torchvision==0.25.0
-    # via lerobot
-tornado==6.5.4
+torchvision==0.22.1
+    # via
+    #   lerobot
+    #   robomimic
+    #   timm
+tornado==6.5.2
    # via meshcat
-tqdm==4.67.3
+tqdm==4.67.1
    # via
    #   datasets
    #   dm-control
    #   huggingface-hub
    #   peft
+    #   robomimic
    #   transformers
 traitlets==5.14.3
    # via
    #   ipython
+    #   jupyter-core
    #   matplotlib-inline
-transformers==5.3.0
+    #   nbformat
+transformers==4.57.1
    # via
    #   lerobot
+    #   libero
    #   peft
 transforms3d==0.4.2
    # via teleop
-typer==0.24.1
-    # via
-    #   huggingface-hub
-    #   transformers
 typing-extensions==4.15.0
    # via
    #   aiosignal
    #   anyio
    #   etils
-    #   faker
+    #   exceptiongroup
    #   fastapi
    #   gymnasium
    #   huggingface-hub
-    #   mypy
+    #   ipython
+    #   multidict
    #   pydantic
    #   pydantic-core
+    #   referencing
    #   rerun-sdk
    #   starlette
    #   torch
    #   typing-inspect
    #   typing-inspection
+    #   uvicorn
+    #   virtualenv
    #   wandb
 typing-inspect==0.9.0
    # via draccus
 typing-inspection==0.4.2
-    # via
-    #   fastapi
-    #   pydantic
-tzdata==2025.3
+    # via pydantic
+tzdata==2025.2
    # via pandas
 u-msgpack-python==2.8.0
    # via meshcat
-urllib3==2.6.3
+urllib3==2.5.0
    # via
    #   requests
    #   sentry-sdk
-uvicorn[standard]==0.41.0
+uvicorn[standard]==0.38.0
    # via teleop
 uvloop==0.22.1
    # via uvicorn
-virtualenv==21.1.0
+virtualenv==20.35.3
    # via pre-commit
-wandb==0.24.2
-    # via lerobot
+wandb==0.21.4
+    # via
+    #   lerobot
+    #   libero
 watchfiles==1.1.1
    # via uvicorn
-wcwidth==0.6.0
+wcwidth==0.2.14
    # via prompt-toolkit
 websocket-client==1.9.0
    # via teleop
-websockets==16.0
+websockets==15.0.1
    # via uvicorn
-wrapt==2.1.2
+werkzeug==3.1.3
+    # via tensorboard
+wrapt==2.0.0
    # via dm-tree
 xxhash==3.6.0
    # via datasets
-yarl==1.23.0
+yarl==1.22.0
    # via aiohttp
 zipp==3.23.0
    # via
@@ -1,12 +1,12 @@
 #
-# This file is autogenerated by pip-compile with Python 3.12
+# This file is autogenerated by pip-compile with Python 3.10
 # by the following command:
 #
 #    pip-compile --output-file=requirements-ubuntu.txt requirements.in
 #
 -e .[all]
    # via -[all]
-absl-py==2.4.0
+absl-py==2.3.1
    # via
    #   dm-control
    #   dm-env
@@ -14,33 +14,30 @@ absl-py==2.4.0
    #   labmaze
    #   mujoco
    #   tensorboard
-accelerate==1.13.0
+accelerate==1.11.0
    # via
    #   lerobot
    #   peft
 aiohappyeyeballs==2.6.1
    # via aiohttp
-aiohttp==3.13.3
+aiohttp==3.13.1
    # via fsspec
 aiosignal==1.4.0
    # via aiohttp
-annotated-doc==0.0.4
-    # via
-    #   fastapi
-    #   typer
 annotated-types==0.7.0
    # via pydantic
 antlr4-python3-runtime==4.9.3
    # via
    #   hydra-core
    #   omegaconf
-anyio==4.12.1
+anyio==4.11.0
    # via
-    #   httpx
    #   starlette
    #   watchfiles
-asttokens==3.0.1
+asttokens==3.0.0
    # via stack-data
+async-timeout==5.0.1
+    # via aiohttp
 attrs==25.4.0
    # via
    #   aiohttp
@@ -50,35 +47,30 @@ attrs==25.4.0
    #   referencing
    #   rerun-sdk
 av==15.1.0
-    # via
-    #   lerobot
-    #   qwen-vl-utils
+    # via lerobot
 bddl==1.0.1
-    # via hf-libero
-certifi==2026.2.25
+    # via libero
+certifi==2025.10.5
    # via
-    #   httpcore
-    #   httpx
    #   requests
    #   sentry-sdk
 cffi==2.0.0
    # via pymunk
-cfgv==3.5.0
+cfgv==3.4.0
    # via pre-commit
-charset-normalizer==3.4.5
+charset-normalizer==3.4.4
    # via requests
-click==8.3.1
+click==8.3.0
    # via
-    #   typer
    #   uvicorn
    #   wandb
-cloudpickle==3.1.2
+cloudpickle==3.1.1
    # via
    #   gymnasium
-    #   hf-libero
-cmake==4.1.3
+    #   libero
+cmake==4.1.0
    # via lerobot
-cmeel==0.59.0
+cmeel==0.57.3
    # via
    #   cmeel-assimp
    #   cmeel-boost
@@ -116,24 +108,20 @@ cmeel-zlib==1.3.1
    # via cmeel-assimp
 coal-library==3.0.1
    # via pin
-contourpy==1.3.3
-    # via
-    #   lerobot
-    #   matplotlib
-coverage[toml]==7.13.4
+contourpy==1.3.2
+    # via matplotlib
+coverage[toml]==7.11.0
    # via pytest-cov
-cuda-bindings==12.9.4
-    # via torch
-cuda-pathfinder==1.4.1
-    # via cuda-bindings
 cycler==0.12.1
    # via matplotlib
-datasets==4.6.1
+datasets==4.1.1
    # via lerobot
-debugpy==1.8.20
+debugpy==1.8.17
    # via lerobot
 decorator==5.2.1
    # via ipython
+decord==0.6.0
+    # via lerobot
 deepdiff==8.6.1
    # via lerobot
 diffusers==0.35.2
@@ -144,7 +132,7 @@ dill==0.4.0
    #   multiprocess
 distlib==0.4.0
    # via virtualenv
-dm-control==1.0.37
+dm-control==1.0.34
    # via gym-aloha
 dm-env==1.6
    # via dm-control
@@ -152,6 +140,7 @@ dm-tree==0.1.9
    # via
    #   dm-control
    #   dm-env
+    #   lerobot
 docopt==0.6.2
    # via num2words
 draccus==0.10.0
@@ -159,60 +148,66 @@ draccus==0.10.0
 dynamixel-sdk==3.8.4
    # via lerobot
 easydict==1.13
-    # via hf-libero
-egl-probe==1.0.2
-    # via robomimic
+    # via libero
+egl-probe @ git+https://github.com/huggingface/egl_probe.git
+    # via
+    #   libero
+    #   robomimic
 eigenpy==3.10.3
    # via coal-library
-einops==0.8.2
+einops==0.8.1
    # via
-    #   hf-libero
+    #   flash-attn
    #   lerobot
+    #   libero
 eiquadprog==1.2.9
    # via placo
-etils[epath,epy]==1.14.0
+etils[epath,epy]==1.13.0
    # via mujoco
-evdev==1.9.3
+evdev==1.9.2
    # via pynput
+exceptiongroup==1.3.0
+    # via
+    #   anyio
+    #   ipython
+    #   pytest
 executing==2.2.1
    # via stack-data
-faker==34.0.2
-    # via lerobot
 farama-notifications==0.0.4
    # via gymnasium
-fastapi==0.135.1
-    # via
-    #   lerobot
-    #   teleop
+fastapi==0.119.1
+    # via teleop
 fastjsonschema==2.21.2
    # via nbformat
 feetech-servo-sdk==1.0.0
    # via lerobot
-filelock==3.25.0
+filelock==3.20.0
    # via
    #   datasets
    #   diffusers
    #   huggingface-hub
-    #   python-discovery
    #   torch
+    #   transformers
    #   virtualenv
-fonttools==4.61.1
+flash-attn==2.8.3
+    # via lerobot
+fonttools==4.60.1
    # via matplotlib
 frozenlist==1.8.0
    # via
    #   aiohttp
    #   aiosignal
-fsspec[http]==2026.2.0
+fsspec[http]==2025.9.0
    # via
    #   datasets
    #   etils
    #   huggingface-hub
    #   torch
 future==1.0.0
-    # via hf-libero
+    # via libero
 gitdb==4.0.12
    # via gitpython
-gitpython==3.1.46
+gitpython==3.1.45
    # via wandb
 glfw==2.10.0
    # via
@@ -235,60 +230,50 @@ gym-hil==0.1.13
    # via lerobot
 gym-pusht==0.1.6
    # via lerobot
-gymnasium==1.2.3
+gymnasium==1.2.1
    # via
    #   gym-aloha
    #   gym-hil
    #   gym-pusht
-    #   hf-libero
    #   lerobot
+    #   libero
    #   metaworld
 h11==0.16.0
-    # via
-    #   httpcore
-    #   uvicorn
-h5py==3.16.0
+    # via uvicorn
+h5py==3.15.1
    # via robomimic
 hebi-py==2.11.0
    # via lerobot
-hf-egl-probe==1.0.2
-    # via hf-libero
-hf-libero==0.1.3
-    # via lerobot
-hf-xet==1.3.2
+hf-transfer==0.1.9
+    # via huggingface-hub
+hf-xet==1.1.10
    # via huggingface-hub
 hidapi==0.14.0.post4
    # via
    #   gym-hil
    #   lerobot
-httpcore==1.0.9
-    # via httpx
 httptools==0.7.1
    # via uvicorn
-httpx==0.28.1
-    # via
-    #   datasets
-    #   huggingface-hub
-huggingface-hub==1.6.0
+huggingface-hub[cli,hf-transfer]==0.35.3
    # via
    #   accelerate
    #   datasets
    #   diffusers
    #   lerobot
    #   peft
+    #   timm
    #   tokenizers
    #   transformers
 hydra-core==1.3.2
-    # via hf-libero
-identify==2.6.17
+    # via libero
+identify==2.6.15
    # via pre-commit
 idna==3.11
    # via
    #   anyio
-    #   httpx
    #   requests
    #   yarl
-imageio[ffmpeg]==2.37.2
+imageio[ffmpeg]==2.37.0
    # via
    #   gym-aloha
    #   gym-hil
@@ -300,14 +285,16 @@ imageio-ffmpeg==0.6.0
    # via
    #   imageio
    #   robomimic
-importlib-metadata==8.7.1
+importlib-metadata==8.7.0
    # via diffusers
+importlib-resources==6.5.2
+    # via etils
 iniconfig==2.3.0
    # via pytest
-ipython==9.11.0
+inquirerpy==0.3.4
+    # via huggingface-hub
+ipython==8.37.0
    # via meshcat
-ipython-pygments-lexers==1.1.1
-    # via ipython
 ischedule==1.2.7
    # via placo
 jedi==0.19.2
@@ -316,41 +303,40 @@ jinja2==3.1.6
    # via torch
 jsonlines==4.0.0
    # via lerobot
-jsonschema==4.26.0
+jsonschema==4.25.1
    # via nbformat
 jsonschema-specifications==2025.9.1
    # via jsonschema
 jupyter-core==5.9.1
    # via nbformat
-jupytext==1.19.1
+jupytext==1.18.1
    # via bddl
 kiwisolver==1.4.9
    # via matplotlib
 labmaze==1.0.6
    # via dm-control
-lazy-loader==0.5
+lazy-loader==0.4
    # via scikit-image
-librt==0.8.1
-    # via mypy
-llvmlite==0.46.0
+libero @ git+https://github.com/huggingface/lerobot-libero.git@main
+    # via lerobot
+llvmlite==0.45.1
    # via numba
 lxml==6.0.2
    # via dm-control
-markdown==3.10.2
+markdown==3.9
    # via tensorboard
 markdown-it-py==4.0.0
    # via
    #   jupytext
    #   mdit-py-plugins
-    #   rich
 markupsafe==3.0.3
    # via
    #   jinja2
    #   werkzeug
-matplotlib==3.10.8
+matplotlib==3.10.7
    # via
-    #   hf-libero
    #   lerobot
+    #   libero
 matplotlib-inline==0.2.1
    # via ipython
 mdit-py-plugins==0.5.0
@@ -367,38 +353,36 @@ mock-serial==0.0.1
    # via lerobot
 mpmath==1.3.0
    # via sympy
-mujoco==3.5.0
+mujoco==3.3.7
    # via
    #   dm-control
    #   gym-aloha
    #   gym-hil
-    #   hf-libero
+    #   libero
    #   metaworld
    #   robosuite
-multidict==6.7.1
+multidict==6.7.0
    # via
    #   aiohttp
    #   yarl
-multiprocess==0.70.18
+multiprocess==0.70.16
    # via datasets
-mypy==1.19.1
-    # via lerobot
 mypy-extensions==1.1.0
-    # via
-    #   mypy
-    #   typing-inspect
+    # via typing-inspect
 nbformat==5.10.4
    # via jupytext
-networkx==3.6.1
+networkx==3.4.2
    # via
    #   bddl
    #   scikit-image
    #   torch
-nodeenv==1.10.0
+ninja==1.13.0
+    # via lerobot
+nodeenv==1.9.1
    # via pre-commit
 num2words==0.5.14
    # via lerobot
-numba==0.64.0
+numba==0.62.1
    # via robosuite
 numpy==2.2.6
    # via
@@ -407,6 +391,7 @@ numpy==2.2.6
    #   cmeel-boost
    #   contourpy
    #   datasets
+    #   decord
    #   diffusers
    #   dm-control
    #   dm-env
@@ -414,10 +399,9 @@ numpy==2.2.6
    #   gymnasium
    #   h5py
    #   hebi-py
-    #   hf-libero
    #   imageio
    #   labmaze
-    #   lerobot
+    #   libero
    #   matplotlib
    #   meshcat
    #   metaworld
@@ -442,51 +426,49 @@ numpy==2.2.6
    #   torchvision
    #   transformers
    #   transforms3d
-nvidia-cublas-cu12==12.8.4.1
+nvidia-cublas-cu12==12.6.4.1
    # via
    #   nvidia-cudnn-cu12
    #   nvidia-cusolver-cu12
    #   torch
-nvidia-cuda-cupti-cu12==12.8.90
+nvidia-cuda-cupti-cu12==12.6.80
    # via torch
-nvidia-cuda-nvrtc-cu12==12.8.93
+nvidia-cuda-nvrtc-cu12==12.6.77
    # via torch
-nvidia-cuda-runtime-cu12==12.8.90
+nvidia-cuda-runtime-cu12==12.6.77
    # via torch
-nvidia-cudnn-cu12==9.10.2.21
+nvidia-cudnn-cu12==9.5.1.17
    # via torch
-nvidia-cufft-cu12==11.3.3.83
+nvidia-cufft-cu12==11.3.0.4
    # via torch
-nvidia-cufile-cu12==1.13.1.3
+nvidia-cufile-cu12==1.11.1.6
    # via torch
-nvidia-curand-cu12==10.3.9.90
+nvidia-curand-cu12==10.3.7.77
    # via torch
-nvidia-cusolver-cu12==11.7.3.90
+nvidia-cusolver-cu12==11.7.1.2
    # via torch
-nvidia-cusparse-cu12==12.5.8.93
+nvidia-cusparse-cu12==12.5.4.2
    # via
    #   nvidia-cusolver-cu12
    #   torch
-nvidia-cusparselt-cu12==0.7.1
+nvidia-cusparselt-cu12==0.6.3
    # via torch
-nvidia-nccl-cu12==2.27.5
+nvidia-nccl-cu12==2.26.2
    # via torch
-nvidia-nvjitlink-cu12==12.8.93
+nvidia-nvjitlink-cu12==12.6.85
    # via
    #   nvidia-cufft-cu12
    #   nvidia-cusolver-cu12
    #   nvidia-cusparse-cu12
    #   torch
-nvidia-nvshmem-cu12==3.4.5
-    # via torch
-nvidia-nvtx-cu12==12.8.90
+nvidia-nvtx-cu12==12.6.77
    # via torch
 omegaconf==2.3.0
    # via hydra-core
-opencv-python==4.13.0.92
+opencv-python==4.12.0.88
    # via
    #   gym-pusht
-    #   hf-libero
+    #   libero
    #   reachy2-sdk
    #   robosuite
 opencv-python-headless==4.12.0.88
@@ -505,7 +487,6 @@ packaging==25.0
    #   matplotlib
    #   peft
    #   pytest
-    #   qwen-vl-utils
    #   reachy2-sdk
    #   scikit-image
    #   tensorboard
@@ -516,21 +497,21 @@ pandas==2.3.3
    # via
    #   datasets
    #   lerobot
-parso==0.8.6
+parso==0.8.5
    # via jedi
-pathspec==1.0.4
-    # via mypy
-peft==0.18.1
+peft==0.17.1
    # via lerobot
 pexpect==4.9.0
    # via ipython
-pillow==12.1.1
+pfzy==0.3.4
+    # via inquirerpy
+pillow==12.0.0
    # via
    #   diffusers
    #   imageio
+    #   lerobot
    #   matplotlib
    #   meshcat
-    #   qwen-vl-utils
    #   rerun-sdk
    #   robosuite
    #   scikit-image
@@ -538,27 +519,28 @@ pillow==12.1.1
    #   torchvision
 pin==3.4.0
    # via placo
-placo==0.9.16
+placo==0.9.14
    # via lerobot
-platformdirs==4.9.4
+platformdirs==4.5.0
    # via
    #   jupyter-core
-    #   python-discovery
    #   virtualenv
    #   wandb
 pluggy==1.6.0
    # via
    #   pytest
    #   pytest-cov
-pre-commit==4.5.1
+pre-commit==4.3.0
    # via lerobot
 prompt-toolkit==3.0.52
-    # via ipython
+    # via
+    #   inquirerpy
+    #   ipython
 propcache==0.4.1
    # via
    #   aiohttp
    #   yarl
-protobuf==6.31.1
+protobuf==6.31.0
    # via
    #   dm-control
    #   grpcio-tools
@@ -568,7 +550,7 @@ protobuf==6.31.1
    #   tensorboard
    #   tensorboardx
    #   wandb
-psutil==7.2.2
+psutil==7.1.1
    # via
    #   accelerate
    #   imageio
@@ -578,17 +560,17 @@ ptyprocess==0.7.0
    # via pexpect
 pure-eval==0.2.3
    # via stack-data
-pyarrow==23.0.1
+pyarrow==21.0.0
    # via
    #   datasets
    #   rerun-sdk
-pycparser==3.0
+pycparser==2.23
    # via cffi
-pydantic==2.12.5
+pydantic==2.12.3
    # via
    #   fastapi
    #   wandb
-pydantic-core==2.41.5
+pydantic-core==2.41.4
    # via pydantic
 pygame==2.6.1
    # via
@@ -598,14 +580,12 @@ pygame==2.6.1
 pygments==2.19.2
    # via
    #   ipython
-    #   ipython-pygments-lexers
    #   pytest
-    #   rich
 pymunk==6.11.1
    # via
    #   gym-pusht
    #   lerobot
-pyngrok==7.5.1
+pyngrok==7.4.1
    # via meshcat
 pynput==1.8.1
    # via
@@ -615,7 +595,7 @@ pyopengl==3.1.10
    # via
    #   dm-control
    #   mujoco
-pyparsing==3.3.2
+pyparsing==3.2.5
    # via
    #   dm-control
    #   matplotlib
@@ -641,16 +621,13 @@ pytest-timeout==2.4.0
    # via lerobot
 python-dateutil==2.9.0.post0
    # via
-    #   faker
    #   matplotlib
    #   pandas
-python-discovery==1.1.1
-    # via virtualenv
-python-dotenv==1.2.2
+python-dotenv==1.1.1
    # via uvicorn
 python-xlib==0.33
    # via pynput
-pytz==2026.1.post1
+pytz==2025.2
    # via pandas
 pyyaml==6.0.3
    # via
@@ -665,6 +642,7 @@ pyyaml==6.0.3
    #   pre-commit
    #   pyngrok
    #   pyyaml-include
+    #   timm
    #   transformers
    #   uvicorn
    #   wandb
@@ -674,9 +652,7 @@ pyzmq==27.1.0
    # via
    #   lerobot
    #   meshcat
-qwen-vl-utils==0.0.14
-    # via lerobot
-reachy2-sdk==1.0.15
+reachy2-sdk==1.0.14
    # via lerobot
 reachy2-sdk-api==1.0.21
    # via reachy2-sdk
@@ -684,7 +660,7 @@ referencing==0.37.0
    # via
    #   jsonschema
    #   jsonschema-specifications
-regex==2026.2.28
+regex==2025.10.23
    # via
    #   diffusers
    #   transformers
@@ -693,62 +669,60 @@ requests==2.32.5
    #   datasets
    #   diffusers
    #   dm-control
-    #   qwen-vl-utils
+    #   huggingface-hub
    #   teleop
+    #   transformers
    #   wandb
-rerun-sdk==0.26.2
+rerun-sdk==0.26.1
    # via lerobot
 rhoban-cmeel-jsoncpp==1.9.4.9
    # via placo
-rich==14.3.3
-    # via typer
 robomimic==0.2.0
-    # via hf-libero
+    # via libero
 robosuite==1.4.0
-    # via hf-libero
-rpds-py==0.30.0
+    # via libero
+rpds-py==0.28.0
    # via
    #   jsonschema
    #   referencing
-safetensors==0.7.0
+safetensors==0.6.2
    # via
    #   accelerate
    #   diffusers
    #   lerobot
    #   peft
+    #   timm
    #   transformers
 scikit-image==0.25.2
    # via
    #   gym-pusht
    #   lerobot
-scipy==1.17.1
+scipy==1.15.3
    # via
    #   dm-control
-    #   lerobot
    #   metaworld
    #   robosuite
    #   scikit-image
-    #   torchdiffeq
-sentry-sdk==2.54.0
+sentry-sdk==2.42.1
    # via wandb
 shapely==2.1.2
    # via gym-pusht
-shellingham==1.5.4
-    # via typer
 six==1.17.0
    # via
    #   pynput
    #   python-dateutil
    #   python-xlib
-smmap==5.0.3
+smmap==5.0.2
    # via gitdb
+sniffio==1.3.1
+    # via anyio
 stack-data==0.6.3
    # via ipython
-starlette==0.52.1
+starlette==0.48.0
    # via fastapi
 sympy==1.14.0
    # via torch
-teleop==0.1.4
+teleop==0.1.2
    # via lerobot
 tensorboard==2.20.0
    # via robomimic
@@ -756,38 +730,46 @@ tensorboard-data-server==0.7.2
    # via tensorboard
 tensorboardx==2.6.4
    # via robomimic
-termcolor==3.3.0
+termcolor==3.1.0
    # via
    #   lerobot
    #   robomimic
 thop==0.1.1.post2209072238
-    # via hf-libero
-tifffile==2026.3.3
+    # via libero
+tifffile==2025.5.10
    # via scikit-image
-tokenizers==0.22.2
+timm==1.0.20
+    # via lerobot
+tokenizers==0.22.1
    # via transformers
 toml==0.10.2
    # via draccus
-torch==2.10.0
+tomli==2.3.0
+    # via
+    #   cmeel
+    #   coverage
+    #   jupytext
+    #   pytest
+torch==2.7.1
    # via
    #   accelerate
+    #   flash-attn
    #   lerobot
    #   peft
    #   robomimic
    #   thop
-    #   torchdiffeq
+    #   timm
    #   torchvision
-torchcodec==0.10.0
+torchcodec==0.5
    # via lerobot
-torchdiffeq==0.2.5
-    # via lerobot
-torchvision==0.25.0
+torchvision==0.22.1
    # via
    #   lerobot
    #   robomimic
-tornado==6.5.4
+    #   timm
+tornado==6.5.2
    # via meshcat
-tqdm==4.67.3
+tqdm==4.67.1
    # via
    #   datasets
    #   dm-control
@@ -801,29 +783,26 @@ traitlets==5.14.3
    #   jupyter-core
    #   matplotlib-inline
    #   nbformat
-transformers==5.3.0
+transformers==4.57.1
    # via
-    #   hf-libero
    #   lerobot
+    #   libero
    #   peft
 transforms3d==0.4.2
    # via teleop
-triton==3.6.0
+triton==3.3.1
    # via torch
-typer==0.24.1
-    # via
-    #   huggingface-hub
-    #   transformers
 typing-extensions==4.15.0
    # via
    #   aiosignal
    #   anyio
    #   etils
-    #   faker
+    #   exceptiongroup
    #   fastapi
    #   gymnasium
    #   huggingface-hub
-    #   mypy
+    #   ipython
+    #   multidict
    #   pydantic
    #   pydantic-core
    #   referencing
@@ -832,46 +811,46 @@ typing-extensions==4.15.0
    #   torch
    #   typing-inspect
    #   typing-inspection
+    #   uvicorn
+    #   virtualenv
    #   wandb
 typing-inspect==0.9.0
    # via draccus
 typing-inspection==0.4.2
-    # via
-    #   fastapi
-    #   pydantic
-tzdata==2025.3
+    # via pydantic
+tzdata==2025.2
    # via pandas
 u-msgpack-python==2.8.0
    # via meshcat
-urllib3==2.6.3
+urllib3==2.5.0
    # via
    #   requests
    #   sentry-sdk
-uvicorn[standard]==0.41.0
+uvicorn[standard]==0.38.0
    # via teleop
 uvloop==0.22.1
    # via uvicorn
-virtualenv==21.1.0
+virtualenv==20.35.3
    # via pre-commit
-wandb==0.24.2
+wandb==0.21.4
    # via
-    #   hf-libero
    #   lerobot
+    #   libero
 watchfiles==1.1.1
    # via uvicorn
-wcwidth==0.6.0
+wcwidth==0.2.14
    # via prompt-toolkit
 websocket-client==1.9.0
    # via teleop
-websockets==16.0
+websockets==15.0.1
    # via uvicorn
-werkzeug==3.1.6
+werkzeug==3.1.3
    # via tensorboard
-wrapt==2.1.2
+wrapt==2.0.0
    # via dm-tree
 xxhash==3.6.0
    # via datasets
-yarl==1.23.0
+yarl==1.22.0
    # via aiohttp
 zipp==3.23.0
    # via
@@ -1,9 +1,9 @@
 # requirements.in

-# requirements-macos.txt was generated on macOS and is platform-specific (macOS 26.3.1 25D2128 arm64).
-# Darwin MacBook-Pro.local 25.3.0 Darwin Kernel Version 25.3.0: Wed Jan 28 20:54:55 PST 2026; root:xnu-12377.91.3~2/RELEASE_ARM64_T8132 arm64
+# requirements-macos.txt was generated on macOS and is platform-specific (macOS 26.0.1 25A362 arm64).
+# Darwin MacBook-Pro.local 25.0.0 Darwin Kernel Version 25.0.0: Wed Sep 17 21:42:08 PDT 2025; root:xnu-12377.1.9~141/RELEASE_ARM64_T8132 arm64

-# requirements-ubuntu.txt was generated on Linux and is platform-specific (Ubuntu 24.04.4 LTS x86_64).
-# Linux lerobot-linux 6.17.0-14-generic #14~24.04.1-Ubuntu SMP PREEMPT_DYNAMIC Thu Jan 15 15:52:10 UTC 2 x86_64 x86_64 x86_64 GNU/Linux
+# requirements-ubuntu.txt was generated on Linux and is platform-specific (Ubuntu 24.04.3 LTS x86_64).
+# Linux mlerobot-linux 6.14.0-33-generic #33~24.04.1-Ubuntu SMP PREEMPT_DYNAMIC Fri Sep 19 17:02:30 UTC 2 x86_64 x86_64 x86_64 GNU/Linux

 -e .[all]
@@ -23,7 +23,7 @@ from typing import Any
 import torch

 from lerobot.configs.types import PolicyFeature
-from lerobot.datasets.feature_utils import build_dataset_frame, hw_to_dataset_features
+from lerobot.datasets.utils import build_dataset_frame, hw_to_dataset_features

 # NOTE: Configs need to be loaded for the client to be able to instantiate the policy config
 from lerobot.policies import (  # noqa: F401
@@ -39,13 +39,15 @@ import grpc
 import torch

 from lerobot.policies.factory import get_policy_class, make_pre_post_processors
-from lerobot.processor import PolicyProcessorPipeline
+from lerobot.processor import (
+    PolicyAction,
+    PolicyProcessorPipeline,
+)
 from lerobot.transport import (
    services_pb2,  # type: ignore
    services_pb2_grpc,  # type: ignore
 )
 from lerobot.transport.utils import receive_bytes_in_chunks
-from lerobot.types import PolicyAction

 from .configs import PolicyServerConfig
 from .constants import SUPPORTED_POLICIES
@@ -63,9 +63,9 @@ from lerobot.transport import (
    services_pb2_grpc,  # type: ignore
 )
 from lerobot.transport.utils import grpc_channel_options, send_bytes_in_chunks
-from lerobot.utils.import_utils import register_third_party_plugins

 from .configs import RobotClientConfig
+from .constants import SUPPORTED_ROBOTS
 from .helpers import (
    Action,
    FPSTracker,
@@ -485,9 +485,8 @@ class RobotClient:
 def async_client(cfg: RobotClientConfig):
    logging.info(pformat(asdict(cfg)))

-    # TODO: Assert if checking robot support is still needed with the plugin system
-    # if cfg.robot.type not in SUPPORTED_ROBOTS:
-    #     raise ValueError(f"Robot {cfg.robot.type} not yet supported!")
+    if cfg.robot.type not in SUPPORTED_ROBOTS:
+        raise ValueError(f"Robot {cfg.robot.type} not yet supported!")

    client = RobotClient(cfg)

@@ -513,5 +512,4 @@ def async_client(cfg: RobotClientConfig):


 if __name__ == "__main__":
-    register_third_party_plugins()
    async_client()  # run the client
@@ -150,7 +150,7 @@ class Camera(abc.ABC):
        """
        pass

-    def read_latest(self, max_age_ms: int = 500) -> NDArray[Any]:
+    def read_latest(self, max_age_ms: int = 1000) -> NDArray[Any]:
        """Return the most recent frame captured immediately (Peeking).

        This method is non-blocking and returns whatever is currently in the
@@ -530,7 +530,7 @@ class OpenCVCamera(Camera):
        return frame

    @check_if_not_connected
-    def read_latest(self, max_age_ms: int = 500) -> NDArray[Any]:
+    def read_latest(self, max_age_ms: int = 1000) -> NDArray[Any]:
        """Return the most recent frame captured immediately (Peeking).

        This method is non-blocking and returns whatever is currently in the
@@ -201,7 +201,7 @@ class Reachy2Camera(Camera):
        return self.read()

    @check_if_not_connected
-    def read_latest(self, max_age_ms: int = 500) -> NDArray[Any]:
+    def read_latest(self, max_age_ms: int = 1000) -> NDArray[Any]:
        """Return the most recent frame captured immediately (Peeking).

        This method is non-blocking and returns whatever is currently in the
@@ -573,7 +573,7 @@ class RealSenseCamera(Camera):

    # NOTE(Steven): Missing implementation for depth for now
    @check_if_not_connected
-    def read_latest(self, max_age_ms: int = 500) -> NDArray[Any]:
+    def read_latest(self, max_age_ms: int = 1000) -> NDArray[Any]:
        """Return the most recent (color) frame captured immediately (Peeking).

        This method is non-blocking and returns whatever is currently in the
@@ -181,7 +181,7 @@ class ZMQCamera(Camera):
        try:
            message = self.socket.recv_string()
        except Exception as e:
-            # zmq is lazy-imported in connect(), so check by name to avoid a top-level import
+            # Check for ZMQ timeout (EAGAIN/Again) without requiring global zmq import
            if type(e).__name__ == "Again":
                raise TimeoutError(f"{self} timeout after {self.timeout_ms}ms") from e
            raise
@@ -23,7 +23,6 @@ import base64
 import contextlib
 import json
 import logging
-import threading
 import time
 from collections import deque

@@ -43,57 +42,10 @@ def encode_image(image: np.ndarray, quality: int = 80) -> str:
    return base64.b64encode(buffer).decode("utf-8")


-class CameraCaptureThread:
-    """Background thread that continuously captures and encodes frames from a camera."""
-
-    def __init__(self, camera: OpenCVCamera, name: str):
-        self.camera = camera
-        self.name = name
-        self.latest_encoded: str | None = None  # Pre-encoded JPEG as base64
-        self.latest_timestamp: float = 0.0
-        self.frame_lock = threading.Lock()
-        self.running = False
-        self.thread: threading.Thread | None = None
-
-    def start(self):
-        """Start the capture thread."""
-        self.running = True
-        self.thread = threading.Thread(target=self._capture_loop, daemon=True)
-        self.thread.start()
-
-    def stop(self):
-        """Stop the capture thread."""
-        self.running = False
-        if self.thread:
-            self.thread.join(timeout=1.0)
-
-    def _capture_loop(self):
-        """Continuously capture and encode frames at the camera's native rate."""
-        while self.running:
-            try:
-                frame = self.camera.read()  # Blocks at camera's native rate
-                timestamp = time.time()
-                # Encode immediately in capture thread (this is the slow part)
-                encoded = encode_image(frame)
-                with self.frame_lock:
-                    self.latest_encoded = encoded
-                    self.latest_timestamp = timestamp
-            except Exception as e:
-                logger.warning(f"Camera {self.name} capture error: {e}")
-                time.sleep(0.01)
-
-    def get_latest(self) -> tuple[str | None, float]:
-        """Get the latest encoded frame and its timestamp."""
-        with self.frame_lock:
-            return self.latest_encoded, self.latest_timestamp
-
-
 class ImageServer:
    def __init__(self, config: dict, port: int = 5555):
-        # fps controls the publish loop rate (how often frames are sent over ZMQ), not the camera capture rate
        self.fps = config.get("fps", 30)
        self.cameras: dict[str, OpenCVCamera] = {}
-        self.capture_threads: dict[str, CameraCaptureThread] = {}

        for name, cfg in config.get("cameras", {}).items():
            shape = cfg.get("shape", [480, 640])
@@ -109,10 +61,6 @@ class ImageServer:
            self.cameras[name] = camera
            logger.info(f"Camera {name}: {shape[1]}x{shape[0]}")

-            # Create capture thread for this camera
-            capture_thread = CameraCaptureThread(camera, name)
-            self.capture_threads[name] = capture_thread
-
        # ZMQ PUB socket
        self.context = zmq.Context()
        self.socket = self.context.socket(zmq.PUB)
@@ -125,18 +73,6 @@ class ImageServer:
    def run(self):
        frame_count = 0
        frame_times = deque(maxlen=60)
-        last_published_ts: dict[str, float] = {}
-
-        # Start all capture threads
-        for capture_thread in self.capture_threads.values():
-            capture_thread.start()
-
-        # Wait for first frames to be captured and encoded
-        logger.info("Waiting for cameras to start capturing...")
-        for name, capture_thread in self.capture_threads.items():
-            while capture_thread.get_latest()[0] is None:
-                time.sleep(0.01)
-            logger.info(f"Camera {name} ready (capture + encode in background)")

        try:
            while True:
@@ -144,12 +80,10 @@ class ImageServer:

                # Build message
                message = {"timestamps": {}, "images": {}}
-                for name, capture_thread in self.capture_threads.items():
-                    encoded, timestamp = capture_thread.get_latest()
-                    if encoded is not None and timestamp > last_published_ts.get(name, 0.0):
-                        message["timestamps"][name] = timestamp
-                        message["images"][name] = encoded
-                        last_published_ts[name] = timestamp
+                for name, cam in self.cameras.items():
+                    frame = cam.read()  # Returns RGB
+                    message["timestamps"][name] = time.time()
+                    message["images"][name] = encode_image(frame)

                # Send as JSON string (suppress if buffer full)
                with contextlib.suppress(zmq.Again):
@@ -168,8 +102,6 @@ class ImageServer:
        except KeyboardInterrupt:
            pass
        finally:
-            for capture_thread in self.capture_threads.values():
-                capture_thread.stop()
            for cam in self.cameras.values():
                cam.disconnect()
            self.socket.close()
@@ -27,7 +27,7 @@ class DatasetConfig:
    # "dataset_index" into the returned item. The index mapping is made according to the order in which the
    # datasets are provided.
    repo_id: str
-    # Root directory where the dataset will be stored (e.g. 'dataset/path'). If None, defaults to $HF_LEROBOT_HOME/repo_id.
+    # Root directory where the dataset will be stored (e.g. 'dataset/path').
    root: str | None = None
    episodes: list[int] | None = None
    image_transforms: ImageTransformsConfig = field(default_factory=ImageTransformsConfig)
@@ -36,16 +36,6 @@ class DatasetConfig:
    video_backend: str = field(default_factory=get_safe_default_codec)
    streaming: bool = False

-    def __post_init__(self) -> None:
-        if self.episodes is not None:
-            if any(ep < 0 for ep in self.episodes):
-                raise ValueError(
-                    f"Episode indices must be non-negative, got: {[ep for ep in self.episodes if ep < 0]}"
-                )
-            if len(self.episodes) != len(set(self.episodes)):
-                duplicates = sorted({ep for ep in self.episodes if self.episodes.count(ep) > 1})
-                raise ValueError(f"Episode indices contain duplicates: {duplicates}")
-

@dataclass
 class WandBConfig:
@@ -57,7 +47,6 @@ class WandBConfig:
    notes: str | None = None
    run_id: str | None = None
    mode: str | None = None  # Allowed values: 'online', 'offline' 'disabled'. Defaults to 'online'
-    add_tags: bool = True  # If True, save configuration as tags in the WandB run.


@dataclass
@@ -30,8 +30,8 @@ from lerobot.configs.types import FeatureType, PolicyFeature
 from lerobot.optim.optimizers import OptimizerConfig
 from lerobot.optim.schedulers import LRSchedulerConfig
 from lerobot.utils.constants import ACTION, OBS_STATE
-from lerobot.utils.device_utils import auto_select_torch_device, is_amp_available, is_torch_device_available
 from lerobot.utils.hub import HubMixin
+from lerobot.utils.utils import auto_select_torch_device, is_amp_available, is_torch_device_available

 T = TypeVar("T", bound="PreTrainedConfig")
 logger = getLogger(__name__)
@@ -50,9 +50,6 @@ class TrainPipelineConfig(HubMixin):
    # `seed` is used for training (eg: model initialization, dataset shuffling)
    # AND for the evaluation environments.
    seed: int | None = 1000
-    # Set to True to use deterministic cuDNN algorithms for reproducibility.
-    # This disables cudnn.benchmark and may reduce training speed by ~10-20 percent.
-    cudnn_deterministic: bool = False
    # Number of workers for the dataloader.
    num_workers: int = 4
    batch_size: int = 8
@@ -746,8 +746,7 @@ def save_annotations_to_dataset(
    dataset_path: Path, annotations: dict[int, SubtaskAnnotation], fps: int, prefix: str = "sparse"
 ):
    """Save annotations to LeRobot dataset parquet format."""
-    from lerobot.datasets.io_utils import load_episodes
-    from lerobot.datasets.utils import DEFAULT_EPISODES_PATH
+    from lerobot.datasets.utils import DEFAULT_EPISODES_PATH, load_episodes

    episodes_dataset = load_episodes(dataset_path)
    if not episodes_dataset or len(episodes_dataset) == 0:
@@ -841,7 +840,7 @@ def generate_auto_sparse_annotations(

 def load_annotations_from_dataset(dataset_path: Path, prefix: str = "sparse") -> dict[int, SubtaskAnnotation]:
    """Load annotations from LeRobot dataset parquet files."""
-    from lerobot.datasets.io_utils import load_episodes
+    from lerobot.datasets.utils import load_episodes

    episodes_dataset = load_episodes(dataset_path)
    if not episodes_dataset or len(episodes_dataset) == 0:
@@ -24,16 +24,7 @@ import pandas as pd
 import tqdm

 from lerobot.datasets.compute_stats import aggregate_stats
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.feature_utils import get_hf_features_from_features
-from lerobot.datasets.io_utils import (
-    get_file_size_in_mb,
-    get_parquet_file_size_in_mb,
-    to_parquet_with_hf_images,
-    write_info,
-    write_stats,
-    write_tasks,
-)
+from lerobot.datasets.lerobot_dataset import LeRobotDatasetMetadata
 from lerobot.datasets.utils import (
    DEFAULT_CHUNK_SIZE,
    DEFAULT_DATA_FILE_SIZE_IN_MB,
@@ -41,7 +32,14 @@ from lerobot.datasets.utils import (
    DEFAULT_EPISODES_PATH,
    DEFAULT_VIDEO_FILE_SIZE_IN_MB,
    DEFAULT_VIDEO_PATH,
+    get_file_size_in_mb,
+    get_hf_features_from_features,
+    get_parquet_file_size_in_mb,
+    to_parquet_with_hf_images,
    update_chunk_file_indices,
+    write_info,
+    write_stats,
+    write_tasks,
 )
 from lerobot.datasets.video_utils import concatenate_video_files, get_video_duration_in_s

@@ -291,9 +289,7 @@ def aggregate_datasets(

    logging.info("Find all tasks")
    unique_tasks = pd.concat([m.tasks for m in all_metadata]).index.unique()
-    dst_meta.tasks = pd.DataFrame(
-        {"task_index": range(len(unique_tasks))}, index=pd.Index(unique_tasks, name="task")
-    )
+    dst_meta.tasks = pd.DataFrame({"task_index": range(len(unique_tasks))}, index=unique_tasks)

    meta_idx = {"chunk": 0, "file": 0}
    data_idx = {"chunk": 0, "file": 0}
@@ -0,0 +1,56 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import packaging.version
+
+V30_MESSAGE = """
+The dataset you requested ({repo_id}) is in {version} format.
+
+We introduced a new format since v3.0 which is not backward compatible with v2.1.
+Please, update your dataset to the new format using this command:
+```
+python -m lerobot.datasets.v30.convert_dataset_v21_to_v30 --repo-id={repo_id}
+```
+
+If you already have a converted version uploaded to the hub, then this error might be because of
+an older version in your local cache. Consider deleting the cached version and retrying.
+
+If you encounter a problem, contact LeRobot maintainers on [Discord](https://discord.com/invite/s3KuuzsPFb)
+or open an [issue on GitHub](https://github.com/huggingface/lerobot/issues/new/choose).
+"""
+
+FUTURE_MESSAGE = """
+The dataset you requested ({repo_id}) is only available in {version} format.
+As we cannot ensure forward compatibility with it, please update your current version of lerobot.
+"""
+
+
+class CompatibilityError(Exception): ...
+
+
+class BackwardCompatibilityError(CompatibilityError):
+    def __init__(self, repo_id: str, version: packaging.version.Version):
+        if version.major == 2 and version.minor == 1:
+            message = V30_MESSAGE.format(repo_id=repo_id, version=version)
+        else:
+            raise NotImplementedError(
+                "Contact the maintainer on [Discord](https://discord.com/invite/s3KuuzsPFb)."
+            )
+        super().__init__(message)
+
+
+class ForwardCompatibilityError(CompatibilityError):
+    def __init__(self, repo_id: str, version: packaging.version.Version):
+        message = FUTURE_MESSAGE.format(repo_id=repo_id, version=version)
+        super().__init__(message)
@@ -7,13 +7,6 @@

 This dataset was created using [LeRobot](https://github.com/huggingface/lerobot).

-{% if repo_id is defined and repo_id %}
-<a class="flex" href="https://huggingface.co/spaces/lerobot/visualize_dataset?path={{ repo_id }}">
-<img class="block dark:hidden" src="https://huggingface.co/datasets/huggingface/badges/resolve/main/visualize-this-dataset-xl.svg"/>
-<img class="hidden dark:block" src="https://huggingface.co/datasets/huggingface/badges/resolve/main/visualize-this-dataset-xl-dark.svg"/>
-</a>
-{% endif %}
-
 ## Dataset Description

 {{ dataset_description | default("", true) }}
@@ -15,7 +15,7 @@
 # limitations under the License.
 import numpy as np

-from lerobot.datasets.io_utils import load_image_as_numpy
+from lerobot.datasets.utils import load_image_as_numpy

 DEFAULT_QUANTILES = [0.01, 0.10, 0.50, 0.90, 0.99]

@@ -1,517 +0,0 @@
-#!/usr/bin/env python
-
-# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-#     http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-from pathlib import Path
-
-import numpy as np
-import packaging.version
-import pandas as pd
-import pyarrow as pa
-import pyarrow.parquet as pq
-from huggingface_hub import snapshot_download
-
-from lerobot.datasets.compute_stats import aggregate_stats
-from lerobot.datasets.feature_utils import _validate_feature_names, create_empty_dataset_info
-from lerobot.datasets.io_utils import (
-    get_file_size_in_mb,
-    load_episodes,
-    load_info,
-    load_stats,
-    load_subtasks,
-    load_tasks,
-    write_info,
-    write_json,
-    write_stats,
-    write_tasks,
-)
-from lerobot.datasets.utils import (
-    DEFAULT_EPISODES_PATH,
-    DEFAULT_FEATURES,
-    INFO_PATH,
-    check_version_compatibility,
-    flatten_dict,
-    get_safe_version,
-    is_valid_version,
-    update_chunk_file_indices,
-)
-from lerobot.datasets.video_utils import get_video_info
-from lerobot.utils.constants import HF_LEROBOT_HOME
-
-CODEBASE_VERSION = "v3.0"
-
-
-class LeRobotDatasetMetadata:
-    def __init__(
-        self,
-        repo_id: str,
-        root: str | Path | None = None,
-        revision: str | None = None,
-        force_cache_sync: bool = False,
-        metadata_buffer_size: int = 10,
-    ):
-        self.repo_id = repo_id
-        self.revision = revision if revision else CODEBASE_VERSION
-        self.root = Path(root) if root is not None else HF_LEROBOT_HOME / repo_id
-        self.writer = None
-        self.latest_episode = None
-        self.metadata_buffer: list[dict] = []
-        self.metadata_buffer_size = metadata_buffer_size
-
-        try:
-            if force_cache_sync:
-                raise FileNotFoundError
-            self.load_metadata()
-        except (FileNotFoundError, NotADirectoryError):
-            if is_valid_version(self.revision):
-                self.revision = get_safe_version(self.repo_id, self.revision)
-
-            (self.root / "meta").mkdir(exist_ok=True, parents=True)
-            self.pull_from_repo(allow_patterns="meta/")
-            self.load_metadata()
-
-    def _flush_metadata_buffer(self) -> None:
-        """Write all buffered episode metadata to parquet file."""
-        if not hasattr(self, "metadata_buffer") or len(self.metadata_buffer) == 0:
-            return
-
-        combined_dict = {}
-        for episode_dict in self.metadata_buffer:
-            for key, value in episode_dict.items():
-                if key not in combined_dict:
-                    combined_dict[key] = []
-                # Extract value and serialize numpy arrays
-                # because PyArrow's from_pydict function doesn't support numpy arrays
-                val = value[0] if isinstance(value, list) else value
-                combined_dict[key].append(val.tolist() if isinstance(val, np.ndarray) else val)
-
-        first_ep = self.metadata_buffer[0]
-        chunk_idx = first_ep["meta/episodes/chunk_index"][0]
-        file_idx = first_ep["meta/episodes/file_index"][0]
-
-        table = pa.Table.from_pydict(combined_dict)
-
-        if not self.writer:
-            path = Path(self.root / DEFAULT_EPISODES_PATH.format(chunk_index=chunk_idx, file_index=file_idx))
-            path.parent.mkdir(parents=True, exist_ok=True)
-            self.writer = pq.ParquetWriter(
-                path, schema=table.schema, compression="snappy", use_dictionary=True
-            )
-
-        self.writer.write_table(table)
-
-        self.latest_episode = self.metadata_buffer[-1]
-        self.metadata_buffer.clear()
-
-    def _close_writer(self) -> None:
-        """Close and cleanup the parquet writer if it exists."""
-        self._flush_metadata_buffer()
-
-        writer = getattr(self, "writer", None)
-        if writer is not None:
-            writer.close()
-            self.writer = None
-
-    def __del__(self):
-        """
-        Trust the user to call .finalize() but as an added safety check call the parquet writer to stop when calling the destructor
-        """
-        self._close_writer()
-
-    def load_metadata(self):
-        self.info = load_info(self.root)
-        check_version_compatibility(self.repo_id, self._version, CODEBASE_VERSION)
-        self.tasks = load_tasks(self.root)
-        self.subtasks = load_subtasks(self.root)
-        self.episodes = load_episodes(self.root)
-        self.stats = load_stats(self.root)
-
-    def pull_from_repo(
-        self,
-        allow_patterns: list[str] | str | None = None,
-        ignore_patterns: list[str] | str | None = None,
-    ) -> None:
-        snapshot_download(
-            self.repo_id,
-            repo_type="dataset",
-            revision=self.revision,
-            local_dir=self.root,
-            allow_patterns=allow_patterns,
-            ignore_patterns=ignore_patterns,
-        )
-
-    @property
-    def url_root(self) -> str:
-        return f"hf://datasets/{self.repo_id}"
-
-    @property
-    def _version(self) -> packaging.version.Version:
-        """Codebase version used to create this dataset."""
-        return packaging.version.parse(self.info["codebase_version"])
-
-    def get_data_file_path(self, ep_index: int) -> Path:
-        if self.episodes is None:
-            self.episodes = load_episodes(self.root)
-        if ep_index >= len(self.episodes):
-            raise IndexError(
-                f"Episode index {ep_index} out of range. Episodes: {len(self.episodes) if self.episodes else 0}"
-            )
-        ep = self.episodes[ep_index]
-        chunk_idx = ep["data/chunk_index"]
-        file_idx = ep["data/file_index"]
-        fpath = self.data_path.format(chunk_index=chunk_idx, file_index=file_idx)
-        return Path(fpath)
-
-    def get_video_file_path(self, ep_index: int, vid_key: str) -> Path:
-        if self.episodes is None:
-            self.episodes = load_episodes(self.root)
-        if ep_index >= len(self.episodes):
-            raise IndexError(
-                f"Episode index {ep_index} out of range. Episodes: {len(self.episodes) if self.episodes else 0}"
-            )
-        ep = self.episodes[ep_index]
-        chunk_idx = ep[f"videos/{vid_key}/chunk_index"]
-        file_idx = ep[f"videos/{vid_key}/file_index"]
-        fpath = self.video_path.format(video_key=vid_key, chunk_index=chunk_idx, file_index=file_idx)
-        return Path(fpath)
-
-    @property
-    def data_path(self) -> str:
-        """Formattable string for the parquet files."""
-        return self.info["data_path"]
-
-    @property
-    def video_path(self) -> str | None:
-        """Formattable string for the video files."""
-        return self.info["video_path"]
-
-    @property
-    def robot_type(self) -> str | None:
-        """Robot type used in recording this dataset."""
-        return self.info["robot_type"]
-
-    @property
-    def fps(self) -> int:
-        """Frames per second used during data collection."""
-        return self.info["fps"]
-
-    @property
-    def features(self) -> dict[str, dict]:
-        """All features contained in the dataset."""
-        return self.info["features"]
-
-    @property
-    def image_keys(self) -> list[str]:
-        """Keys to access visual modalities stored as images."""
-        return [key for key, ft in self.features.items() if ft["dtype"] == "image"]
-
-    @property
-    def video_keys(self) -> list[str]:
-        """Keys to access visual modalities stored as videos."""
-        return [key for key, ft in self.features.items() if ft["dtype"] == "video"]
-
-    @property
-    def camera_keys(self) -> list[str]:
-        """Keys to access visual modalities (regardless of their storage method)."""
-        return [key for key, ft in self.features.items() if ft["dtype"] in ["video", "image"]]
-
-    @property
-    def names(self) -> dict[str, list | dict]:
-        """Names of the various dimensions of vector modalities."""
-        return {key: ft["names"] for key, ft in self.features.items()}
-
-    @property
-    def shapes(self) -> dict:
-        """Shapes for the different features."""
-        return {key: tuple(ft["shape"]) for key, ft in self.features.items()}
-
-    @property
-    def total_episodes(self) -> int:
-        """Total number of episodes available."""
-        return self.info["total_episodes"]
-
-    @property
-    def total_frames(self) -> int:
-        """Total number of frames saved in this dataset."""
-        return self.info["total_frames"]
-
-    @property
-    def total_tasks(self) -> int:
-        """Total number of different tasks performed in this dataset."""
-        return self.info["total_tasks"]
-
-    @property
-    def chunks_size(self) -> int:
-        """Max number of files per chunk."""
-        return self.info["chunks_size"]
-
-    @property
-    def data_files_size_in_mb(self) -> int:
-        """Max size of data file in mega bytes."""
-        return self.info["data_files_size_in_mb"]
-
-    @property
-    def video_files_size_in_mb(self) -> int:
-        """Max size of video file in mega bytes."""
-        return self.info["video_files_size_in_mb"]
-
-    def get_task_index(self, task: str) -> int | None:
-        """
-        Given a task in natural language, returns its task_index if the task already exists in the dataset,
-        otherwise return None.
-        """
-        if task in self.tasks.index:
-            return int(self.tasks.loc[task].task_index)
-        else:
-            return None
-
-    def save_episode_tasks(self, tasks: list[str]):
-        if len(set(tasks)) != len(tasks):
-            raise ValueError(f"Tasks are not unique: {tasks}")
-
-        if self.tasks is None:
-            new_tasks = tasks
-            task_indices = range(len(tasks))
-            self.tasks = pd.DataFrame({"task_index": task_indices}, index=pd.Index(tasks, name="task"))
-        else:
-            new_tasks = [task for task in tasks if task not in self.tasks.index]
-            new_task_indices = range(len(self.tasks), len(self.tasks) + len(new_tasks))
-            for task_idx, task in zip(new_task_indices, new_tasks, strict=False):
-                self.tasks.loc[task] = task_idx
-
-        if len(new_tasks) > 0:
-            # Update on disk
-            write_tasks(self.tasks, self.root)
-
-    def _save_episode_metadata(self, episode_dict: dict) -> None:
-        """Buffer episode metadata and write to parquet in batches for efficiency.
-
-        This function accumulates episode metadata in a buffer and flushes it when the buffer
-        reaches the configured size. This reduces I/O overhead by writing multiple episodes
-        at once instead of one row at a time.
-
-        Notes: We both need to update parquet files and HF dataset:
-        - `pandas` loads parquet file in RAM
-        - `datasets` relies on a memory mapping from pyarrow (no RAM). It either converts parquet files to a pyarrow cache on disk,
-          or loads directly from pyarrow cache.
-        """
-        # Convert to list format for each value
-        episode_dict = {key: [value] for key, value in episode_dict.items()}
-        num_frames = episode_dict["length"][0]
-
-        if self.latest_episode is None:
-            # Initialize indices and frame count for a new dataset made of the first episode data
-            chunk_idx, file_idx = 0, 0
-            if self.episodes is not None and len(self.episodes) > 0:
-                # It means we are resuming recording, so we need to load the latest episode
-                # Update the indices to avoid overwriting the latest episode
-                chunk_idx = self.episodes[-1]["meta/episodes/chunk_index"]
-                file_idx = self.episodes[-1]["meta/episodes/file_index"]
-                latest_num_frames = self.episodes[-1]["dataset_to_index"]
-                episode_dict["dataset_from_index"] = [latest_num_frames]
-                episode_dict["dataset_to_index"] = [latest_num_frames + num_frames]
-
-                # When resuming, move to the next file
-                chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, self.chunks_size)
-            else:
-                episode_dict["dataset_from_index"] = [0]
-                episode_dict["dataset_to_index"] = [num_frames]
-
-            episode_dict["meta/episodes/chunk_index"] = [chunk_idx]
-            episode_dict["meta/episodes/file_index"] = [file_idx]
-        else:
-            chunk_idx = self.latest_episode["meta/episodes/chunk_index"][0]
-            file_idx = self.latest_episode["meta/episodes/file_index"][0]
-
-            latest_path = (
-                self.root / DEFAULT_EPISODES_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
-                if self.writer is None
-                else self.writer.where
-            )
-
-            if Path(latest_path).exists():
-                latest_size_in_mb = get_file_size_in_mb(Path(latest_path))
-                latest_num_frames = self.latest_episode["episode_index"][0]
-
-                av_size_per_frame = latest_size_in_mb / latest_num_frames if latest_num_frames > 0 else 0.0
-
-                if latest_size_in_mb + av_size_per_frame * num_frames >= self.data_files_size_in_mb:
-                    # Size limit is reached, flush buffer and prepare new parquet file
-                    self._flush_metadata_buffer()
-                    chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, self.chunks_size)
-                    self._close_writer()
-
-            # Update the existing pandas dataframe with new row
-            episode_dict["meta/episodes/chunk_index"] = [chunk_idx]
-            episode_dict["meta/episodes/file_index"] = [file_idx]
-            episode_dict["dataset_from_index"] = [self.latest_episode["dataset_to_index"][0]]
-            episode_dict["dataset_to_index"] = [self.latest_episode["dataset_to_index"][0] + num_frames]
-
-        # Add to buffer
-        self.metadata_buffer.append(episode_dict)
-        self.latest_episode = episode_dict
-
-        if len(self.metadata_buffer) >= self.metadata_buffer_size:
-            self._flush_metadata_buffer()
-
-    def save_episode(
-        self,
-        episode_index: int,
-        episode_length: int,
-        episode_tasks: list[str],
-        episode_stats: dict[str, dict],
-        episode_metadata: dict,
-    ) -> None:
-        episode_dict = {
-            "episode_index": episode_index,
-            "tasks": episode_tasks,
-            "length": episode_length,
-        }
-        episode_dict.update(episode_metadata)
-        episode_dict.update(flatten_dict({"stats": episode_stats}))
-        self._save_episode_metadata(episode_dict)
-
-        # Update info
-        self.info["total_episodes"] += 1
-        self.info["total_frames"] += episode_length
-        self.info["total_tasks"] = len(self.tasks)
-        self.info["splits"] = {"train": f"0:{self.info['total_episodes']}"}
-
-        write_info(self.info, self.root)
-
-        self.stats = aggregate_stats([self.stats, episode_stats]) if self.stats is not None else episode_stats
-        write_stats(self.stats, self.root)
-
-    def update_video_info(self, video_key: str | None = None) -> None:
-        """
-        Warning: this function writes info from first episode videos, implicitly assuming that all videos have
-        been encoded the same way. Also, this means it assumes the first episode exists.
-        """
-        if video_key is not None and video_key not in self.video_keys:
-            raise ValueError(f"Video key {video_key} not found in dataset")
-
-        video_keys = [video_key] if video_key is not None else self.video_keys
-        for key in video_keys:
-            if not self.features[key].get("info", None):
-                video_path = self.root / self.video_path.format(video_key=key, chunk_index=0, file_index=0)
-                self.info["features"][key]["info"] = get_video_info(video_path)
-
-    def update_chunk_settings(
-        self,
-        chunks_size: int | None = None,
-        data_files_size_in_mb: int | None = None,
-        video_files_size_in_mb: int | None = None,
-    ) -> None:
-        """Update chunk and file size settings after dataset creation.
-
-        This allows users to customize storage organization without modifying the constructor.
-        These settings control how episodes are chunked and how large files can grow before
-        creating new ones.
-
-        Args:
-            chunks_size: Maximum number of files per chunk directory. If None, keeps current value.
-            data_files_size_in_mb: Maximum size for data parquet files in MB. If None, keeps current value.
-            video_files_size_in_mb: Maximum size for video files in MB. If None, keeps current value.
-        """
-        if chunks_size is not None:
-            if chunks_size <= 0:
-                raise ValueError(f"chunks_size must be positive, got {chunks_size}")
-            self.info["chunks_size"] = chunks_size
-
-        if data_files_size_in_mb is not None:
-            if data_files_size_in_mb <= 0:
-                raise ValueError(f"data_files_size_in_mb must be positive, got {data_files_size_in_mb}")
-            self.info["data_files_size_in_mb"] = data_files_size_in_mb
-
-        if video_files_size_in_mb is not None:
-            if video_files_size_in_mb <= 0:
-                raise ValueError(f"video_files_size_in_mb must be positive, got {video_files_size_in_mb}")
-            self.info["video_files_size_in_mb"] = video_files_size_in_mb
-
-        # Update the info file on disk
-        write_info(self.info, self.root)
-
-    def get_chunk_settings(self) -> dict[str, int]:
-        """Get current chunk and file size settings.
-
-        Returns:
-            Dict containing chunks_size, data_files_size_in_mb, and video_files_size_in_mb.
-        """
-        return {
-            "chunks_size": self.chunks_size,
-            "data_files_size_in_mb": self.data_files_size_in_mb,
-            "video_files_size_in_mb": self.video_files_size_in_mb,
-        }
-
-    def __repr__(self):
-        feature_keys = list(self.features)
-        return (
-            f"{self.__class__.__name__}({{\n"
-            f"    Repository ID: '{self.repo_id}',\n"
-            f"    Total episodes: '{self.total_episodes}',\n"
-            f"    Total frames: '{self.total_frames}',\n"
-            f"    Features: '{feature_keys}',\n"
-            "})',\n"
-        )
-
-    @classmethod
-    def create(
-        cls,
-        repo_id: str,
-        fps: int,
-        features: dict,
-        robot_type: str | None = None,
-        root: str | Path | None = None,
-        use_videos: bool = True,
-        metadata_buffer_size: int = 10,
-        chunks_size: int | None = None,
-        data_files_size_in_mb: int | None = None,
-        video_files_size_in_mb: int | None = None,
-    ) -> "LeRobotDatasetMetadata":
-        """Creates metadata for a LeRobotDataset."""
-        obj = cls.__new__(cls)
-        obj.repo_id = repo_id
-        obj.root = Path(root) if root is not None else HF_LEROBOT_HOME / repo_id
-
-        obj.root.mkdir(parents=True, exist_ok=False)
-
-        features = {**features, **DEFAULT_FEATURES}
-        _validate_feature_names(features)
-
-        obj.tasks = None
-        obj.subtasks = None
-        obj.episodes = None
-        obj.stats = None
-        obj.info = create_empty_dataset_info(
-            CODEBASE_VERSION,
-            fps,
-            features,
-            use_videos,
-            robot_type,
-            chunks_size,
-            data_files_size_in_mb,
-            video_files_size_in_mb,
-        )
-        if len(obj.video_keys) > 0 and not use_videos:
-            raise ValueError(
-                f"Features contain video keys {obj.video_keys}, but 'use_videos' is set to False. "
-                "Either remove video features from the features dict, or set 'use_videos=True'."
-            )
-        write_json(obj.info, obj.root / INFO_PATH)
-        obj.revision = None
-        obj.writer = None
-        obj.latest_episode = None
-        obj.metadata_buffer = []
-        obj.metadata_buffer_size = metadata_buffer_size
-        return obj
@@ -38,22 +38,19 @@ from tqdm import tqdm

 from lerobot.datasets.aggregate import aggregate_datasets
 from lerobot.datasets.compute_stats import aggregate_stats
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.io_utils import (
-    get_parquet_file_size_in_mb,
-    load_episodes,
-    write_info,
-    write_stats,
-    write_tasks,
-)
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.lerobot_dataset import LeRobotDataset, LeRobotDatasetMetadata
 from lerobot.datasets.utils import (
    DATA_DIR,
    DEFAULT_CHUNK_SIZE,
    DEFAULT_DATA_FILE_SIZE_IN_MB,
    DEFAULT_DATA_PATH,
    DEFAULT_EPISODES_PATH,
+    get_parquet_file_size_in_mb,
+    load_episodes,
    update_chunk_file_indices,
+    write_info,
+    write_stats,
+    write_tasks,
 )
 from lerobot.datasets.video_utils import encode_video_frames, get_video_info
 from lerobot.utils.constants import HF_LEROBOT_HOME, OBS_IMAGE
@@ -92,8 +89,8 @@ def delete_episodes(
    Args:
        dataset: The source LeRobotDataset.
        episode_indices: List of episode indices to delete.
-        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
-        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+        output_dir: Directory to save the new dataset. If None, uses default location.
+        repo_id: Repository ID for the new dataset. If None, appends "_modified" to original.
    """
    if not episode_indices:
        raise ValueError("No episodes to delete")
@@ -155,7 +152,7 @@ def split_dataset(
        dataset: The source LeRobotDataset to split.
        splits: Either a dict mapping split names to episode indices, or a dict mapping
                split names to fractions (must sum to <= 1.0).
-        output_dir: Root directory where the split datasets will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id.
+        output_dir: Base directory for output datasets. If None, uses default location.

    Examples:
      Split by specific episodes
@@ -246,8 +243,8 @@ def merge_datasets(

    Args:
        datasets: List of LeRobotDatasets to merge.
-        output_repo_id: Merged dataset identifier.
-        output_dir: Root directory where the merged dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/output_repo_id.
+        output_repo_id: Repository ID for the merged dataset.
+        output_dir: Directory to save the merged dataset. If None, uses default location.
    """
    if not datasets:
        raise ValueError("No datasets to merge")
@@ -291,8 +288,8 @@ def modify_features(
        dataset: The source LeRobotDataset.
        add_features: Optional dict mapping feature names to (feature_values, feature_info) tuples.
        remove_features: Optional feature name(s) to remove. Can be a single string or list.
-        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
-        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+        output_dir: Directory to save the new dataset. If None, uses default location.
+        repo_id: Repository ID for the new dataset. If None, appends "_modified" to original.

    Returns:
        New dataset with features modified.
@@ -393,8 +390,8 @@ def add_features(
    Args:
        dataset: The source LeRobotDataset.
        features: Dictionary mapping feature names to (feature_values, feature_info) tuples.
-        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
-        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+        output_dir: Directory to save the new dataset. If None, uses default location.
+        repo_id: Repository ID for the new dataset. If None, appends "_modified" to original.

    Returns:
        New dataset with all features added.
@@ -430,8 +427,8 @@ def remove_feature(
    Args:
        dataset: The source LeRobotDataset.
        feature_names: Name(s) of features to remove. Can be a single string or list.
-        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
-        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+        output_dir: Directory to save the new dataset. If None, uses default location.
+        repo_id: Repository ID for the new dataset. If None, appends "_modified" to original.

    Returns:
        New dataset with features removed.
@@ -570,22 +567,20 @@ def _copy_and_reindex_data(
 def _keep_episodes_from_video_with_av(
    input_path: Path,
    output_path: Path,
-    episodes_to_keep: list[tuple[int, int]],
+    episodes_to_keep: list[tuple[float, float]],
    fps: float,
    vcodec: str = "libsvtav1",
    pix_fmt: str = "yuv420p",
 ) -> None:
    """Keep only specified episodes from a video file using PyAV.

-    This function decodes frames from specified frame ranges and re-encodes them with
+    This function decodes frames from specified time ranges and re-encodes them with
    properly reset timestamps to ensure monotonic progression.

    Args:
        input_path: Source video file path.
        output_path: Destination video file path.
-        episodes_to_keep: List of (start_frame, end_frame) tuples for episodes to keep.
-            Ranges are half-open intervals: [start_frame, end_frame), where start_frame
-            is inclusive and end_frame is exclusive.
+        episodes_to_keep: List of (start_time, end_time) tuples for episodes to keep.
        fps: Frame rate of the video.
        vcodec: Video codec to use for encoding.
        pix_fmt: Pixel format for output video.
@@ -627,10 +622,9 @@ def _keep_episodes_from_video_with_av(

    # Create set of (start, end) ranges for fast lookup.
    # Convert to a sorted list for efficient checking.
-    frame_ranges = sorted(episodes_to_keep)
+    time_ranges = sorted(episodes_to_keep)

    # Track frame index for setting PTS and current range being processed.
-    src_frame_count = 0
    frame_count = 0
    range_idx = 0

@@ -640,20 +634,21 @@ def _keep_episodes_from_video_with_av(
            if frame is None:
                continue

-            # Check if frame is in any of our desired frame ranges.
+            # Get frame timestamp.
+            frame_time = float(frame.pts * frame.time_base) if frame.pts is not None else 0.0
+
+            # Check if frame is in any of our desired time ranges.
            # Skip ranges that have already passed.
-            while range_idx < len(frame_ranges) and src_frame_count >= frame_ranges[range_idx][1]:
+            while range_idx < len(time_ranges) and frame_time >= time_ranges[range_idx][1]:
                range_idx += 1

            # If we've passed all ranges, stop processing.
-            if range_idx >= len(frame_ranges):
+            if range_idx >= len(time_ranges):
                break

            # Check if frame is in current range.
-            start_frame = frame_ranges[range_idx][0]
-
-            if src_frame_count < start_frame:
-                src_frame_count += 1
+            start_ts, end_ts = time_ranges[range_idx]
+            if frame_time < start_ts:
                continue

            # Frame is in range - create a new frame with reset timestamps.
@@ -666,7 +661,6 @@ def _keep_episodes_from_video_with_av(
            for pkt in v_out.encode(new_frame):
                out.mux(pkt)

-            src_frame_count += 1
            frame_count += 1

    # Flush encoder.
@@ -755,17 +749,15 @@ def _copy_and_reindex_videos(
                        f"videos/{video_key}/to_timestamp"
                    ]
            else:
-                # Build list of frame ranges to keep, in sorted order.
+                # Build list of time ranges to keep, in sorted order.
                sorted_keep_episodes = sorted(episodes_in_file, key=lambda x: episode_mapping[x])
-                episodes_to_keep_ranges: list[tuple[int, int]] = []
+                episodes_to_keep_ranges: list[tuple[float, float]] = []
+
                for old_idx in sorted_keep_episodes:
                    src_ep = src_dataset.meta.episodes[old_idx]
-                    from_frame = round(src_ep[f"videos/{video_key}/from_timestamp"] * src_dataset.meta.fps)
-                    to_frame = round(src_ep[f"videos/{video_key}/to_timestamp"] * src_dataset.meta.fps)
-                    assert src_ep["length"] == to_frame - from_frame, (
-                        f"Episode length mismatch: {src_ep['length']} vs {to_frame - from_frame}"
-                    )
-                    episodes_to_keep_ranges.append((from_frame, to_frame))
+                    from_ts = src_ep[f"videos/{video_key}/from_timestamp"]
+                    to_ts = src_ep[f"videos/{video_key}/to_timestamp"]
+                    episodes_to_keep_ranges.append((from_ts, to_ts))

                # Use PyAV filters to efficiently re-encode only the desired segments.
                assert src_dataset.meta.video_path is not None
@@ -918,8 +910,7 @@ def _write_parquet(df: pd.DataFrame, path: Path, meta: LeRobotDatasetMetadata) -

    This ensures images are properly embedded and the file can be loaded correctly by HF datasets.
    """
-    from lerobot.datasets.feature_utils import get_hf_features_from_features
-    from lerobot.datasets.io_utils import embed_images
+    from lerobot.datasets.utils import embed_images, get_hf_features_from_features

    hf_features = get_hf_features_from_features(meta.features)
    ep_dataset = datasets.Dataset.from_dict(df.to_dict(orient="list"), features=hf_features, split="train")
@@ -1479,9 +1470,7 @@ def modify_tasks(

    # Collect all unique tasks and create new task mapping
    unique_tasks = sorted(set(episode_to_task.values()))
-    new_task_df = pd.DataFrame(
-        {"task_index": list(range(len(unique_tasks)))}, index=pd.Index(unique_tasks, name="task")
-    )
+    new_task_df = pd.DataFrame({"task_index": list(range(len(unique_tasks)))}, index=unique_tasks)
    task_to_index = {task: idx for idx, task in enumerate(unique_tasks)}

    logging.info(f"Modifying tasks in {dataset.repo_id}")
@@ -1535,7 +1524,7 @@ def modify_tasks(

 def convert_image_to_video_dataset(
    dataset: LeRobotDataset,
-    output_dir: Path | None = None,
+    output_dir: Path,
    repo_id: str | None = None,
    vcodec: str = "libsvtav1",
    pix_fmt: str = "yuv420p",
@@ -1554,8 +1543,8 @@ def convert_image_to_video_dataset(

    Args:
        dataset: The source LeRobot dataset with images
-        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
-        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+        output_dir: Directory to save the new video dataset
+        repo_id: Repository ID for the new dataset (default: original_id + "_video")
        vcodec: Video codec (default: libsvtav1)
        pix_fmt: Pixel format (default: yuv420p)
        g: Group of pictures size (default: 2)
@@ -1606,7 +1595,6 @@ def convert_image_to_video_dataset(
            # Video info will be updated after episodes are encoded

    # Create new metadata for video dataset
-    output_dir = Path(output_dir) if output_dir is not None else HF_LEROBOT_HOME / repo_id
    new_meta = LeRobotDatasetMetadata.create(
        repo_id=repo_id,
        fps=dataset.meta.fps,
@@ -20,9 +20,11 @@ import torch

 from lerobot.configs.policies import PreTrainedConfig
 from lerobot.configs.train import TrainPipelineConfig
-from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
-from lerobot.datasets.multi_dataset import MultiLeRobotDataset
+from lerobot.datasets.lerobot_dataset import (
+    LeRobotDataset,
+    LeRobotDatasetMetadata,
+    MultiLeRobotDataset,
+)
 from lerobot.datasets.streaming_dataset import StreamingLeRobotDataset
 from lerobot.datasets.transforms import ImageTransforms
 from lerobot.utils.constants import ACTION, OBS_PREFIX, REWARD
@@ -1,552 +0,0 @@
-#!/usr/bin/env python
-
-# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-#     http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-from pprint import pformat
-from typing import Any
-
-import datasets
-import numpy as np
-from PIL import Image as PILImage
-
-from lerobot.configs.types import FeatureType, PolicyFeature
-from lerobot.datasets.utils import (
-    DEFAULT_CHUNK_SIZE,
-    DEFAULT_DATA_FILE_SIZE_IN_MB,
-    DEFAULT_DATA_PATH,
-    DEFAULT_FEATURES,
-    DEFAULT_VIDEO_FILE_SIZE_IN_MB,
-    DEFAULT_VIDEO_PATH,
-)
-from lerobot.utils.constants import ACTION, OBS_ENV_STATE, OBS_STR
-from lerobot.utils.utils import is_valid_numpy_dtype_string
-
-
-def get_hf_features_from_features(features: dict) -> datasets.Features:
-    """Convert a LeRobot features dictionary to a `datasets.Features` object.
-
-    Args:
-        features (dict): A LeRobot-style feature dictionary.
-
-    Returns:
-        datasets.Features: The corresponding Hugging Face `datasets.Features` object.
-
-    Raises:
-        ValueError: If a feature has an unsupported shape.
-    """
-    hf_features = {}
-    for key, ft in features.items():
-        if ft["dtype"] == "video":
-            continue
-        elif ft["dtype"] == "image":
-            hf_features[key] = datasets.Image()
-        elif ft["shape"] == (1,):
-            hf_features[key] = datasets.Value(dtype=ft["dtype"])
-        elif len(ft["shape"]) == 1:
-            hf_features[key] = datasets.Sequence(
-                length=ft["shape"][0], feature=datasets.Value(dtype=ft["dtype"])
-            )
-        elif len(ft["shape"]) == 2:
-            hf_features[key] = datasets.Array2D(shape=ft["shape"], dtype=ft["dtype"])
-        elif len(ft["shape"]) == 3:
-            hf_features[key] = datasets.Array3D(shape=ft["shape"], dtype=ft["dtype"])
-        elif len(ft["shape"]) == 4:
-            hf_features[key] = datasets.Array4D(shape=ft["shape"], dtype=ft["dtype"])
-        elif len(ft["shape"]) == 5:
-            hf_features[key] = datasets.Array5D(shape=ft["shape"], dtype=ft["dtype"])
-        else:
-            raise ValueError(f"Corresponding feature is not valid: {ft}")
-
-    return datasets.Features(hf_features)
-
-
-def _validate_feature_names(features: dict[str, dict]) -> None:
-    """Validate that feature names do not contain invalid characters.
-
-    Args:
-        features (dict): The LeRobot features dictionary.
-
-    Raises:
-        ValueError: If any feature name contains '/'.
-    """
-    invalid_features = {name: ft for name, ft in features.items() if "/" in name}
-    if invalid_features:
-        raise ValueError(f"Feature names should not contain '/'. Found '/' in '{invalid_features}'.")
-
-
-def hw_to_dataset_features(
-    hw_features: dict[str, type | tuple], prefix: str, use_video: bool = True
-) -> dict[str, dict]:
-    """Convert hardware-specific features to a LeRobot dataset feature dictionary.
-
-    This function takes a dictionary describing hardware outputs (like joint states
-    or camera image shapes) and formats it into the standard LeRobot feature
-    specification.
-
-    Args:
-        hw_features (dict): Dictionary mapping feature names to their type (float for
-            joints) or shape (tuple for images).
-        prefix (str): The prefix to add to the feature keys (e.g., "observation"
-            or "action").
-        use_video (bool): If True, image features are marked as "video", otherwise "image".
-
-    Returns:
-        dict: A LeRobot features dictionary.
-    """
-    features = {}
-    joint_fts = {
-        key: ftype
-        for key, ftype in hw_features.items()
-        if ftype is float or (isinstance(ftype, PolicyFeature) and ftype.type != FeatureType.VISUAL)
-    }
-    cam_fts = {key: shape for key, shape in hw_features.items() if isinstance(shape, tuple)}
-
-    if joint_fts and prefix == ACTION:
-        features[prefix] = {
-            "dtype": "float32",
-            "shape": (len(joint_fts),),
-            "names": list(joint_fts),
-        }
-
-    if joint_fts and prefix == OBS_STR:
-        features[f"{prefix}.state"] = {
-            "dtype": "float32",
-            "shape": (len(joint_fts),),
-            "names": list(joint_fts),
-        }
-
-    for key, shape in cam_fts.items():
-        features[f"{prefix}.images.{key}"] = {
-            "dtype": "video" if use_video else "image",
-            "shape": shape,
-            "names": ["height", "width", "channels"],
-        }
-
-    _validate_feature_names(features)
-    return features
-
-
-def build_dataset_frame(
-    ds_features: dict[str, dict], values: dict[str, Any], prefix: str
-) -> dict[str, np.ndarray]:
-    """Construct a single data frame from raw values based on dataset features.
-
-    A "frame" is a dictionary containing all the data for a single timestep,
-    formatted as numpy arrays according to the feature specification.
-
-    Args:
-        ds_features (dict): The LeRobot dataset features dictionary.
-        values (dict): A dictionary of raw values from the hardware/environment.
-        prefix (str): The prefix to filter features by (e.g., "observation"
-            or "action").
-
-    Returns:
-        dict: A dictionary representing a single frame of data.
-    """
-    frame = {}
-    for key, ft in ds_features.items():
-        if key in DEFAULT_FEATURES or not key.startswith(prefix):
-            continue
-        elif ft["dtype"] == "float32" and len(ft["shape"]) == 1:
-            frame[key] = np.array([values[name] for name in ft["names"]], dtype=np.float32)
-        elif ft["dtype"] in ["image", "video"]:
-            frame[key] = values[key.removeprefix(f"{prefix}.images.")]
-
-    return frame
-
-
-def dataset_to_policy_features(features: dict[str, dict]) -> dict[str, PolicyFeature]:
-    """Convert dataset features to policy features.
-
-    This function transforms the dataset's feature specification into a format
-    that a policy can use, classifying features by type (e.g., visual, state,
-    action) and ensuring correct shapes (e.g., channel-first for images).
-
-    Args:
-        features (dict): The LeRobot dataset features dictionary.
-
-    Returns:
-        dict: A dictionary mapping feature keys to `PolicyFeature` objects.
-
-    Raises:
-        ValueError: If an image feature does not have a 3D shape.
-    """
-    # TODO(aliberts): Implement "type" in dataset features and simplify this
-    policy_features = {}
-    for key, ft in features.items():
-        shape = ft["shape"]
-        if ft["dtype"] in ["image", "video"]:
-            type = FeatureType.VISUAL
-            if len(shape) != 3:
-                raise ValueError(f"Number of dimensions of {key} != 3 (shape={shape})")
-
-            names = ft["names"]
-            # Backward compatibility for "channel" which is an error introduced in LeRobotDataset v2.0 for ported datasets.
-            if names[2] in ["channel", "channels"]:  # (h, w, c) -> (c, h, w)
-                shape = (shape[2], shape[0], shape[1])
-        elif key == OBS_ENV_STATE:
-            type = FeatureType.ENV
-        elif key.startswith(OBS_STR):
-            type = FeatureType.STATE
-        elif key.startswith(ACTION):
-            type = FeatureType.ACTION
-        else:
-            continue
-
-        policy_features[key] = PolicyFeature(
-            type=type,
-            shape=shape,
-        )
-
-    return policy_features
-
-
-def combine_feature_dicts(*dicts: dict) -> dict:
-    """Merge LeRobot grouped feature dicts.
-
-    - For 1D numeric specs (dtype not image/video/string) with "names": we merge the names and recompute the shape.
-    - For others (e.g. `observation.images.*`), the last one wins (if they are identical).
-
-    Args:
-        *dicts: A variable number of LeRobot feature dictionaries to merge.
-
-    Returns:
-        dict: A single merged feature dictionary.
-
-    Raises:
-        ValueError: If there's a dtype mismatch for a feature being merged.
-    """
-    out: dict = {}
-    for d in dicts:
-        for key, value in d.items():
-            if not isinstance(value, dict):
-                out[key] = value
-                continue
-
-            dtype = value.get("dtype")
-            shape = value.get("shape")
-            is_vector = (
-                dtype not in ("image", "video", "string")
-                and isinstance(shape, tuple)
-                and len(shape) == 1
-                and "names" in value
-            )
-
-            if is_vector:
-                # Initialize or retrieve the accumulating dict for this feature key
-                target = out.setdefault(key, {"dtype": dtype, "names": [], "shape": (0,)})
-                # Ensure consistent data types across merged entries
-                if "dtype" in target and dtype != target["dtype"]:
-                    raise ValueError(f"dtype mismatch for '{key}': {target['dtype']} vs {dtype}")
-
-                # Merge feature names: append only new ones to preserve order without duplicates
-                seen = set(target["names"])
-                for n in value["names"]:
-                    if n not in seen:
-                        target["names"].append(n)
-                        seen.add(n)
-                # Recompute the shape to reflect the updated number of features
-                target["shape"] = (len(target["names"]),)
-            else:
-                # For images/videos and non-1D entries: override with the latest definition
-                out[key] = value
-    return out
-
-
-def create_empty_dataset_info(
-    codebase_version: str,
-    fps: int,
-    features: dict,
-    use_videos: bool,
-    robot_type: str | None = None,
-    chunks_size: int | None = None,
-    data_files_size_in_mb: int | None = None,
-    video_files_size_in_mb: int | None = None,
-) -> dict:
-    """Create a template dictionary for a new dataset's `info.json`.
-
-    Args:
-        codebase_version (str): The version of the LeRobot codebase.
-        fps (int): The frames per second of the data.
-        features (dict): The LeRobot features dictionary for the dataset.
-        use_videos (bool): Whether the dataset will store videos.
-        robot_type (str | None): The type of robot used, if any.
-
-    Returns:
-        dict: A dictionary with the initial dataset metadata.
-    """
-    return {
-        "codebase_version": codebase_version,
-        "robot_type": robot_type,
-        "total_episodes": 0,
-        "total_frames": 0,
-        "total_tasks": 0,
-        "chunks_size": chunks_size or DEFAULT_CHUNK_SIZE,
-        "data_files_size_in_mb": data_files_size_in_mb or DEFAULT_DATA_FILE_SIZE_IN_MB,
-        "video_files_size_in_mb": video_files_size_in_mb or DEFAULT_VIDEO_FILE_SIZE_IN_MB,
-        "fps": fps,
-        "splits": {},
-        "data_path": DEFAULT_DATA_PATH,
-        "video_path": DEFAULT_VIDEO_PATH if use_videos else None,
-        "features": features,
-    }
-
-
-def check_delta_timestamps(
-    delta_timestamps: dict[str, list[float]], fps: int, tolerance_s: float, raise_value_error: bool = True
-) -> bool:
-    """Check if delta timestamps are multiples of 1/fps +/- tolerance.
-
-    This ensures that adding these delta timestamps to any existing timestamp in
-    the dataset will result in a value that aligns with the dataset's frame rate.
-
-    Args:
-        delta_timestamps (dict): A dictionary where values are lists of time
-            deltas in seconds.
-        fps (int): The frames per second of the dataset.
-        tolerance_s (float): The allowed tolerance in seconds.
-        raise_value_error (bool): If True, raises an error on failure.
-
-    Returns:
-        bool: True if all deltas are valid, False otherwise.
-
-    Raises:
-        ValueError: If any delta is outside the tolerance and `raise_value_error` is True.
-    """
-    outside_tolerance = {}
-    for key, delta_ts in delta_timestamps.items():
-        within_tolerance = [abs(ts * fps - round(ts * fps)) / fps <= tolerance_s for ts in delta_ts]
-        if not all(within_tolerance):
-            outside_tolerance[key] = [
-                ts for ts, is_within in zip(delta_ts, within_tolerance, strict=True) if not is_within
-            ]
-
-    if len(outside_tolerance) > 0:
-        if raise_value_error:
-            raise ValueError(
-                f"""
-                The following delta_timestamps are found outside of tolerance range.
-                Please make sure they are multiples of 1/{fps} +/- tolerance and adjust
-                their values accordingly.
-                \n{pformat(outside_tolerance)}
-                """
-            )
-        return False
-
-    return True
-
-
-def get_delta_indices(delta_timestamps: dict[str, list[float]], fps: int) -> dict[str, list[int]]:
-    """Convert delta timestamps in seconds to delta indices in frames.
-
-    Args:
-        delta_timestamps (dict): A dictionary of time deltas in seconds.
-        fps (int): The frames per second of the dataset.
-
-    Returns:
-        dict: A dictionary of frame delta indices.
-    """
-    delta_indices = {}
-    for key, delta_ts in delta_timestamps.items():
-        delta_indices[key] = [round(d * fps) for d in delta_ts]
-
-    return delta_indices
-
-
-def validate_frame(frame: dict, features: dict) -> None:
-    expected_features = set(features) - set(DEFAULT_FEATURES)
-    actual_features = set(frame)
-
-    # task is a special required field that's not part of regular features
-    if "task" not in actual_features:
-        raise ValueError("Feature mismatch in `frame` dictionary:\nMissing features: {'task'}\n")
-
-    # Remove task from actual_features for regular feature validation
-    actual_features_for_validation = actual_features - {"task"}
-
-    error_message = validate_features_presence(actual_features_for_validation, expected_features)
-
-    common_features = actual_features_for_validation & expected_features
-    for name in common_features:
-        error_message += validate_feature_dtype_and_shape(name, features[name], frame[name])
-
-    if error_message:
-        raise ValueError(error_message)
-
-
-def validate_features_presence(actual_features: set[str], expected_features: set[str]) -> str:
-    """Check for missing or extra features in a frame.
-
-    Args:
-        actual_features (set[str]): The set of feature names present in the frame.
-        expected_features (set[str]): The set of feature names expected in the frame.
-
-    Returns:
-        str: An error message string if there's a mismatch, otherwise an empty string.
-    """
-    error_message = ""
-    missing_features = expected_features - actual_features
-    extra_features = actual_features - expected_features
-
-    if missing_features or extra_features:
-        error_message += "Feature mismatch in `frame` dictionary:\n"
-        if missing_features:
-            error_message += f"Missing features: {missing_features}\n"
-        if extra_features:
-            error_message += f"Extra features: {extra_features}\n"
-
-    return error_message
-
-
-def validate_feature_dtype_and_shape(
-    name: str, feature: dict, value: np.ndarray | PILImage.Image | str
-) -> str:
-    """Validate the dtype and shape of a single feature's value.
-
-    Args:
-        name (str): The name of the feature.
-        feature (dict): The feature specification from the LeRobot features dictionary.
-        value: The value of the feature to validate.
-
-    Returns:
-        str: An error message if validation fails, otherwise an empty string.
-
-    Raises:
-        NotImplementedError: If the feature dtype is not supported for validation.
-    """
-    expected_dtype = feature["dtype"]
-    expected_shape = feature["shape"]
-    if is_valid_numpy_dtype_string(expected_dtype):
-        return validate_feature_numpy_array(name, expected_dtype, expected_shape, value)
-    elif expected_dtype in ["image", "video"]:
-        return validate_feature_image_or_video(name, expected_shape, value)
-    elif expected_dtype == "string":
-        return validate_feature_string(name, value)
-    else:
-        raise NotImplementedError(f"The feature dtype '{expected_dtype}' is not implemented yet.")
-
-
-def validate_feature_numpy_array(
-    name: str, expected_dtype: str, expected_shape: list[int], value: np.ndarray
-) -> str:
-    """Validate a feature that is expected to be a numpy array.
-
-    Args:
-        name (str): The name of the feature.
-        expected_dtype (str): The expected numpy dtype as a string.
-        expected_shape (list[int]): The expected shape.
-        value (np.ndarray): The numpy array to validate.
-
-    Returns:
-        str: An error message if validation fails, otherwise an empty string.
-    """
-    error_message = ""
-    if isinstance(value, np.ndarray):
-        actual_dtype = value.dtype
-        actual_shape = value.shape
-
-        if actual_dtype != np.dtype(expected_dtype):
-            error_message += f"The feature '{name}' of dtype '{actual_dtype}' is not of the expected dtype '{expected_dtype}'.\n"
-
-        if actual_shape != expected_shape:
-            error_message += f"The feature '{name}' of shape '{actual_shape}' does not have the expected shape '{expected_shape}'.\n"
-    else:
-        error_message += f"The feature '{name}' is not a 'np.ndarray'. Expected type is '{expected_dtype}', but type '{type(value)}' provided instead.\n"
-
-    return error_message
-
-
-def validate_feature_image_or_video(
-    name: str, expected_shape: list[str], value: np.ndarray | PILImage.Image
-) -> str:
-    """Validate a feature that is expected to be an image or video frame.
-
-    Accepts `np.ndarray` (channel-first or channel-last) or `PIL.Image.Image`.
-
-    Args:
-        name (str): The name of the feature.
-        expected_shape (list[str]): The expected shape (C, H, W).
-        value: The image data to validate.
-
-    Returns:
-        str: An error message if validation fails, otherwise an empty string.
-    """
-    # Note: The check of pixels range ([0,1] for float and [0,255] for uint8) is done by the image writer threads.
-    error_message = ""
-    if isinstance(value, np.ndarray):
-        actual_shape = value.shape
-        c, h, w = expected_shape
-        if len(actual_shape) != 3 or (actual_shape != (c, h, w) and actual_shape != (h, w, c)):
-            error_message += f"The feature '{name}' of shape '{actual_shape}' does not have the expected shape '{(c, h, w)}' or '{(h, w, c)}'.\n"
-    elif isinstance(value, PILImage.Image):
-        pass
-    else:
-        error_message += f"The feature '{name}' is expected to be of type 'PIL.Image' or 'np.ndarray' channel first or channel last, but type '{type(value)}' provided instead.\n"
-
-    return error_message
-
-
-def validate_feature_string(name: str, value: str) -> str:
-    """Validate a feature that is expected to be a string.
-
-    Args:
-        name (str): The name of the feature.
-        value (str): The value to validate.
-
-    Returns:
-        str: An error message if validation fails, otherwise an empty string.
-    """
-    if not isinstance(value, str):
-        return f"The feature '{name}' is expected to be of type 'str', but type '{type(value)}' provided instead.\n"
-    return ""
-
-
-def validate_episode_buffer(episode_buffer: dict, total_episodes: int, features: dict) -> None:
-    """Validate the episode buffer before it's written to disk.
-
-    Ensures the buffer has the required keys, contains at least one frame, and
-    has features consistent with the dataset's specification.
-
-    Args:
-        episode_buffer (dict): The buffer containing data for a single episode.
-        total_episodes (int): The current total number of episodes in the dataset.
-        features (dict): The LeRobot features dictionary for the dataset.
-
-    Raises:
-        ValueError: If the buffer is invalid.
-        NotImplementedError: If the episode index is manually set and doesn't match.
-    """
-    if "size" not in episode_buffer:
-        raise ValueError("size key not found in episode_buffer")
-
-    if "task" not in episode_buffer:
-        raise ValueError("task key not found in episode_buffer")
-
-    if episode_buffer["episode_index"] != total_episodes:
-        # TODO(aliberts): Add option to use existing episode_index
-        raise NotImplementedError(
-            "You might have manually provided the episode_buffer with an episode_index that doesn't "
-            "match the total number of episodes already in the dataset. This is not supported for now."
-        )
-
-    if episode_buffer["size"] == 0:
-        raise ValueError("You must add one or several frames with `add_frame` before calling `add_episode`.")
-
-    buffer_keys = set(episode_buffer.keys()) - {"task", "size"}
-    if not buffer_keys == set(features):
-        raise ValueError(
-            f"Features from `episode_buffer` don't match the ones in `features`."
-            f"In episode_buffer not in features: {buffer_keys - set(features)}"
-            f"In features not in episode_buffer: {set(features) - buffer_keys}"
-        )
@@ -13,7 +13,6 @@
 # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 # See the License for the specific language governing permissions and
 # limitations under the License.
-import logging
 import multiprocessing
 import queue
 import threading
@@ -23,8 +22,6 @@ import numpy as np
 import PIL.Image
 import torch

-logger = logging.getLogger(__name__)
-

 def safe_stop_image_writer(func):
    def wrapper(*args, **kwargs):
@@ -34,7 +31,7 @@ def safe_stop_image_writer(func):
            dataset = kwargs.get("dataset")
            image_writer = getattr(dataset, "image_writer", None) if dataset else None
            if image_writer is not None:
-                logger.warning("Waiting for image writer to terminate...")
+                print("Waiting for image writer to terminate...")
                image_writer.stop()
            raise e

@@ -92,7 +89,8 @@ def write_image(image: np.ndarray | PIL.Image.Image, fpath: Path, compress_level
            PIL.Image.Image object.

    Side Effects:
-        Logs an error message if the image writing process fails for any reason.
+        Prints an error message to the console if the image writing process
+        fails for any reason.
    """
    try:
        if isinstance(image, np.ndarray):
@@ -103,7 +101,7 @@ def write_image(image: np.ndarray | PIL.Image.Image, fpath: Path, compress_level
            raise TypeError(f"Unsupported image type: {type(image)}")
        img.save(fpath, compress_level=compress_level)
    except Exception as e:
-        logger.error("Error writing image %s: %s", fpath, e)
+        print(f"Error writing image {fpath}: {e}")


 def worker_thread_loop(queue: queue.Queue):
@@ -1,342 +0,0 @@
-#!/usr/bin/env python
-
-# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-#     http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-import json
-from pathlib import Path
-from typing import Any
-
-import datasets
-import numpy as np
-import pandas
-import pandas as pd
-import pyarrow.dataset as pa_ds
-import pyarrow.parquet as pq
-import torch
-from datasets import Dataset
-from datasets.table import embed_table_storage
-from PIL import Image as PILImage
-from torchvision import transforms
-
-from lerobot.datasets.utils import (
-    DEFAULT_DATA_FILE_SIZE_IN_MB,
-    DEFAULT_EPISODES_PATH,
-    DEFAULT_SUBTASKS_PATH,
-    DEFAULT_TASKS_PATH,
-    EPISODES_DIR,
-    INFO_PATH,
-    STATS_PATH,
-    flatten_dict,
-    serialize_dict,
-    unflatten_dict,
-)
-from lerobot.utils.utils import SuppressProgressBars
-
-
-def get_parquet_file_size_in_mb(parquet_path: str | Path) -> float:
-    metadata = pq.read_metadata(parquet_path)
-    total_uncompressed_size = 0
-    for row_group in range(metadata.num_row_groups):
-        rg_metadata = metadata.row_group(row_group)
-        for column in range(rg_metadata.num_columns):
-            col_metadata = rg_metadata.column(column)
-            total_uncompressed_size += col_metadata.total_uncompressed_size
-    return total_uncompressed_size / (1024**2)
-
-
-def get_hf_dataset_size_in_mb(hf_ds: Dataset) -> int:
-    return hf_ds.data.nbytes // (1024**2)
-
-
-def load_nested_dataset(
-    pq_dir: Path, features: datasets.Features | None = None, episodes: list[int] | None = None
-) -> Dataset:
-    """Find parquet files in provided directory {pq_dir}/chunk-xxx/file-xxx.parquet
-    Convert parquet files to pyarrow memory mapped in a cache folder for efficient RAM usage
-    Concatenate all pyarrow references to return HF Dataset format
-
-    Args:
-        pq_dir: Directory containing parquet files
-        features: Optional features schema to ensure consistent loading of complex types like images
-        episodes: Optional list of episode indices to filter. Uses PyArrow predicate pushdown for efficiency.
-    """
-    paths = sorted(pq_dir.glob("*/*.parquet"))
-    if len(paths) == 0:
-        raise FileNotFoundError(f"Provided directory does not contain any parquet file: {pq_dir}")
-
-    with SuppressProgressBars():
-        # We use .from_parquet() memory-mapped loading for efficiency
-        filters = pa_ds.field("episode_index").isin(episodes) if episodes is not None else None
-        return Dataset.from_parquet([str(path) for path in paths], filters=filters, features=features)
-
-
-def get_parquet_num_frames(parquet_path: str | Path) -> int:
-    metadata = pq.read_metadata(parquet_path)
-    return metadata.num_rows
-
-
-def get_file_size_in_mb(file_path: Path) -> float:
-    """Get file size on disk in megabytes.
-
-    Args:
-        file_path (Path): Path to the file.
-    """
-    file_size_bytes = file_path.stat().st_size
-    return file_size_bytes / (1024**2)
-
-
-def embed_images(dataset: datasets.Dataset) -> datasets.Dataset:
-    """Embed image bytes into the dataset table before saving to Parquet.
-
-    This function prepares a Hugging Face dataset for serialization by converting
-    image objects into an embedded format that can be stored in Arrow/Parquet.
-
-    Args:
-        dataset (datasets.Dataset): The input dataset, possibly containing image features.
-
-    Returns:
-        datasets.Dataset: The dataset with images embedded in the table storage.
-    """
-    # Embed image bytes into the table before saving to parquet
-    format = dataset.format
-    dataset = dataset.with_format("arrow")
-    dataset = dataset.map(embed_table_storage, batched=False)
-    dataset = dataset.with_format(**format)
-    return dataset
-
-
-def load_json(fpath: Path) -> Any:
-    """Load data from a JSON file.
-
-    Args:
-        fpath (Path): Path to the JSON file.
-
-    Returns:
-        Any: The data loaded from the JSON file.
-    """
-    with open(fpath) as f:
-        return json.load(f)
-
-
-def write_json(data: dict, fpath: Path) -> None:
-    """Write data to a JSON file.
-
-    Creates parent directories if they don't exist.
-
-    Args:
-        data (dict): The dictionary to write.
-        fpath (Path): The path to the output JSON file.
-    """
-    fpath.parent.mkdir(exist_ok=True, parents=True)
-    with open(fpath, "w") as f:
-        json.dump(data, f, indent=4, ensure_ascii=False)
-
-
-def write_info(info: dict, local_dir: Path) -> None:
-    write_json(info, local_dir / INFO_PATH)
-
-
-def load_info(local_dir: Path) -> dict:
-    """Load dataset info metadata from its standard file path.
-
-    Also converts shape lists to tuples for consistency.
-
-    Args:
-        local_dir (Path): The root directory of the dataset.
-
-    Returns:
-        dict: The dataset information dictionary.
-    """
-    info = load_json(local_dir / INFO_PATH)
-    for ft in info["features"].values():
-        ft["shape"] = tuple(ft["shape"])
-    return info
-
-
-def write_stats(stats: dict, local_dir: Path) -> None:
-    """Serialize and write dataset statistics to their standard file path.
-
-    Args:
-        stats (dict): The statistics dictionary (can contain tensors/numpy arrays).
-        local_dir (Path): The root directory of the dataset.
-    """
-    serialized_stats = serialize_dict(stats)
-    write_json(serialized_stats, local_dir / STATS_PATH)
-
-
-def cast_stats_to_numpy(stats: dict) -> dict[str, dict[str, np.ndarray]]:
-    """Recursively cast numerical values in a stats dictionary to numpy arrays.
-
-    Args:
-        stats (dict): The statistics dictionary.
-
-    Returns:
-        dict: The statistics dictionary with values cast to numpy arrays.
-    """
-    stats = {key: np.array(value) for key, value in flatten_dict(stats).items()}
-    return unflatten_dict(stats)
-
-
-def load_stats(local_dir: Path) -> dict[str, dict[str, np.ndarray]] | None:
-    """Load dataset statistics and cast numerical values to numpy arrays.
-
-    Returns None if the stats file doesn't exist.
-
-    Args:
-        local_dir (Path): The root directory of the dataset.
-
-    Returns:
-        A dictionary of statistics or None if the file is not found.
-    """
-    if not (local_dir / STATS_PATH).exists():
-        return None
-    stats = load_json(local_dir / STATS_PATH)
-    return cast_stats_to_numpy(stats)
-
-
-def write_tasks(tasks: pandas.DataFrame, local_dir: Path) -> None:
-    path = local_dir / DEFAULT_TASKS_PATH
-    path.parent.mkdir(parents=True, exist_ok=True)
-    tasks.to_parquet(path)
-
-
-def load_tasks(local_dir: Path) -> pandas.DataFrame:
-    tasks = pd.read_parquet(local_dir / DEFAULT_TASKS_PATH)
-    tasks.index.name = "task"
-    return tasks
-
-
-def load_subtasks(local_dir: Path) -> pandas.DataFrame | None:
-    """Load subtasks from subtasks.parquet if it exists."""
-    subtasks_path = local_dir / DEFAULT_SUBTASKS_PATH
-    if subtasks_path.exists():
-        return pd.read_parquet(subtasks_path)
-    return None
-
-
-def write_episodes(episodes: Dataset, local_dir: Path) -> None:
-    """Write episode metadata to a parquet file in the LeRobot v3.0 format.
-    This function writes episode-level metadata to a single parquet file.
-    Used primarily during dataset conversion (v2.1 → v3.0) and in test fixtures.
-
-    Args:
-        episodes: HuggingFace Dataset containing episode metadata
-        local_dir: Root directory where the dataset will be stored
-    """
-    episode_size_mb = get_hf_dataset_size_in_mb(episodes)
-    if episode_size_mb > DEFAULT_DATA_FILE_SIZE_IN_MB:
-        raise NotImplementedError(
-            f"Episodes dataset is too large ({episode_size_mb} MB) to write to a single file. "
-            f"The current limit is {DEFAULT_DATA_FILE_SIZE_IN_MB} MB. "
-            "This function only supports single-file episode metadata. "
-        )
-
-    fpath = local_dir / DEFAULT_EPISODES_PATH.format(chunk_index=0, file_index=0)
-    fpath.parent.mkdir(parents=True, exist_ok=True)
-    episodes.to_parquet(fpath)
-
-
-def load_episodes(local_dir: Path) -> datasets.Dataset:
-    episodes = load_nested_dataset(local_dir / EPISODES_DIR)
-    # Select episode features/columns containing references to episode data and videos
-    # (e.g. tasks, dataset_from_index, dataset_to_index, data/chunk_index, data/file_index, etc.)
-    # This is to speedup access to these data, instead of having to load episode stats.
-    episodes = episodes.select_columns([key for key in episodes.features if not key.startswith("stats/")])
-    return episodes
-
-
-def load_image_as_numpy(
-    fpath: str | Path, dtype: np.dtype = np.float32, channel_first: bool = True
-) -> np.ndarray:
-    """Load an image from a file into a numpy array.
-
-    Args:
-        fpath (str | Path): Path to the image file.
-        dtype (np.dtype): The desired data type of the output array. If floating,
-            pixels are scaled to [0, 1].
-        channel_first (bool): If True, converts the image to (C, H, W) format.
-            Otherwise, it remains in (H, W, C) format.
-
-    Returns:
-        np.ndarray: The image as a numpy array.
-    """
-    img = PILImage.open(fpath).convert("RGB")
-    img_array = np.array(img, dtype=dtype)
-    if channel_first:  # (H, W, C) -> (C, H, W)
-        img_array = np.transpose(img_array, (2, 0, 1))
-    if np.issubdtype(dtype, np.floating):
-        img_array /= 255.0
-    return img_array
-
-
-def hf_transform_to_torch(items_dict: dict[str, list[Any]]) -> dict[str, list[torch.Tensor | str]]:
-    """Convert a batch from a Hugging Face dataset to torch tensors.
-
-    This transform function converts items from Hugging Face dataset format (pyarrow)
-    to torch tensors. Importantly, images are converted from PIL objects (H, W, C, uint8)
-    to a torch image representation (C, H, W, float32) in the range [0, 1]. Other
-    types are converted to torch.tensor.
-
-    Args:
-        items_dict (dict): A dictionary representing a batch of data from a
-            Hugging Face dataset.
-
-    Returns:
-        dict: The batch with items converted to torch tensors.
-    """
-    for key in items_dict:
-        first_item = items_dict[key][0]
-        if isinstance(first_item, PILImage.Image):
-            to_tensor = transforms.ToTensor()
-            items_dict[key] = [to_tensor(img) for img in items_dict[key]]
-        elif first_item is None:
-            pass
-        else:
-            items_dict[key] = [x if isinstance(x, str) else torch.tensor(x) for x in items_dict[key]]
-    return items_dict
-
-
-def to_parquet_with_hf_images(
-    df: pandas.DataFrame, path: Path, features: datasets.Features | None = None
-) -> None:
-    """This function correctly writes to parquet a panda DataFrame that contains images encoded by HF dataset.
-    This way, it can be loaded by HF dataset and correctly formatted images are returned.
-
-    Args:
-        df: DataFrame to write to parquet.
-        path: Path to write the parquet file.
-        features: Optional HuggingFace Features schema. If provided, ensures image columns
-                  are properly typed as Image() in the parquet schema.
-    """
-    # TODO(qlhoest): replace this weird synthax by `df.to_parquet(path)` only
-    ds = datasets.Dataset.from_dict(df.to_dict(orient="list"), features=features)
-    ds.to_parquet(path)
-
-
-def item_to_torch(item: dict) -> dict:
-    """Convert all items in a dictionary to PyTorch tensors where appropriate.
-
-    This function is used to convert an item from a streaming dataset to PyTorch tensors.
-
-    Args:
-        item (dict): Dictionary of items from a dataset.
-
-    Returns:
-        dict: Dictionary with all tensor-like items converted to torch.Tensor.
-    """
-    for key, val in item.items():
-        if isinstance(val, (np.ndarray | list)) and key not in ["task"]:
-            # Convert numpy arrays and lists to torch tensors
-            item[key] = torch.tensor(val)
-    return item
@@ -1,210 +0,0 @@
-#!/usr/bin/env python
-
-# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-#     http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-import logging
-from collections.abc import Callable
-from pathlib import Path
-
-import datasets
-import torch
-import torch.utils
-
-from lerobot.datasets.compute_stats import aggregate_stats
-from lerobot.datasets.lerobot_dataset import LeRobotDataset
-from lerobot.datasets.video_utils import VideoFrame
-from lerobot.utils.constants import HF_LEROBOT_HOME
-
-logger = logging.getLogger(__name__)
-
-
-class MultiLeRobotDataset(torch.utils.data.Dataset):
-    """A dataset consisting of multiple underlying `LeRobotDataset`s.
-
-    The underlying `LeRobotDataset`s are effectively concatenated, and this class adopts much of the API
-    structure of `LeRobotDataset`.
-    """
-
-    def __init__(
-        self,
-        repo_ids: list[str],
-        root: str | Path | None = None,
-        episodes: dict | None = None,
-        image_transforms: Callable | None = None,
-        delta_timestamps: dict[str, list[float]] | None = None,
-        tolerances_s: dict | None = None,
-        download_videos: bool = True,
-        video_backend: str | None = None,
-    ):
-        super().__init__()
-        self.repo_ids = repo_ids
-        self.root = Path(root) if root else HF_LEROBOT_HOME
-        self.tolerances_s = tolerances_s if tolerances_s else dict.fromkeys(repo_ids, 0.0001)
-        # Construct the underlying datasets passing everything but `transform` and `delta_timestamps` which
-        # are handled by this class.
-        self._datasets = [
-            LeRobotDataset(
-                repo_id,
-                root=self.root / repo_id,
-                episodes=episodes[repo_id] if episodes else None,
-                image_transforms=image_transforms,
-                delta_timestamps=delta_timestamps,
-                tolerance_s=self.tolerances_s[repo_id],
-                download_videos=download_videos,
-                video_backend=video_backend,
-            )
-            for repo_id in repo_ids
-        ]
-
-        # Disable any data keys that are not common across all of the datasets. Note: we may relax this
-        # restriction in future iterations of this class. For now, this is necessary at least for being able
-        # to use PyTorch's default DataLoader collate function.
-        self.disabled_features = set()
-        intersection_features = set(self._datasets[0].features)
-        for ds in self._datasets:
-            intersection_features.intersection_update(ds.features)
-        if len(intersection_features) == 0:
-            raise RuntimeError(
-                "Multiple datasets were provided but they had no keys common to all of them. "
-                "The multi-dataset functionality currently only keeps common keys."
-            )
-        for repo_id, ds in zip(self.repo_ids, self._datasets, strict=True):
-            extra_keys = set(ds.features).difference(intersection_features)
-            if extra_keys:
-                logger.warning(
-                    f"keys {extra_keys} of {repo_id} were disabled as they are not contained in all the "
-                    "other datasets."
-                )
-                self.disabled_features.update(extra_keys)
-
-        self.image_transforms = image_transforms
-        self.delta_timestamps = delta_timestamps
-        # TODO(rcadene, aliberts): We should not perform this aggregation for datasets
-        # with multiple robots of different ranges. Instead we should have one normalization
-        # per robot.
-        self.stats = aggregate_stats([dataset.meta.stats for dataset in self._datasets])
-
-    @property
-    def repo_id_to_index(self):
-        """Return a mapping from dataset repo_id to a dataset index automatically created by this class.
-
-        This index is incorporated as a data key in the dictionary returned by `__getitem__`.
-        """
-        return {repo_id: i for i, repo_id in enumerate(self.repo_ids)}
-
-    @property
-    def fps(self) -> int:
-        """Frames per second used during data collection.
-
-        NOTE: Fow now, this relies on a check in __init__ to make sure all sub-datasets have the same info.
-        """
-        return self._datasets[0].meta.info["fps"]
-
-    @property
-    def video(self) -> bool:
-        """Returns True if this dataset loads video frames from mp4 files.
-
-        Returns False if it only loads images from png files.
-
-        NOTE: Fow now, this relies on a check in __init__ to make sure all sub-datasets have the same info.
-        """
-        return self._datasets[0].meta.info.get("video", False)
-
-    @property
-    def features(self) -> datasets.Features:
-        features = {}
-        for dataset in self._datasets:
-            features.update({k: v for k, v in dataset.hf_features.items() if k not in self.disabled_features})
-        return features
-
-    @property
-    def camera_keys(self) -> list[str]:
-        """Keys to access image and video stream from cameras."""
-        keys = []
-        for key, feats in self.features.items():
-            if isinstance(feats, (datasets.Image | VideoFrame)):
-                keys.append(key)
-        return keys
-
-    @property
-    def video_frame_keys(self) -> list[str]:
-        """Keys to access video frames that requires to be decoded into images.
-
-        Note: It is empty if the dataset contains images only,
-        or equal to `self.cameras` if the dataset contains videos only,
-        or can even be a subset of `self.cameras` in a case of a mixed image/video dataset.
-        """
-        video_frame_keys = []
-        for key, feats in self.features.items():
-            if isinstance(feats, VideoFrame):
-                video_frame_keys.append(key)
-        return video_frame_keys
-
-    @property
-    def num_frames(self) -> int:
-        """Number of samples/frames."""
-        return sum(d.num_frames for d in self._datasets)
-
-    @property
-    def num_episodes(self) -> int:
-        """Number of episodes."""
-        return sum(d.num_episodes for d in self._datasets)
-
-    @property
-    def tolerance_s(self) -> float:
-        """Tolerance in seconds used to discard loaded frames when their timestamps
-        are not close enough from the requested frames. It is only used when `delta_timestamps`
-        is provided or when loading video frames from mp4 files.
-        """
-        # 1e-4 to account for possible numerical error
-        return 1 / self.fps - 1e-4
-
-    def __len__(self):
-        return self.num_frames
-
-    def __getitem__(self, idx: int) -> dict[str, torch.Tensor]:
-        if idx >= len(self):
-            raise IndexError(f"Index {idx} out of bounds.")
-        # Determine which dataset to get an item from based on the index.
-        start_idx = 0
-        dataset_idx = 0
-        for dataset in self._datasets:
-            if idx >= start_idx + dataset.num_frames:
-                start_idx += dataset.num_frames
-                dataset_idx += 1
-                continue
-            break
-        else:
-            raise AssertionError("We expect the loop to break out as long as the index is within bounds.")
-        item = self._datasets[dataset_idx][idx - start_idx]
-        item["dataset_index"] = torch.tensor(dataset_idx)
-        for data_key in self.disabled_features:
-            if data_key in item:
-                del item[data_key]
-
-        return item
-
-    def __repr__(self):
-        return (
-            f"{self.__class__.__name__}(\n"
-            f"  Repository IDs: '{self.repo_ids}',\n"
-            f"  Number of Samples: {self.num_frames},\n"
-            f"  Number of Episodes: {self.num_episodes},\n"
-            f"  Type: {'video (.mp4)' if self.video else 'image (.png)'},\n"
-            f"  Recorded Frames per Second: {self.fps},\n"
-            f"  Camera Keys: {self.camera_keys},\n"
-            f"  Video Frame Keys: {self.video_frame_keys if self.video else 'N/A'},\n"
-            f"  Transformations: {self.image_transforms},\n"
-            f")"
-        )
@@ -0,0 +1,382 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""An online buffer for the online training loop in train.py
+
+Note to maintainers: This duplicates some logic from LeRobotDataset and EpisodeAwareSampler. We should
+consider converging to one approach. Here we have opted to use numpy.memmap to back the data buffer. It's much
+faster than using HuggingFace Datasets as there's no conversion to an intermediate non-python object. Also it
+supports in-place slicing and mutation which is very handy for a dynamic buffer.
+"""
+
+import os
+from pathlib import Path
+from typing import Any
+
+import numpy as np
+import torch
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+
+def _make_memmap_safe(**kwargs) -> np.memmap:
+    """Make a numpy memmap with checks on available disk space first.
+
+    Expected kwargs are: "filename", "dtype" (must by np.dtype), "mode" and "shape"
+
+    For information on dtypes:
+    https://numpy.org/doc/stable/reference/arrays.dtypes.html#arrays-dtypes-constructing
+    """
+    if kwargs["mode"].startswith("w"):
+        required_space = kwargs["dtype"].itemsize * np.prod(kwargs["shape"])  # bytes
+        stats = os.statvfs(Path(kwargs["filename"]).parent)
+        available_space = stats.f_bavail * stats.f_frsize  # bytes
+        if required_space >= available_space * 0.8:
+            raise RuntimeError(
+                f"You're about to take up {required_space} of {available_space} bytes available."
+            )
+    return np.memmap(**kwargs)
+
+
+class OnlineBuffer(torch.utils.data.Dataset):
+    """FIFO data buffer for the online training loop in train.py.
+
+    Follows the protocol of LeRobotDataset as much as is required to have it be used by the online training
+    loop in the same way that a LeRobotDataset would be used.
+
+    The underlying data structure will have data inserted in a circular fashion. Always insert after the
+    last index, and when you reach the end, wrap around to the start.
+
+    The data is stored in a numpy memmap.
+    """
+
+    NEXT_INDEX_KEY = "_next_index"
+    OCCUPANCY_MASK_KEY = "_occupancy_mask"
+    INDEX_KEY = "index"
+    FRAME_INDEX_KEY = "frame_index"
+    EPISODE_INDEX_KEY = "episode_index"
+    TIMESTAMP_KEY = "timestamp"
+    IS_PAD_POSTFIX = "_is_pad"
+
+    def __init__(
+        self,
+        write_dir: str | Path,
+        data_spec: dict[str, Any] | None,
+        buffer_capacity: int | None,
+        fps: float | None = None,
+        delta_timestamps: dict[str, list[float]] | dict[str, np.ndarray] | None = None,
+    ):
+        """
+        The online buffer can be provided from scratch or you can load an existing online buffer by passing
+        a `write_dir` associated with an existing buffer.
+
+        Args:
+            write_dir: Where to keep the numpy memmap files. One memmap file will be stored for each data key.
+                Note that if the files already exist, they are opened in read-write mode (used for training
+                resumption.)
+            data_spec: A mapping from data key to data specification, like {data_key: {"shape": tuple[int],
+                "dtype": np.dtype}}. This should include all the data that you wish to record into the buffer,
+                but note that "index", "frame_index" and "episode_index" are already accounted for by this
+                class, so you don't need to include them.
+            buffer_capacity: How many frames should be stored in the buffer as a maximum. Be aware of your
+                system's available disk space when choosing this.
+            fps: Same as the fps concept in LeRobot dataset. Here it needs to be provided for the
+                 delta_timestamps logic. You can pass None if you are not using delta_timestamps.
+            delta_timestamps: Same as the delta_timestamps concept in LeRobotDataset. This is internally
+                converted to dict[str, np.ndarray] for optimization purposes.
+
+        """
+        self.set_delta_timestamps(delta_timestamps)
+        self._fps = fps
+        # Tolerance in seconds used to discard loaded frames when their timestamps are not close enough from
+        # the requested frames. It is only used when `delta_timestamps` is provided.
+        # minus 1e-4 to account for possible numerical error
+        self.tolerance_s = 1 / self.fps - 1e-4 if fps is not None else None
+        self._buffer_capacity = buffer_capacity
+        data_spec = self._make_data_spec(data_spec, buffer_capacity)
+        Path(write_dir).mkdir(parents=True, exist_ok=True)
+        self._data = {}
+        for k, v in data_spec.items():
+            self._data[k] = _make_memmap_safe(
+                filename=Path(write_dir) / k,
+                dtype=v["dtype"] if v is not None else None,
+                mode="r+" if (Path(write_dir) / k).exists() else "w+",
+                shape=tuple(v["shape"]) if v is not None else None,
+            )
+
+    @property
+    def delta_timestamps(self) -> dict[str, np.ndarray] | None:
+        return self._delta_timestamps
+
+    def set_delta_timestamps(self, value: dict[str, list[float]] | None):
+        """Set delta_timestamps converting the values to numpy arrays.
+
+        The conversion is for an optimization in the __getitem__. The loop is much slower if the arrays
+        need to be converted into numpy arrays.
+        """
+        if value is not None:
+            self._delta_timestamps = {k: np.array(v) for k, v in value.items()}
+        else:
+            self._delta_timestamps = None
+
+    def _make_data_spec(self, data_spec: dict[str, Any], buffer_capacity: int) -> dict[str, dict[str, Any]]:
+        """Makes the data spec for np.memmap."""
+        if any(k.startswith("_") for k in data_spec):
+            raise ValueError(
+                "data_spec keys should not start with '_'. This prefix is reserved for internal logic."
+            )
+        preset_keys = {
+            OnlineBuffer.INDEX_KEY,
+            OnlineBuffer.FRAME_INDEX_KEY,
+            OnlineBuffer.EPISODE_INDEX_KEY,
+            OnlineBuffer.TIMESTAMP_KEY,
+        }
+        if len(intersection := set(data_spec).intersection(preset_keys)) > 0:
+            raise ValueError(
+                f"data_spec should not contain any of {preset_keys} as these are handled internally. "
+                f"The provided data_spec has {intersection}."
+            )
+        complete_data_spec = {
+            # _next_index will be a pointer to the next index that we should start filling from when we add
+            # more data.
+            OnlineBuffer.NEXT_INDEX_KEY: {"dtype": np.dtype("int64"), "shape": ()},
+            # Since the memmap is initialized with all-zeros, this keeps track of which indices are occupied
+            # with real data rather than the dummy initialization.
+            OnlineBuffer.OCCUPANCY_MASK_KEY: {"dtype": np.dtype("?"), "shape": (buffer_capacity,)},
+            OnlineBuffer.INDEX_KEY: {"dtype": np.dtype("int64"), "shape": (buffer_capacity,)},
+            OnlineBuffer.FRAME_INDEX_KEY: {"dtype": np.dtype("int64"), "shape": (buffer_capacity,)},
+            OnlineBuffer.EPISODE_INDEX_KEY: {"dtype": np.dtype("int64"), "shape": (buffer_capacity,)},
+            OnlineBuffer.TIMESTAMP_KEY: {"dtype": np.dtype("float64"), "shape": (buffer_capacity,)},
+        }
+        for k, v in data_spec.items():
+            complete_data_spec[k] = {"dtype": v["dtype"], "shape": (buffer_capacity, *v["shape"])}
+        return complete_data_spec
+
+    def add_data(self, data: dict[str, np.ndarray]):
+        """Add new data to the buffer, which could potentially mean shifting old data out.
+
+        The new data should contain all the frames (in order) of any number of episodes. The indices should
+        start from 0 (note to the developer: this can easily be generalized). See the `rollout` and
+        `eval_policy` functions in `eval.py` for more information on how the data is constructed.
+
+        Shift the incoming data index and episode_index to continue on from the last frame. Note that this
+        will be done in place!
+        """
+        if len(missing_keys := (set(self.data_keys).difference(set(data)))) > 0:
+            raise ValueError(f"Missing data keys: {missing_keys}")
+        new_data_length = len(data[self.data_keys[0]])
+        if not all(len(data[k]) == new_data_length for k in self.data_keys):
+            raise ValueError("All data items should have the same length")
+
+        next_index = self._data[OnlineBuffer.NEXT_INDEX_KEY]
+
+        # Sanity check to make sure that the new data indices start from 0.
+        assert data[OnlineBuffer.EPISODE_INDEX_KEY][0].item() == 0
+        assert data[OnlineBuffer.INDEX_KEY][0].item() == 0
+
+        # Shift the incoming indices if necessary.
+        if self.num_frames > 0:
+            last_episode_index = self._data[OnlineBuffer.EPISODE_INDEX_KEY][next_index - 1]
+            last_data_index = self._data[OnlineBuffer.INDEX_KEY][next_index - 1]
+            data[OnlineBuffer.EPISODE_INDEX_KEY] += last_episode_index + 1
+            data[OnlineBuffer.INDEX_KEY] += last_data_index + 1
+
+        # Insert the new data starting from next_index. It may be necessary to wrap around to the start.
+        n_surplus = max(0, new_data_length - (self._buffer_capacity - next_index))
+        for k in self.data_keys:
+            if n_surplus == 0:
+                slc = slice(next_index, next_index + new_data_length)
+                self._data[k][slc] = data[k]
+                self._data[OnlineBuffer.OCCUPANCY_MASK_KEY][slc] = True
+            else:
+                self._data[k][next_index:] = data[k][:-n_surplus]
+                self._data[OnlineBuffer.OCCUPANCY_MASK_KEY][next_index:] = True
+                self._data[k][:n_surplus] = data[k][-n_surplus:]
+        if n_surplus == 0:
+            self._data[OnlineBuffer.NEXT_INDEX_KEY] = next_index + new_data_length
+        else:
+            self._data[OnlineBuffer.NEXT_INDEX_KEY] = n_surplus
+
+    @property
+    def data_keys(self) -> list[str]:
+        keys = set(self._data)
+        keys.remove(OnlineBuffer.OCCUPANCY_MASK_KEY)
+        keys.remove(OnlineBuffer.NEXT_INDEX_KEY)
+        return sorted(keys)
+
+    @property
+    def fps(self) -> float | None:
+        return self._fps
+
+    @property
+    def num_episodes(self) -> int:
+        return len(
+            np.unique(self._data[OnlineBuffer.EPISODE_INDEX_KEY][self._data[OnlineBuffer.OCCUPANCY_MASK_KEY]])
+        )
+
+    @property
+    def num_frames(self) -> int:
+        return np.count_nonzero(self._data[OnlineBuffer.OCCUPANCY_MASK_KEY])
+
+    def __len__(self):
+        return self.num_frames
+
+    def _item_to_tensors(self, item: dict) -> dict:
+        item_ = {}
+        for k, v in item.items():
+            if isinstance(v, torch.Tensor):
+                item_[k] = v
+            elif isinstance(v, np.ndarray):
+                item_[k] = torch.from_numpy(v)
+            else:
+                item_[k] = torch.tensor(v)
+        return item_
+
+    def __getitem__(self, idx: int) -> dict[str, torch.Tensor]:
+        if idx >= len(self) or idx < -len(self):
+            raise IndexError
+
+        item = {k: v[idx] for k, v in self._data.items() if not k.startswith("_")}
+
+        if self.delta_timestamps is None:
+            return self._item_to_tensors(item)
+
+        episode_index = item[OnlineBuffer.EPISODE_INDEX_KEY]
+        current_ts = item[OnlineBuffer.TIMESTAMP_KEY]
+        episode_data_indices = np.where(
+            np.bitwise_and(
+                self._data[OnlineBuffer.EPISODE_INDEX_KEY] == episode_index,
+                self._data[OnlineBuffer.OCCUPANCY_MASK_KEY],
+            )
+        )[0]
+        episode_timestamps = self._data[OnlineBuffer.TIMESTAMP_KEY][episode_data_indices]
+
+        for data_key in self.delta_timestamps:
+            # Note: The logic in this loop is copied from `load_previous_and_future_frames`.
+            # Get timestamps used as query to retrieve data of previous/future frames.
+            query_ts = current_ts + self.delta_timestamps[data_key]
+
+            # Compute distances between each query timestamp and all timestamps of all the frames belonging to
+            # the episode.
+            dist = np.abs(query_ts[:, None] - episode_timestamps[None, :])
+            argmin_ = np.argmin(dist, axis=1)
+            min_ = dist[np.arange(dist.shape[0]), argmin_]
+
+            is_pad = min_ > self.tolerance_s
+
+            # Check violated query timestamps are all outside the episode range.
+            assert (
+                (query_ts[is_pad] < episode_timestamps[0]) | (episode_timestamps[-1] < query_ts[is_pad])
+            ).all(), (
+                f"One or several timestamps unexpectedly violate the tolerance ({min_} > {self.tolerance_s=}"
+                ") inside the episode range."
+            )
+
+            # Load frames for this data key.
+            item[data_key] = self._data[data_key][episode_data_indices[argmin_]]
+
+            item[f"{data_key}{OnlineBuffer.IS_PAD_POSTFIX}"] = is_pad
+
+        return self._item_to_tensors(item)
+
+    def get_data_by_key(self, key: str) -> torch.Tensor:
+        """Returns all data for a given data key as a Tensor."""
+        return torch.from_numpy(self._data[key][self._data[OnlineBuffer.OCCUPANCY_MASK_KEY]])
+
+
+def compute_sampler_weights(
+    offline_dataset: LeRobotDataset,
+    offline_drop_n_last_frames: int = 0,
+    online_dataset: OnlineBuffer | None = None,
+    online_sampling_ratio: float | None = None,
+    online_drop_n_last_frames: int = 0,
+) -> torch.Tensor:
+    """Compute the sampling weights for the online training dataloader in train.py.
+
+    Args:
+        offline_dataset: The LeRobotDataset used for offline pre-training.
+        online_drop_n_last_frames: Number of frames to drop from the end of each offline dataset episode.
+        online_dataset: The OnlineBuffer used in online training.
+        online_sampling_ratio: The proportion of data that should be sampled from the online dataset. If an
+            online dataset is provided, this value must also be provided.
+        online_drop_n_first_frames: See `offline_drop_n_last_frames`. This is the same, but for the online
+            dataset.
+    Returns:
+        Tensor of weights for [offline_dataset; online_dataset], normalized to 1.
+
+    Notes to maintainers:
+        - This duplicates some logic from EpisodeAwareSampler. We should consider converging to one approach.
+        - When used with `torch.utils.data.WeightedRandomSampler`, it could completely replace
+          `EpisodeAwareSampler` as the online dataset related arguments are optional. The only missing feature
+          is the ability to turn shuffling off.
+        - Options `drop_first_n_frames` and `episode_indices_to_use` can be added easily. They were not
+          included here to avoid adding complexity.
+    """
+    if len(offline_dataset) == 0 and (online_dataset is None or len(online_dataset) == 0):
+        raise ValueError("At least one of `offline_dataset` or `online_dataset` should be contain data.")
+    if (online_dataset is None) ^ (online_sampling_ratio is None):
+        raise ValueError(
+            "`online_dataset` and `online_sampling_ratio` must be provided together or not at all."
+        )
+    offline_sampling_ratio = 0 if online_sampling_ratio is None else 1 - online_sampling_ratio
+
+    weights = []
+
+    if len(offline_dataset) > 0:
+        offline_data_mask_indices = []
+        for start_index, end_index in zip(
+            offline_dataset.meta.episodes["dataset_from_index"],
+            offline_dataset.meta.episodes["dataset_to_index"],
+            strict=True,
+        ):
+            offline_data_mask_indices.extend(range(start_index, end_index - offline_drop_n_last_frames))
+        offline_data_mask = torch.zeros(len(offline_dataset), dtype=torch.bool)
+        offline_data_mask[torch.tensor(offline_data_mask_indices)] = True
+        weights.append(
+            torch.full(
+                size=(len(offline_dataset),),
+                fill_value=offline_sampling_ratio / offline_data_mask.sum(),
+            )
+            * offline_data_mask
+        )
+
+    if online_dataset is not None and len(online_dataset) > 0:
+        online_data_mask_indices = []
+        episode_indices = online_dataset.get_data_by_key("episode_index")
+        for episode_idx in torch.unique(episode_indices):
+            where_episode = torch.where(episode_indices == episode_idx)
+            start_index = where_episode[0][0]
+            end_index = where_episode[0][-1] + 1
+            online_data_mask_indices.extend(
+                range(start_index.item(), end_index.item() - online_drop_n_last_frames)
+            )
+        online_data_mask = torch.zeros(len(online_dataset), dtype=torch.bool)
+        online_data_mask[torch.tensor(online_data_mask_indices)] = True
+        weights.append(
+            torch.full(
+                size=(len(online_dataset),),
+                fill_value=online_sampling_ratio / online_data_mask.sum(),
+            )
+            * online_data_mask
+        )
+
+    weights = torch.cat(weights)
+
+    if weights.sum() == 0:
+        weights += 1 / len(weights)
+    else:
+        weights /= weights.sum()
+
+    return weights
@@ -17,9 +17,8 @@ from collections.abc import Sequence
 from typing import Any

 from lerobot.configs.types import PipelineFeatureType
-from lerobot.datasets.feature_utils import hw_to_dataset_features
-from lerobot.processor import DataProcessorPipeline
-from lerobot.types import RobotAction, RobotObservation
+from lerobot.datasets.utils import hw_to_dataset_features
+from lerobot.processor import DataProcessorPipeline, RobotAction, RobotObservation
 from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE, OBS_STR


@@ -44,11 +43,11 @@ def create_initial_features(
    return features


-# Helper to filter state/action keys based on compiled regex patterns.
-def should_keep(key: str, patterns: tuple[re.Pattern] | None) -> bool:
+# Helper to filter state/action keys based on regex patterns.
+def should_keep(key: str, patterns: tuple[str]) -> bool:
    if patterns is None:
        return True
-    return any(pat.search(key) for pat in patterns)
+    return any(re.search(pat, key) for pat in patterns)


 def strip_prefix(key: str, prefixes_to_strip: tuple[str]) -> str:
@@ -89,8 +88,6 @@ def aggregate_pipeline_dataset_features(
    Returns:
        A dictionary of features formatted for a Hugging Face LeRobot Dataset.
    """
-    compiled_patterns = tuple(re.compile(p) for p in patterns) if patterns is not None else None
-
    all_features = pipeline.transform_features(initial_features)

    # Intermediate storage for categorized and filtered features.
@@ -122,7 +119,7 @@ def aggregate_pipeline_dataset_features(
            # 2. Apply filtering rules.
            if is_image and not use_videos:
                continue
-            if not is_image and not should_keep(key, compiled_patterns):
+            if not is_image and not should_keep(key, patterns):
                continue

            # 3. Add the feature to the appropriate group with a clean name.
@@ -0,0 +1,73 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import datasets
+import torch
+
+
+# TODO(aliberts): remove
+def calculate_episode_data_index(hf_dataset: datasets.Dataset) -> dict[str, torch.Tensor]:
+    """
+    Calculate episode data index for the provided HuggingFace Dataset. Relies on episode_index column of hf_dataset.
+
+    Parameters:
+    - hf_dataset (datasets.Dataset): A HuggingFace dataset containing the episode index.
+
+    Returns:
+    - episode_data_index: A dictionary containing the data index for each episode. The dictionary has two keys:
+        - "from": A tensor containing the starting index of each episode.
+        - "to": A tensor containing the ending index of each episode.
+    """
+    episode_data_index = {"from": [], "to": []}
+
+    current_episode = None
+    """
+    The episode_index is a list of integers, each representing the episode index of the corresponding example.
+    For instance, the following is a valid episode_index:
+      [0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 2]
+
+    Below, we iterate through the episode_index and populate the episode_data_index dictionary with the starting and
+    ending index of each episode. For the episode_index above, the episode_data_index dictionary will look like this:
+        {
+            "from": [0, 3, 7],
+            "to": [3, 7, 12]
+        }
+    """
+    if len(hf_dataset) == 0:
+        episode_data_index = {
+            "from": torch.tensor([]),
+            "to": torch.tensor([]),
+        }
+        return episode_data_index
+    for idx, episode_idx in enumerate(hf_dataset["episode_index"]):
+        if episode_idx != current_episode:
+            # We encountered a new episode, so we append its starting location to the "from" list
+            episode_data_index["from"].append(idx)
+            # If this is not the first episode, we append the ending location of the previous episode to the "to" list
+            if current_episode is not None:
+                episode_data_index["to"].append(idx)
+            # Let's keep track of the current episode index
+            current_episode = episode_idx
+        else:
+            # We are still in the same episode, so there is nothing for us to do here
+            pass
+    # We have reached the end of the dataset, so we append the ending location of the last episode to the "to" list
+    episode_data_index["to"].append(idx + 1)
+
+    for k in ["from", "to"]:
+        episode_data_index[k] = torch.tensor(episode_data_index[k])
+
+    return episode_data_index
--- a/Show More
+++ b/Show More