refactor(dataset): split LeRobotDataset into DatasetReader & DatasetWriter (+ API cleanup) (#3180)

* refactor(dataset): split reader and writer * chore(dataset): remove proxys * refactor(dataset): better reader & writer encapsulation * refactor(datasets): clean API + reduce leaky implementations * refactor(dataset): API cleaning for writer, reader and meta * refactor(dataset): expose writer & reader + other minor improvements * refactor(dataset): improve teardown routine * refactor(dataset): add hf_dataset property at the facade level * chore(dataset): add init for datasset module * docs(dataset): add docstrings for public API of the dataset classes * tests(dataset): add tests for new classes * fix(dataset): remove circular dependecy
2026-07-22 17:32:07 +00:00 · 2026-03-26 19:09:25 +01:00
parent 017ff73fbf
commit 123495250b
28 changed files with 2742 additions and 1158 deletions
@@ -78,7 +78,7 @@ def replay(cfg: ReplayConfig):

    robot = make_robot_from_config(cfg.robot)
    dataset = LeRobotDataset(cfg.dataset.repo_id, root=cfg.dataset.root, episodes=[cfg.dataset.episode])
-    actions = dataset.hf_dataset.select_columns(ACTION)
+    actions = dataset.select_columns(ACTION)
    robot.connect()

    try:
@@ -88,9 +88,8 @@ def main():
    # The previous metadata class is contained in the 'meta' attribute of the dataset:
    print(dataset.meta)

-    # LeRobotDataset actually wraps an underlying Hugging Face dataset
-    # (see https://huggingface.co/docs/datasets for more information).
-    print(dataset.hf_dataset)
+    # You can inspect the dataset using its repr:
+    print(dataset)

    # LeRobot datasets also subclasses PyTorch datasets so you can do everything you know and love from working
    # with the latter, like iterating through the dataset.
@@ -35,9 +35,7 @@ def main():

    # Fetch the dataset to replay
    dataset = LeRobotDataset("<hf_username>/<dataset_repo_id>", episodes=[EPISODE_IDX])
-    # Filter dataset to only include frames from the specified episode since episodes are chunked in dataset V3.0
-    episode_frames = dataset.hf_dataset.filter(lambda x: x["episode_index"] == EPISODE_IDX)
-    actions = episode_frames.select_columns(ACTION)
+    actions = dataset.select_columns(ACTION)

    # Connect to the robot
    robot.connect()
@@ -48,7 +46,7 @@ def main():

        print("Starting replay loop...")
        log_say(f"Replaying episode {EPISODE_IDX}")
-        for idx in range(len(episode_frames)):
+        for idx in range(dataset.num_frames):
            t0 = time.perf_counter()

            # Get recorded action from dataset
@@ -67,9 +67,7 @@ def main():

    # Fetch the dataset to replay
    dataset = LeRobotDataset(HF_REPO_ID, episodes=[EPISODE_IDX])
-    # Filter dataset to only include frames from the specified episode since episodes are chunked in dataset V3.0
-    episode_frames = dataset.hf_dataset.filter(lambda x: x["episode_index"] == EPISODE_IDX)
-    actions = episode_frames.select_columns(ACTION)
+    actions = dataset.select_columns(ACTION)

    # Connect to the robot
    robot.connect()
@@ -80,7 +78,7 @@ def main():

        print("Starting replay loop...")
        log_say(f"Replaying episode {EPISODE_IDX}")
-        for idx in range(len(episode_frames)):
+        for idx in range(dataset.num_frames):
            t0 = time.perf_counter()

            # Get recorded action from dataset
@@ -68,9 +68,7 @@ def main():

    # Fetch the dataset to replay
    dataset = LeRobotDataset(HF_REPO_ID, episodes=[EPISODE_IDX])
-    # Filter dataset to only include frames from the specified episode since episodes are chunked in dataset V3.0
-    episode_frames = dataset.hf_dataset.filter(lambda x: x["episode_index"] == EPISODE_IDX)
-    actions = episode_frames.select_columns(ACTION)
+    actions = dataset.select_columns(ACTION)

    # Connect to the robot
    robot.connect()
@@ -81,7 +79,7 @@ def main():

        print("Starting replay loop...")
        log_say(f"Replaying episode {EPISODE_IDX}")
-        for idx in range(len(episode_frames)):
+        for idx in range(dataset.num_frames):
            t0 = time.perf_counter()

            # Get recorded action from dataset