mirror of
https://github.com/huggingface/lerobot.git
synced 2026-07-26 11:16:00 +00:00
Merge branch 'main' into feat/vlabench-benchmark
Made-with: Cursor
This commit is contained in:
@@ -633,6 +633,217 @@ jobs:
|
|||||||
path: /tmp/robocerebra-artifacts/metrics.json
|
path: /tmp/robocerebra-artifacts/metrics.json
|
||||||
if-no-files-found: warn
|
if-no-files-found: warn
|
||||||
|
|
||||||
|
# ── ROBOMME ───────────────────────────────────────────────────────────────
|
||||||
|
# Isolated image: mani-skill/SAPIEN/Vulkan chain with gymnasium and numpy
|
||||||
|
# overrides (robomme can't be a pyproject extra due to numpy<2 pin).
|
||||||
|
robomme-integration-test:
|
||||||
|
name: RoboMME — build image + 1-episode eval
|
||||||
|
runs-on:
|
||||||
|
group: aws-g6-4xlarge-plus
|
||||||
|
env:
|
||||||
|
HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
|
||||||
|
ROBOMME_POLICY: lerobot/smolvla_robomme
|
||||||
|
ROBOMME_TASKS: PickXtimes,BinFill,StopCube,MoveCube,InsertPeg,SwingXtimes,VideoUnmask,ButtonUnmask,PickHighlight,PatternLock
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||||
|
with:
|
||||||
|
persist-credentials: false
|
||||||
|
lfs: true
|
||||||
|
|
||||||
|
- name: Set up Docker Buildx
|
||||||
|
uses: docker/setup-buildx-action@v3 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
cache-binary: false
|
||||||
|
|
||||||
|
- name: Login to Docker Hub
|
||||||
|
if: ${{ env.DOCKERHUB_USERNAME != '' }}
|
||||||
|
uses: docker/login-action@v3 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
|
||||||
|
password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
|
||||||
|
env:
|
||||||
|
DOCKERHUB_USERNAME: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
|
||||||
|
|
||||||
|
- name: Build RoboMME benchmark image
|
||||||
|
uses: docker/build-push-action@v6 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
context: .
|
||||||
|
file: docker/Dockerfile.benchmark.robomme
|
||||||
|
push: false
|
||||||
|
load: true
|
||||||
|
tags: lerobot-benchmark-robomme:ci
|
||||||
|
|
||||||
|
- name: Run RoboMME smoke eval (10 tasks, 1 episode each)
|
||||||
|
if: env.HF_USER_TOKEN != ''
|
||||||
|
run: |
|
||||||
|
docker run --name robomme-eval --gpus all \
|
||||||
|
--shm-size=4g \
|
||||||
|
-e HF_HOME=/tmp/hf \
|
||||||
|
-e HF_USER_TOKEN="${HF_USER_TOKEN}" \
|
||||||
|
-e HF_HUB_DOWNLOAD_TIMEOUT=300 \
|
||||||
|
-e ROBOMME_POLICY="${ROBOMME_POLICY}" \
|
||||||
|
-e ROBOMME_TASKS="${ROBOMME_TASKS}" \
|
||||||
|
lerobot-benchmark-robomme:ci \
|
||||||
|
bash -c "
|
||||||
|
hf auth login --token \"\$HF_USER_TOKEN\" --add-to-git-credential 2>/dev/null || true
|
||||||
|
lerobot-eval \
|
||||||
|
--policy.path=\"\$ROBOMME_POLICY\" \
|
||||||
|
--env.type=robomme \
|
||||||
|
--env.task=\"\$ROBOMME_TASKS\" \
|
||||||
|
--env.dataset_split=test \
|
||||||
|
--env.task_ids=[0] \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.n_episodes=1 \
|
||||||
|
--eval.use_async_envs=false \
|
||||||
|
--policy.device=cuda \
|
||||||
|
'--rename_map={\"observation.images.image\": \"observation.images.camera1\", \"observation.images.wrist_image\": \"observation.images.camera2\"}' \
|
||||||
|
--policy.empty_cameras=3 \
|
||||||
|
--output_dir=/tmp/eval-artifacts
|
||||||
|
python scripts/ci/extract_task_descriptions.py \
|
||||||
|
--env robomme --task \"\$ROBOMME_TASKS\" \
|
||||||
|
--output /tmp/eval-artifacts/task_descriptions.json
|
||||||
|
"
|
||||||
|
|
||||||
|
- name: Copy RoboMME artifacts from container
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
mkdir -p /tmp/robomme-artifacts
|
||||||
|
docker cp robomme-eval:/tmp/eval-artifacts/. /tmp/robomme-artifacts/ 2>/dev/null || true
|
||||||
|
docker rm -f robomme-eval || true
|
||||||
|
|
||||||
|
- name: Parse RoboMME eval metrics
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
python3 scripts/ci/parse_eval_metrics.py \
|
||||||
|
--artifacts-dir /tmp/robomme-artifacts \
|
||||||
|
--env robomme \
|
||||||
|
--task "${ROBOMME_TASKS}" \
|
||||||
|
--policy "${ROBOMME_POLICY}"
|
||||||
|
|
||||||
|
- name: Upload RoboMME rollout video
|
||||||
|
if: always()
|
||||||
|
uses: actions/upload-artifact@v4 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
name: robomme-rollout-video
|
||||||
|
path: /tmp/robomme-artifacts/videos/
|
||||||
|
if-no-files-found: warn
|
||||||
|
|
||||||
|
- name: Upload RoboMME eval metrics
|
||||||
|
if: always()
|
||||||
|
uses: actions/upload-artifact@v4 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
name: robomme-metrics
|
||||||
|
path: /tmp/robomme-artifacts/metrics.json
|
||||||
|
if-no-files-found: warn
|
||||||
|
|
||||||
|
# ── LIBERO-plus ───────────────────────────────────────────────────────────
|
||||||
|
# Isolated image: LIBERO-plus fork cloned into /home/user_lerobot on top of
|
||||||
|
# huggingface/lerobot-gpu (see docker/Dockerfile.benchmark.libero_plus).
|
||||||
|
libero-plus-integration-test:
|
||||||
|
name: LIBERO-plus — build image + 1-episode eval
|
||||||
|
runs-on:
|
||||||
|
group: aws-g6-4xlarge-plus
|
||||||
|
env:
|
||||||
|
HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
|
||||||
|
LIBERO_PLUS_SUITE: libero_spatial
|
||||||
|
LIBERO_PLUS_POLICY: lerobot/smolvla_libero_plus
|
||||||
|
LIBERO_PLUS_TASK_IDS: "[0,100,260,500,1000,1500,2000,2400]"
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||||
|
with:
|
||||||
|
persist-credentials: false
|
||||||
|
lfs: true
|
||||||
|
|
||||||
|
- name: Set up Docker Buildx
|
||||||
|
uses: docker/setup-buildx-action@v3 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
cache-binary: false
|
||||||
|
|
||||||
|
- name: Login to Docker Hub
|
||||||
|
if: ${{ env.DOCKERHUB_USERNAME != '' }}
|
||||||
|
uses: docker/login-action@v3 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
|
||||||
|
password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
|
||||||
|
env:
|
||||||
|
DOCKERHUB_USERNAME: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
|
||||||
|
|
||||||
|
- name: Build LIBERO-plus benchmark image
|
||||||
|
uses: docker/build-push-action@v6 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
context: .
|
||||||
|
file: docker/Dockerfile.benchmark.libero_plus
|
||||||
|
push: false
|
||||||
|
load: true
|
||||||
|
tags: lerobot-benchmark-libero-plus:ci
|
||||||
|
cache-from: type=local,src=/tmp/.buildx-cache-libero-plus
|
||||||
|
cache-to: type=local,dest=/tmp/.buildx-cache-libero-plus,mode=max
|
||||||
|
|
||||||
|
- name: Run LIBERO-plus smoke eval (1 episode)
|
||||||
|
if: env.HF_USER_TOKEN != ''
|
||||||
|
run: |
|
||||||
|
docker run --name libero-plus-eval --gpus all \
|
||||||
|
--shm-size=4g \
|
||||||
|
-e HF_HOME=/tmp/hf \
|
||||||
|
-e HF_USER_TOKEN="${HF_USER_TOKEN}" \
|
||||||
|
-e HF_HUB_DOWNLOAD_TIMEOUT=300 \
|
||||||
|
-e LIBERO_PLUS_SUITE="${LIBERO_PLUS_SUITE}" \
|
||||||
|
-e LIBERO_PLUS_POLICY="${LIBERO_PLUS_POLICY}" \
|
||||||
|
-e LIBERO_PLUS_TASK_IDS="${LIBERO_PLUS_TASK_IDS}" \
|
||||||
|
lerobot-benchmark-libero-plus:ci \
|
||||||
|
bash -c "
|
||||||
|
hf auth login --token \"\$HF_USER_TOKEN\" --add-to-git-credential 2>/dev/null || true
|
||||||
|
lerobot-eval \
|
||||||
|
--policy.path=\"\$LIBERO_PLUS_POLICY\" \
|
||||||
|
--env.type=libero_plus \
|
||||||
|
--env.task=\"\$LIBERO_PLUS_SUITE\" \
|
||||||
|
--env.task_ids=\"\$LIBERO_PLUS_TASK_IDS\" \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.n_episodes=1 \
|
||||||
|
--eval.use_async_envs=false \
|
||||||
|
--policy.device=cuda \
|
||||||
|
'--env.camera_name_mapping={\"agentview_image\": \"camera1\", \"robot0_eye_in_hand_image\": \"camera2\"}' \
|
||||||
|
--policy.empty_cameras=1 \
|
||||||
|
--output_dir=/tmp/eval-artifacts
|
||||||
|
python scripts/ci/extract_task_descriptions.py \
|
||||||
|
--env libero_plus --task \"\$LIBERO_PLUS_SUITE\" \
|
||||||
|
--output /tmp/eval-artifacts/task_descriptions.json
|
||||||
|
"
|
||||||
|
|
||||||
|
- name: Copy LIBERO-plus artifacts from container
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
mkdir -p /tmp/libero-plus-artifacts
|
||||||
|
docker cp libero-plus-eval:/tmp/eval-artifacts/. /tmp/libero-plus-artifacts/ 2>/dev/null || true
|
||||||
|
docker rm -f libero-plus-eval || true
|
||||||
|
|
||||||
|
- name: Parse LIBERO-plus eval metrics
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
python3 scripts/ci/parse_eval_metrics.py \
|
||||||
|
--artifacts-dir /tmp/libero-plus-artifacts \
|
||||||
|
--env libero_plus \
|
||||||
|
--task "${LIBERO_PLUS_SUITE}" \
|
||||||
|
--policy "${LIBERO_PLUS_POLICY}"
|
||||||
|
|
||||||
|
- name: Upload LIBERO-plus rollout video
|
||||||
|
if: always()
|
||||||
|
uses: actions/upload-artifact@v4 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
name: libero-plus-rollout-video
|
||||||
|
path: /tmp/libero-plus-artifacts/videos/
|
||||||
|
if-no-files-found: warn
|
||||||
|
|
||||||
|
- name: Upload LIBERO-plus eval metrics
|
||||||
|
if: always()
|
||||||
|
uses: actions/upload-artifact@v4 # zizmor: ignore[unpinned-uses]
|
||||||
|
with:
|
||||||
|
name: libero-plus-metrics
|
||||||
|
path: /tmp/libero-plus-artifacts/metrics.json
|
||||||
|
if-no-files-found: warn
|
||||||
|
|
||||||
# ── VLABENCH ─────────────────────────────────────────────────────────────
|
# ── VLABENCH ─────────────────────────────────────────────────────────────
|
||||||
# Isolated image: lerobot[vlabench] only (VLABench, mujoco==3.2.2, dm-control chain)
|
# Isolated image: lerobot[vlabench] only (VLABench, mujoco==3.2.2, dm-control chain)
|
||||||
vlabench-integration-test:
|
vlabench-integration-test:
|
||||||
|
|||||||
@@ -0,0 +1,84 @@
|
|||||||
|
# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
# Benchmark image for LIBERO-plus integration tests.
|
||||||
|
# Extends the nightly GPU image (which has lerobot[all]) with the LIBERO-plus
|
||||||
|
# fork source + its 6.4 GB perturbation assets.
|
||||||
|
#
|
||||||
|
# Build: docker build -f docker/Dockerfile.benchmark.libero_plus -t lerobot-benchmark-libero-plus .
|
||||||
|
# Run: docker run --gpus all --rm lerobot-benchmark-libero-plus lerobot-eval ...
|
||||||
|
|
||||||
|
FROM huggingface/lerobot-gpu:latest
|
||||||
|
ENV MUJOCO_GL=egl
|
||||||
|
|
||||||
|
# unzip for the 6.4 GB assets.zip; the rest are LIBERO-plus build-time extras
|
||||||
|
# (wand / ImageMagick / fontconfig) not in the nightly base.
|
||||||
|
USER root
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install -y --no-install-recommends \
|
||||||
|
unzip libexpat1 libfontconfig1-dev libmagickwand-dev \
|
||||||
|
&& apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||||
|
USER user_lerobot
|
||||||
|
|
||||||
|
# robosuite==1.4.1 is mandatory (the fork uses `single_arm_env` removed in
|
||||||
|
# v1.5+). The rest are LIBERO-plus runtime deps pulled from its setup.py.
|
||||||
|
# We install these explicitly instead of via the [libero_plus] extra because
|
||||||
|
# the extra's `libero @ git+...` dep installs as a namespace package and then
|
||||||
|
# clone and PYTHONPATH-override it below.
|
||||||
|
RUN uv pip install --no-cache \
|
||||||
|
"robosuite==1.4.1" \
|
||||||
|
"bddl==1.0.1" \
|
||||||
|
"easydict==1.13" \
|
||||||
|
"mujoco==3.7.0" \
|
||||||
|
"matplotlib==3.10.8" \
|
||||||
|
"Wand==0.6.13" \
|
||||||
|
"scikit-image==0.25.2" \
|
||||||
|
"gym==0.26.2"
|
||||||
|
|
||||||
|
# Clone LIBERO-plus and make it importable as `libero`. The nightly base has
|
||||||
|
# hf-libero (10 tasks) preinstalled via lerobot[libero]; uninstall it so
|
||||||
|
# Python resolves `import libero` to the 2402-task LIBERO-plus module instead.
|
||||||
|
# Pinned to the current upstream main SHA so benchmark builds stay reproducible.
|
||||||
|
ARG LIBERO_PLUS_SHA=4976dc3
|
||||||
|
ENV LIBERO_PLUS_ROOT=/home/user_lerobot/libero-plus/libero/libero
|
||||||
|
RUN git clone https://github.com/sylvestf/LIBERO-plus.git /home/user_lerobot/libero-plus \
|
||||||
|
&& git -C /home/user_lerobot/libero-plus checkout ${LIBERO_PLUS_SHA} \
|
||||||
|
&& cd /home/user_lerobot/libero-plus && uv pip install --no-cache --no-deps -e "." \
|
||||||
|
&& (uv pip uninstall hf-libero 2>/dev/null || true)
|
||||||
|
ENV PYTHONPATH="/home/user_lerobot/libero-plus:${PYTHONPATH}"
|
||||||
|
|
||||||
|
# Perturbation textures/scenes: bddl_base_domain.py resolves XMLs via
|
||||||
|
# DIR_PATH/../assets (package-relative, ignoring ~/.libero/config.yaml). All
|
||||||
|
# 2402 tasks reference files that ship only in Sylvest/LIBERO-plus's
|
||||||
|
# assets.zip (6.4 GB) under a deep author-internal prefix — extract and
|
||||||
|
# flatten it under ${LIBERO_PLUS_ROOT}/assets.
|
||||||
|
RUN python -c "\
|
||||||
|
from huggingface_hub import hf_hub_download; \
|
||||||
|
hf_hub_download(repo_id='Sylvest/LIBERO-plus', repo_type='dataset', \
|
||||||
|
filename='assets.zip', local_dir='/tmp/libero-plus-dl')" \
|
||||||
|
&& unzip -q /tmp/libero-plus-dl/assets.zip -d /tmp/libero-plus-dl/extract \
|
||||||
|
&& ASSETS_DIR=$(find /tmp/libero-plus-dl/extract -type d -name assets | head -1) \
|
||||||
|
&& mv "${ASSETS_DIR}" ${LIBERO_PLUS_ROOT}/assets \
|
||||||
|
&& rm -rf /tmp/libero-plus-dl
|
||||||
|
|
||||||
|
# Point ~/.libero/config.yaml at the clone so LIBERO-plus's imports are
|
||||||
|
# non-interactive (it calls input() when the config is missing).
|
||||||
|
RUN mkdir -p /home/user_lerobot/.libero \
|
||||||
|
&& printf "assets: ${LIBERO_PLUS_ROOT}/assets\nbddl_files: ${LIBERO_PLUS_ROOT}/bddl_files\ndatasets: ${LIBERO_PLUS_ROOT}/../datasets\ninit_states: ${LIBERO_PLUS_ROOT}/init_files\n" \
|
||||||
|
> /home/user_lerobot/.libero/config.yaml
|
||||||
|
|
||||||
|
# Overlay the PR's source code on top of the nightly image.
|
||||||
|
COPY --chown=user_lerobot:user_lerobot . .
|
||||||
|
|
||||||
|
CMD ["/bin/bash"]
|
||||||
@@ -0,0 +1,56 @@
|
|||||||
|
# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
# Benchmark image for RoboMME integration tests.
|
||||||
|
# Extends the nightly GPU image (which has lerobot[all]) with Vulkan system
|
||||||
|
# libs for ManiSkill/SAPIEN and the robomme extra. robomme isn't in [all]
|
||||||
|
# because mani-skill hard-pins gymnasium==0.29.1 and numpy<2.0.0 which
|
||||||
|
# conflict with lerobot's defaults; both are safe at runtime:
|
||||||
|
# - gymnasium 0.29.x has the same 5-tuple step() API as 1.x (since 0.26)
|
||||||
|
# - numpy 1.26.4 is API-compatible with lerobot's actual usage.
|
||||||
|
#
|
||||||
|
# Build: docker build -f docker/Dockerfile.benchmark.robomme -t lerobot-benchmark-robomme .
|
||||||
|
# Run: docker run --gpus all --rm lerobot-benchmark-robomme lerobot-eval ...
|
||||||
|
|
||||||
|
FROM huggingface/lerobot-gpu:latest
|
||||||
|
|
||||||
|
# NVIDIA Container Toolkit: expose Vulkan driver capability for headless rendering.
|
||||||
|
ENV NVIDIA_DRIVER_CAPABILITIES=all \
|
||||||
|
VK_ICD_FILENAMES=/usr/share/vulkan/icd.d/nvidia_icd.json
|
||||||
|
|
||||||
|
# ManiSkill/SAPIEN's renderer needs Vulkan, which isn't in the base image.
|
||||||
|
USER root
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install -y --no-install-recommends \
|
||||||
|
libvulkan1 libvulkan-dev mesa-vulkan-drivers \
|
||||||
|
&& mkdir -p /usr/share/vulkan/icd.d \
|
||||||
|
&& echo '{"file_format_version":"1.0.0","ICD":{"library_path":"libGLX_nvidia.so.0","api_version":"1.3.0"}}' \
|
||||||
|
> /usr/share/vulkan/icd.d/nvidia_icd.json \
|
||||||
|
&& apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||||
|
USER user_lerobot
|
||||||
|
|
||||||
|
# Install smolvla + av-dep via the PR's pyproject, then layer robomme on top
|
||||||
|
# with gymnasium/numpy overrides. robomme isn't a pyproject extra because its
|
||||||
|
# mani-skill pin conflicts with lerobot's base numpy>=2 (see pyproject.toml).
|
||||||
|
COPY --chown=user_lerobot:user_lerobot setup.py pyproject.toml uv.lock README.md MANIFEST.in ./
|
||||||
|
RUN printf 'gymnasium==0.29.1\nnumpy==1.26.4\n' > /tmp/robomme_override.txt \
|
||||||
|
&& uv pip install --no-cache --override /tmp/robomme_override.txt \
|
||||||
|
-e ".[smolvla,av-dep]" \
|
||||||
|
"robomme @ git+https://github.com/RoboMME/robomme_benchmark.git@main" \
|
||||||
|
&& python -c "import robomme; print('robomme import OK')"
|
||||||
|
|
||||||
|
# Overlay the PR's source code on top of the nightly image.
|
||||||
|
COPY --chown=user_lerobot:user_lerobot . .
|
||||||
|
|
||||||
|
CMD ["/bin/bash"]
|
||||||
@@ -77,6 +77,8 @@
|
|||||||
title: Adding a New Benchmark
|
title: Adding a New Benchmark
|
||||||
- local: libero
|
- local: libero
|
||||||
title: LIBERO
|
title: LIBERO
|
||||||
|
- local: libero_plus
|
||||||
|
title: LIBERO-plus
|
||||||
- local: metaworld
|
- local: metaworld
|
||||||
title: Meta-World
|
title: Meta-World
|
||||||
- local: robotwin
|
- local: robotwin
|
||||||
@@ -85,6 +87,8 @@
|
|||||||
title: RoboCasa365
|
title: RoboCasa365
|
||||||
- local: robocerebra
|
- local: robocerebra
|
||||||
title: RoboCerebra
|
title: RoboCerebra
|
||||||
|
- local: robomme
|
||||||
|
title: RoboMME
|
||||||
- local: envhub_isaaclab_arena
|
- local: envhub_isaaclab_arena
|
||||||
title: NVIDIA IsaacLab Arena Environments
|
title: NVIDIA IsaacLab Arena Environments
|
||||||
- local: vlabench
|
- local: vlabench
|
||||||
|
|||||||
@@ -0,0 +1,188 @@
|
|||||||
|
# LIBERO-plus
|
||||||
|
|
||||||
|
LIBERO-plus is a **robustness benchmark** for Vision-Language-Action (VLA) models built on top of [LIBERO](./libero). It systematically stress-tests policies by applying **seven independent perturbation dimensions** to the original LIBERO task set, exposing failure modes that standard benchmarks miss.
|
||||||
|
|
||||||
|
- Paper: [In-depth Robustness Analysis of Vision-Language-Action Models](https://arxiv.org/abs/2510.13626)
|
||||||
|
- GitHub: [sylvestf/LIBERO-plus](https://github.com/sylvestf/LIBERO-plus)
|
||||||
|
- Dataset: [lerobot/libero_plus](https://huggingface.co/datasets/lerobot/libero_plus)
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
## Perturbation dimensions
|
||||||
|
|
||||||
|
LIBERO-plus creates ~10 000 task variants by perturbing each original LIBERO task along these axes:
|
||||||
|
|
||||||
|
| Dimension | What changes |
|
||||||
|
| --------------------- | ----------------------------------------------------- |
|
||||||
|
| Objects layout | Target position, presence of confounding objects |
|
||||||
|
| Camera viewpoints | Camera position, orientation, field-of-view |
|
||||||
|
| Robot initial states | Manipulator start pose |
|
||||||
|
| Language instructions | LLM-rewritten task description (paraphrase / synonym) |
|
||||||
|
| Light conditions | Intensity, direction, color, shadow |
|
||||||
|
| Background textures | Scene surface and object appearance |
|
||||||
|
| Sensor noise | Photometric distortions and image degradation |
|
||||||
|
|
||||||
|
## Available task suites
|
||||||
|
|
||||||
|
LIBERO-plus covers the same five suites as LIBERO:
|
||||||
|
|
||||||
|
| Suite | CLI name | Tasks | Max steps | Description |
|
||||||
|
| -------------- | ---------------- | ----- | --------- | -------------------------------------------------- |
|
||||||
|
| LIBERO-Spatial | `libero_spatial` | 10 | 280 | Tasks requiring reasoning about spatial relations |
|
||||||
|
| LIBERO-Object | `libero_object` | 10 | 280 | Tasks centered on manipulating different objects |
|
||||||
|
| LIBERO-Goal | `libero_goal` | 10 | 300 | Goal-conditioned tasks with changing targets |
|
||||||
|
| LIBERO-90 | `libero_90` | 90 | 400 | Short-horizon tasks from the LIBERO-100 collection |
|
||||||
|
| LIBERO-Long | `libero_10` | 10 | 520 | Long-horizon tasks from the LIBERO-100 collection |
|
||||||
|
|
||||||
|
<Tip warning={true}>
|
||||||
|
Installing LIBERO-plus **replaces** vanilla LIBERO — it uninstalls `hf-libero`
|
||||||
|
so that `import libero` resolves to the LIBERO-plus fork. You cannot have both
|
||||||
|
installed at the same time. To switch back to vanilla LIBERO, uninstall the
|
||||||
|
fork and reinstall with `pip install -e ".[libero]"`.
|
||||||
|
</Tip>
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
### System dependencies (Linux only)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
sudo apt install libexpat1 libfontconfig1-dev libmagickwand-dev
|
||||||
|
```
|
||||||
|
|
||||||
|
### Python package
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install -e ".[libero]" "robosuite==1.4.1" bddl easydict mujoco wand scikit-image gym
|
||||||
|
git clone https://github.com/sylvestf/LIBERO-plus.git
|
||||||
|
cd LIBERO-plus && pip install --no-deps -e .
|
||||||
|
pip uninstall -y hf-libero # so `import libero` resolves to the fork
|
||||||
|
```
|
||||||
|
|
||||||
|
LIBERO-plus is installed from its GitHub fork rather than a pyproject extra — the fork ships as a namespace package that pip can't handle, so it must be cloned and added to `PYTHONPATH`. See `docker/Dockerfile.benchmark.libero_plus` for the canonical install. MuJoCo is required, so only Linux is supported.
|
||||||
|
|
||||||
|
<Tip>
|
||||||
|
Set the MuJoCo rendering backend before running evaluation:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export MUJOCO_GL=egl # headless / HPC / cloud
|
||||||
|
```
|
||||||
|
|
||||||
|
</Tip>
|
||||||
|
|
||||||
|
### Download LIBERO-plus assets
|
||||||
|
|
||||||
|
LIBERO-plus ships its extended asset pack separately. Download `assets.zip` from the [Hugging Face dataset](https://huggingface.co/datasets/Sylvest/LIBERO-plus/tree/main) and extract it into the LIBERO-plus package directory:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# After installing the package, find where it was installed:
|
||||||
|
python -c "import libero; print(libero.__file__)"
|
||||||
|
# Then extract assets.zip into <package_root>/libero/assets/
|
||||||
|
```
|
||||||
|
|
||||||
|
## Evaluation
|
||||||
|
|
||||||
|
### Default evaluation (recommended)
|
||||||
|
|
||||||
|
Evaluate across the four standard suites (10 episodes per task):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
lerobot-eval \
|
||||||
|
--policy.path="your-policy-id" \
|
||||||
|
--env.type=libero_plus \
|
||||||
|
--env.task=libero_spatial,libero_object,libero_goal,libero_10 \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.n_episodes=10 \
|
||||||
|
--env.max_parallel_tasks=1
|
||||||
|
```
|
||||||
|
|
||||||
|
### Single-suite evaluation
|
||||||
|
|
||||||
|
Evaluate on one LIBERO-plus suite:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
lerobot-eval \
|
||||||
|
--policy.path="your-policy-id" \
|
||||||
|
--env.type=libero_plus \
|
||||||
|
--env.task=libero_spatial \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.n_episodes=10
|
||||||
|
```
|
||||||
|
|
||||||
|
- `--env.task` picks the suite (`libero_spatial`, `libero_object`, etc.).
|
||||||
|
- `--env.task_ids` restricts to specific task indices (`[0]`, `[1,2,3]`, etc.). Omit to run all tasks in the suite.
|
||||||
|
- `--eval.batch_size` controls how many environments run in parallel.
|
||||||
|
- `--eval.n_episodes` sets how many episodes to run per task.
|
||||||
|
|
||||||
|
### Multi-suite evaluation
|
||||||
|
|
||||||
|
Benchmark a policy across multiple suites at once by passing a comma-separated list:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
lerobot-eval \
|
||||||
|
--policy.path="your-policy-id" \
|
||||||
|
--env.type=libero_plus \
|
||||||
|
--env.task=libero_spatial,libero_object \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.n_episodes=10
|
||||||
|
```
|
||||||
|
|
||||||
|
### Control mode
|
||||||
|
|
||||||
|
LIBERO-plus supports two control modes — `relative` (default) and `absolute`. Different VLA checkpoints are trained with different action parameterizations, so make sure the mode matches your policy:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
--env.control_mode=relative # or "absolute"
|
||||||
|
```
|
||||||
|
|
||||||
|
### Policy inputs and outputs
|
||||||
|
|
||||||
|
**Observations:**
|
||||||
|
|
||||||
|
- `observation.state` — 8-dim proprioceptive features (eef position, axis-angle orientation, gripper qpos)
|
||||||
|
- `observation.images.image` — main camera view (`agentview_image`), HWC uint8
|
||||||
|
- `observation.images.image2` — wrist camera view (`robot0_eye_in_hand_image`), HWC uint8
|
||||||
|
|
||||||
|
**Actions:**
|
||||||
|
|
||||||
|
- Continuous control in `Box(-1, 1, shape=(7,))` — 6D end-effector delta + 1D gripper
|
||||||
|
|
||||||
|
### Recommended evaluation episodes
|
||||||
|
|
||||||
|
For reproducible benchmarking, use **10 episodes per task** across all four standard suites (Spatial, Object, Goal, Long). This gives 400 total episodes and matches the protocol used for published results.
|
||||||
|
|
||||||
|
## Training
|
||||||
|
|
||||||
|
### Dataset
|
||||||
|
|
||||||
|
A LeRobot-format training dataset for LIBERO-plus is available at:
|
||||||
|
|
||||||
|
- [lerobot/libero_plus](https://huggingface.co/datasets/lerobot/libero_plus)
|
||||||
|
|
||||||
|
### Example training command
|
||||||
|
|
||||||
|
```bash
|
||||||
|
lerobot-train \
|
||||||
|
--policy.type=smolvla \
|
||||||
|
--policy.repo_id=${HF_USER}/smolvla_libero_plus \
|
||||||
|
--policy.load_vlm_weights=true \
|
||||||
|
--dataset.repo_id=lerobot/libero_plus \
|
||||||
|
--env.type=libero_plus \
|
||||||
|
--env.task=libero_spatial \
|
||||||
|
--output_dir=./outputs/ \
|
||||||
|
--steps=100000 \
|
||||||
|
--batch_size=4 \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.n_episodes=1 \
|
||||||
|
--eval_freq=1000
|
||||||
|
```
|
||||||
|
|
||||||
|
## Relationship to LIBERO
|
||||||
|
|
||||||
|
LIBERO-plus is a drop-in extension of LIBERO:
|
||||||
|
|
||||||
|
- Same Python gym interface (`LiberoEnv`, `LiberoProcessorStep`)
|
||||||
|
- Same camera names and observation/action format
|
||||||
|
- Same task suite names
|
||||||
|
- Installs under the same `libero` Python package name (different GitHub repo)
|
||||||
|
|
||||||
|
To use the original LIBERO benchmark, see [LIBERO](./libero) and use `--env.type=libero`.
|
||||||
@@ -0,0 +1,130 @@
|
|||||||
|
# RoboMME
|
||||||
|
|
||||||
|
[RoboMME](https://robomme.github.io) is a memory-augmented manipulation benchmark built on ManiSkill (SAPIEN). It evaluates a robot's ability to retain and use information across an episode — counting, object permanence, reference, and imitation.
|
||||||
|
|
||||||
|
- **16 tasks** across 4 memory-skill suites
|
||||||
|
- **1,600 training demos** (100 per task, 50 val, 50 test)
|
||||||
|
- **Dataset**: [`lerobot/robomme`](https://huggingface.co/datasets/lerobot/robomme) — LeRobot v3.0, 768K frames at 10 fps
|
||||||
|
- **Simulator**: ManiSkill / SAPIEN, Panda arm, Linux only
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
## Tasks
|
||||||
|
|
||||||
|
| Suite | Tasks |
|
||||||
|
| --------------------------------- | ------------------------------------------------------------- |
|
||||||
|
| **Counting** (temporal memory) | BinFill, PickXtimes, SwingXtimes, StopCube |
|
||||||
|
| **Permanence** (spatial memory) | VideoUnmask, VideoUnmaskSwap, ButtonUnmask, ButtonUnmaskSwap |
|
||||||
|
| **Reference** (object memory) | PickHighlight, VideoRepick, VideoPlaceButton, VideoPlaceOrder |
|
||||||
|
| **Imitation** (procedural memory) | MoveCube, InsertPeg, PatternLock, RouteStick |
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
> RoboMME requires **Linux** (ManiSkill/SAPIEN uses Vulkan rendering). Docker is recommended to isolate dependency conflicts.
|
||||||
|
|
||||||
|
### Native (Linux)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install --override <(printf 'gymnasium==0.29.1\nnumpy==1.26.4\n') \
|
||||||
|
-e '.[smolvla,av-dep]' \
|
||||||
|
'robomme @ git+https://github.com/RoboMME/robomme_benchmark.git@main'
|
||||||
|
```
|
||||||
|
|
||||||
|
> **Dependency note**: `mani-skill` (pulled by `robomme`) pins `gymnasium==0.29.1` and `numpy<2.0.0`, which conflict with lerobot's base `numpy>=2.0.0`. That's why `robomme` is not a pyproject extra — use the override install above, or the Docker approach below to avoid conflicts entirely.
|
||||||
|
|
||||||
|
### Docker (recommended)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Build base image first (from repo root)
|
||||||
|
docker build -f docker/Dockerfile.eval-base -t lerobot-eval-base .
|
||||||
|
|
||||||
|
# Build RoboMME eval image (applies gymnasium + numpy pin overrides)
|
||||||
|
docker build -f docker/Dockerfile.benchmark.robomme -t lerobot-robomme .
|
||||||
|
```
|
||||||
|
|
||||||
|
The `docker/Dockerfile.benchmark.robomme` image overrides `gymnasium==0.29.1` and `numpy==1.26.4` after lerobot's install. Both versions are runtime-safe for lerobot's actual API usage.
|
||||||
|
|
||||||
|
## Running Evaluation
|
||||||
|
|
||||||
|
### Default (single task, single episode)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
lerobot-eval \
|
||||||
|
--policy.path=<your_policy_repo> \
|
||||||
|
--env.type=robomme \
|
||||||
|
--env.task=PickXtimes \
|
||||||
|
--env.dataset_split=test \
|
||||||
|
--env.task_ids=[0] \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.n_episodes=1
|
||||||
|
```
|
||||||
|
|
||||||
|
### Multi-task evaluation
|
||||||
|
|
||||||
|
Evaluate multiple tasks in one run by comma-separating task names. Use `task_ids` to control which episodes are evaluated per task. Recommended: 50 episodes per task for the test split.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
lerobot-eval \
|
||||||
|
--policy.path=<your_policy_repo> \
|
||||||
|
--env.type=robomme \
|
||||||
|
--env.task=PickXtimes,BinFill,StopCube,MoveCube,InsertPeg \
|
||||||
|
--env.dataset_split=test \
|
||||||
|
--env.task_ids=[0,1,2,3,4,5,6,7,8,9] \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.n_episodes=50
|
||||||
|
```
|
||||||
|
|
||||||
|
### Key CLI options for `env.type=robomme`
|
||||||
|
|
||||||
|
| Option | Default | Description |
|
||||||
|
| -------------------- | ------------- | -------------------------------------------------- |
|
||||||
|
| `env.task` | `PickXtimes` | Any of the 16 task names above (comma-separated) |
|
||||||
|
| `env.dataset_split` | `test` | `train`, `val`, or `test` |
|
||||||
|
| `env.action_space` | `joint_angle` | `joint_angle` (8-D) or `ee_pose` (7-D) |
|
||||||
|
| `env.episode_length` | `300` | Max steps per episode |
|
||||||
|
| `env.task_ids` | `null` | List of episode indices to evaluate (null = `[0]`) |
|
||||||
|
|
||||||
|
## Dataset
|
||||||
|
|
||||||
|
The dataset [`lerobot/robomme`](https://huggingface.co/datasets/lerobot/robomme) is in **LeRobot v3.0 format** and can be loaded directly:
|
||||||
|
|
||||||
|
```python
|
||||||
|
from lerobot.datasets.lerobot_dataset import LeRobotDataset
|
||||||
|
|
||||||
|
dataset = LeRobotDataset("lerobot/robomme")
|
||||||
|
```
|
||||||
|
|
||||||
|
### Dataset features
|
||||||
|
|
||||||
|
| Feature | Shape | Description |
|
||||||
|
| ------------------ | ------------- | ------------------------------- |
|
||||||
|
| `image` | (256, 256, 3) | Front camera RGB |
|
||||||
|
| `wrist_image` | (256, 256, 3) | Wrist camera RGB |
|
||||||
|
| `actions` | (8,) | Joint angles + gripper |
|
||||||
|
| `state` | (8,) | Joint positions + gripper state |
|
||||||
|
| `simple_subgoal` | str | High-level language annotation |
|
||||||
|
| `grounded_subgoal` | str | Grounded language annotation |
|
||||||
|
| `episode_index` | int | Episode ID |
|
||||||
|
| `frame_index` | int | Frame within episode |
|
||||||
|
|
||||||
|
### Feature key alignment (training)
|
||||||
|
|
||||||
|
The env wrapper exposes `pixels/image` and `pixels/wrist_image` as observation keys. The `features_map` in `RoboMMEEnv` maps these to `observation.images.image` and `observation.images.wrist_image` for the policy. State is exposed as `agent_pos` and maps to `observation.state`.
|
||||||
|
|
||||||
|
The dataset's `image` and `wrist_image` columns already align with the policy input keys, so no renaming is needed when fine-tuning.
|
||||||
|
|
||||||
|
## Action Spaces
|
||||||
|
|
||||||
|
| Type | Dim | Description |
|
||||||
|
| ------------- | --- | --------------------------------------------------------- |
|
||||||
|
| `joint_angle` | 8 | 7 joint angles + 1 gripper (−1 closed, +1 open, absolute) |
|
||||||
|
| `ee_pose` | 7 | xyz + roll/pitch/yaw + gripper |
|
||||||
|
|
||||||
|
Set via `--env.action_space=joint_angle` (default) or `--env.action_space=ee_pose`.
|
||||||
|
|
||||||
|
## Platform Notes
|
||||||
|
|
||||||
|
- **Linux only**: ManiSkill requires SAPIEN/Vulkan. macOS and Windows are not supported.
|
||||||
|
- **GPU recommended**: Rendering is CPU-capable but slow; CUDA + Vulkan gives full speed.
|
||||||
|
- **gymnasium / numpy conflict**: See installation note above. Docker image handles this automatically.
|
||||||
|
- **ManiSkill fork**: `robomme` depends on a specific ManiSkill fork (`YinpeiDai/ManiSkill`), pulled in automatically via the `robomme` package.
|
||||||
@@ -217,6 +217,10 @@ metaworld = ["lerobot[dataset]", "metaworld==3.0.0", "lerobot[scipy-dep]"]
|
|||||||
# release), so any `vlabench>=X` pip spec is unresolvable. Install it
|
# release), so any `vlabench>=X` pip spec is unresolvable. Install it
|
||||||
# manually alongside MuJoCo / dm-control — see docs/source/vlabench.mdx
|
# manually alongside MuJoCo / dm-control — see docs/source/vlabench.mdx
|
||||||
# for the recipe.
|
# for the recipe.
|
||||||
|
# NOTE: robomme is NOT a pyproject extra — mani-skill hard-pins numpy<2
|
||||||
|
# which conflicts with lerobot's numpy>=2 base pin, so the two trees can't
|
||||||
|
# resolve into a single env. Install it only in the RoboMME Docker image
|
||||||
|
# via `uv pip install --override` (see docker/Dockerfile.benchmark.robomme).
|
||||||
# NOTE: robocasa is NOT exposed as a `lerobot` extra. Its setup.py pins
|
# NOTE: robocasa is NOT exposed as a `lerobot` extra. Its setup.py pins
|
||||||
# `lerobot==0.3.3` in install_requires, which cyclically shadows our own
|
# `lerobot==0.3.3` in install_requires, which cyclically shadows our own
|
||||||
# workspace `lerobot` and makes the graph unsolvable under any resolver
|
# workspace `lerobot` and makes the graph unsolvable under any resolver
|
||||||
|
|||||||
@@ -31,9 +31,23 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import json
|
import json
|
||||||
|
import re
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
# LIBERO-plus derives task.language by space-joining the perturbation-variant
|
||||||
|
# filename (grab_language_from_filename in libero/libero/benchmark/__init__.py),
|
||||||
|
# so non-_language_ variants inherit a trailing metadata blob like
|
||||||
|
# "view 0 0 100 0 0 initstate 0 noise 45" or "add 16". Strip those tokens so
|
||||||
|
# the description matches the base instruction used in the training dataset.
|
||||||
|
_LIBERO_PERTURBATION_TAIL_RE = re.compile(
|
||||||
|
r"(?:\s(?:view|initstate|noise|add|tb|table|light|level)(?:\s\d+)+)+$"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _strip_libero_perturbation_tail(instruction: str) -> str:
|
||||||
|
return _LIBERO_PERTURBATION_TAIL_RE.sub("", instruction).strip()
|
||||||
|
|
||||||
|
|
||||||
def _libero_descriptions(task_suite: str) -> dict[str, str]:
|
def _libero_descriptions(task_suite: str) -> dict[str, str]:
|
||||||
from libero.libero import benchmark # type: ignore[import-untyped]
|
from libero.libero import benchmark # type: ignore[import-untyped]
|
||||||
@@ -47,7 +61,10 @@ def _libero_descriptions(task_suite: str) -> dict[str, str]:
|
|||||||
)
|
)
|
||||||
return {}
|
return {}
|
||||||
suite = suite_dict[task_suite]()
|
suite = suite_dict[task_suite]()
|
||||||
return {f"{task_suite}_{i}": suite.get_task(i).language for i in range(suite.n_tasks)}
|
return {
|
||||||
|
f"{task_suite}_{i}": _strip_libero_perturbation_tail(suite.get_task(i).language)
|
||||||
|
for i in range(suite.n_tasks)
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def _metaworld_descriptions(task_name: str) -> dict[str, str]:
|
def _metaworld_descriptions(task_name: str) -> dict[str, str]:
|
||||||
@@ -92,6 +109,39 @@ def _robocasa_descriptions(task_spec: str) -> dict[str, str]:
|
|||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
_ROBOMME_DESCRIPTIONS = {
|
||||||
|
"BinFill": "Fill the target bin with the correct number of cubes",
|
||||||
|
"PickXtimes": "Pick the indicated cube the specified number of times",
|
||||||
|
"SwingXtimes": "Swing the object the specified number of times",
|
||||||
|
"StopCube": "Grasp and stop the moving cube",
|
||||||
|
"VideoUnmask": "Pick the cube shown in the reference video",
|
||||||
|
"VideoUnmaskSwap": "Pick the cube matching the reference video after a swap",
|
||||||
|
"ButtonUnmask": "Press the button indicated by the reference",
|
||||||
|
"ButtonUnmaskSwap": "Press the correct button after objects are swapped",
|
||||||
|
"PickHighlight": "Pick the highlighted cube",
|
||||||
|
"VideoRepick": "Repick the cube shown in the reference video",
|
||||||
|
"VideoPlaceButton": "Place the cube on the button shown in the video",
|
||||||
|
"VideoPlaceOrder": "Place cubes in the order shown in the video",
|
||||||
|
"MoveCube": "Move the cube to the target location",
|
||||||
|
"InsertPeg": "Insert the peg into the target hole",
|
||||||
|
"PatternLock": "Unlock the pattern by pressing buttons in sequence",
|
||||||
|
"RouteStick": "Route the stick through the required waypoints",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _robomme_descriptions(task_names: str, task_ids: list[int] | None = None) -> dict[str, str]:
|
||||||
|
"""Return descriptions for each requested RoboMME task. Keys match the
|
||||||
|
video filename pattern `<task>_<task_id>` used by the eval script."""
|
||||||
|
if task_ids is None:
|
||||||
|
task_ids = [0]
|
||||||
|
out: dict[str, str] = {}
|
||||||
|
for name in (t.strip() for t in task_names.split(",") if t.strip()):
|
||||||
|
desc = _ROBOMME_DESCRIPTIONS.get(name, name)
|
||||||
|
for tid in task_ids:
|
||||||
|
out[f"{name}_{tid}"] = desc
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
def _vlabench_descriptions(task_spec: str) -> dict[str, str]:
|
def _vlabench_descriptions(task_spec: str) -> dict[str, str]:
|
||||||
"""For each task in the comma-separated list, emit a cleaned-name label.
|
"""For each task in the comma-separated list, emit a cleaned-name label.
|
||||||
|
|
||||||
@@ -111,12 +161,22 @@ def main() -> int:
|
|||||||
parser = argparse.ArgumentParser(description=__doc__)
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
parser.add_argument("--env", required=True, help="Environment family (libero, metaworld, ...)")
|
parser.add_argument("--env", required=True, help="Environment family (libero, metaworld, ...)")
|
||||||
parser.add_argument("--task", required=True, help="Task/suite name (e.g. libero_spatial)")
|
parser.add_argument("--task", required=True, help="Task/suite name (e.g. libero_spatial)")
|
||||||
|
parser.add_argument(
|
||||||
|
"--task-ids",
|
||||||
|
type=str,
|
||||||
|
default=None,
|
||||||
|
help="Comma-separated task IDs (e.g. '0,1,2'). Default: [0]",
|
||||||
|
)
|
||||||
parser.add_argument("--output", required=True, help="Path to write task_descriptions.json")
|
parser.add_argument("--output", required=True, help="Path to write task_descriptions.json")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
task_ids: list[int] | None = None
|
||||||
|
if args.task_ids:
|
||||||
|
task_ids = [int(x.strip()) for x in args.task_ids.split(",")]
|
||||||
|
|
||||||
descriptions: dict[str, str] = {}
|
descriptions: dict[str, str] = {}
|
||||||
try:
|
try:
|
||||||
if args.env == "libero":
|
if args.env == ("libero", "libero_plus"):
|
||||||
descriptions = _libero_descriptions(args.task)
|
descriptions = _libero_descriptions(args.task)
|
||||||
elif args.env == "metaworld":
|
elif args.env == "metaworld":
|
||||||
descriptions = _metaworld_descriptions(args.task)
|
descriptions = _metaworld_descriptions(args.task)
|
||||||
@@ -124,6 +184,8 @@ def main() -> int:
|
|||||||
descriptions = _robotwin_descriptions(args.task)
|
descriptions = _robotwin_descriptions(args.task)
|
||||||
elif args.env == "robocasa":
|
elif args.env == "robocasa":
|
||||||
descriptions = _robocasa_descriptions(args.task)
|
descriptions = _robocasa_descriptions(args.task)
|
||||||
|
elif args.env == "robomme":
|
||||||
|
descriptions = _robomme_descriptions(args.task, task_ids=task_ids)
|
||||||
elif args.env == "vlabench":
|
elif args.env == "vlabench":
|
||||||
descriptions = _vlabench_descriptions(args.task)
|
descriptions = _vlabench_descriptions(args.task)
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -331,6 +331,7 @@ class LiberoEnv(EnvConfig):
|
|||||||
camera_name_mapping: dict[str, str] | None = None
|
camera_name_mapping: dict[str, str] | None = None
|
||||||
observation_height: int = 360
|
observation_height: int = 360
|
||||||
observation_width: int = 360
|
observation_width: int = 360
|
||||||
|
is_libero_plus: bool = False
|
||||||
features: dict[str, PolicyFeature] = field(
|
features: dict[str, PolicyFeature] = field(
|
||||||
default_factory=lambda: {
|
default_factory=lambda: {
|
||||||
ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
|
ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
|
||||||
@@ -432,6 +433,7 @@ class LiberoEnv(EnvConfig):
|
|||||||
control_mode=self.control_mode,
|
control_mode=self.control_mode,
|
||||||
episode_length=self.episode_length,
|
episode_length=self.episode_length,
|
||||||
camera_name_mapping=self.camera_name_mapping,
|
camera_name_mapping=self.camera_name_mapping,
|
||||||
|
is_libero_plus=self.is_libero_plus,
|
||||||
)
|
)
|
||||||
|
|
||||||
def get_env_processors(self):
|
def get_env_processors(self):
|
||||||
@@ -716,6 +718,30 @@ class IsaaclabArenaEnv(HubEnvConfig):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@EnvConfig.register_subclass("libero_plus")
|
||||||
|
@dataclass
|
||||||
|
class LiberoPlusEnv(LiberoEnv):
|
||||||
|
"""Config for LIBERO-plus robustness benchmark evaluation.
|
||||||
|
|
||||||
|
LIBERO-plus extends LIBERO with 7 perturbation dimensions (camera viewpoints,
|
||||||
|
object layouts, robot initial states, language instructions, lighting, background
|
||||||
|
textures, sensor noise) producing ~10k task variants.
|
||||||
|
|
||||||
|
The gym interface is identical to LIBERO so this class reuses ``LiberoEnv``
|
||||||
|
entirely — only the registered name and default task suite differ.
|
||||||
|
|
||||||
|
Install: see docker/Dockerfile.benchmark.libero_plus — LIBERO-plus ships
|
||||||
|
as a namespace package from a git fork and must be cloned + PYTHONPATH'd
|
||||||
|
rather than installed as a pyproject extra.
|
||||||
|
|
||||||
|
See Also:
|
||||||
|
https://github.com/sylvestf/LIBERO-plus
|
||||||
|
"""
|
||||||
|
|
||||||
|
task: str = "libero_spatial"
|
||||||
|
is_libero_plus: bool = True
|
||||||
|
|
||||||
|
|
||||||
@EnvConfig.register_subclass("robotwin")
|
@EnvConfig.register_subclass("robotwin")
|
||||||
@dataclass
|
@dataclass
|
||||||
class RoboTwinEnvConfig(EnvConfig):
|
class RoboTwinEnvConfig(EnvConfig):
|
||||||
@@ -801,3 +827,60 @@ class RoboTwinEnvConfig(EnvConfig):
|
|||||||
observation_width=self.observation_width,
|
observation_width=self.observation_width,
|
||||||
episode_length=self.episode_length,
|
episode_length=self.episode_length,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@EnvConfig.register_subclass("robomme")
|
||||||
|
@dataclass
|
||||||
|
class RoboMMEEnv(EnvConfig):
|
||||||
|
"""RoboMME memory-augmented manipulation benchmark (ManiSkill/SAPIEN).
|
||||||
|
|
||||||
|
16 tasks across 4 suites: Counting, Permanence, Reference, Imitation.
|
||||||
|
Dataset: lerobot/robomme (LeRobot v3.0, 1,600 episodes).
|
||||||
|
Benchmark: https://github.com/RoboMME/robomme_benchmark
|
||||||
|
|
||||||
|
Requires the `robomme` git package installed separately (Linux only);
|
||||||
|
see docker/Dockerfile.benchmark.robomme for the canonical install.
|
||||||
|
"""
|
||||||
|
|
||||||
|
task: str = "PickXtimes"
|
||||||
|
fps: int = 10
|
||||||
|
episode_length: int = 300
|
||||||
|
action_space: str = "joint_angle" # or "ee_pose" (7-D)
|
||||||
|
dataset_split: str = "test" # "train" | "val" | "test"
|
||||||
|
task_ids: list[int] | None = None
|
||||||
|
features: dict[str, PolicyFeature] = field(default_factory=dict)
|
||||||
|
features_map: dict[str, str] = field(
|
||||||
|
default_factory=lambda: {
|
||||||
|
ACTION: ACTION,
|
||||||
|
"pixels/image": f"{OBS_IMAGES}.image",
|
||||||
|
"pixels/wrist_image": f"{OBS_IMAGES}.wrist_image",
|
||||||
|
"agent_pos": OBS_STATE,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
def __post_init__(self):
|
||||||
|
action_dim = 8 if self.action_space == "joint_angle" else 7
|
||||||
|
self.features = {
|
||||||
|
ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(action_dim,)),
|
||||||
|
"pixels/image": PolicyFeature(type=FeatureType.VISUAL, shape=(256, 256, 3)),
|
||||||
|
"pixels/wrist_image": PolicyFeature(type=FeatureType.VISUAL, shape=(256, 256, 3)),
|
||||||
|
"agent_pos": PolicyFeature(type=FeatureType.STATE, shape=(8,)),
|
||||||
|
}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def gym_kwargs(self) -> dict:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
def create_envs(self, n_envs: int, use_async_envs: bool = True):
|
||||||
|
from lerobot.envs.robomme import create_robomme_envs
|
||||||
|
|
||||||
|
env_cls = _make_vec_env_cls(use_async_envs, n_envs)
|
||||||
|
return create_robomme_envs(
|
||||||
|
task=self.task,
|
||||||
|
n_envs=n_envs,
|
||||||
|
action_space_type=self.action_space,
|
||||||
|
dataset=self.dataset_split,
|
||||||
|
episode_length=self.episode_length,
|
||||||
|
task_ids=self.task_ids,
|
||||||
|
env_cls=env_cls,
|
||||||
|
)
|
||||||
|
|||||||
@@ -16,6 +16,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from collections.abc import Callable, Iterable, Mapping, Sequence
|
from collections.abc import Callable, Iterable, Mapping, Sequence
|
||||||
from functools import partial
|
from functools import partial
|
||||||
@@ -56,14 +57,34 @@ def _select_task_ids(total_tasks: int, task_ids: Iterable[int] | None) -> list[i
|
|||||||
return ids
|
return ids
|
||||||
|
|
||||||
|
|
||||||
def get_task_init_states(task_suite: Any, i: int) -> np.ndarray:
|
# LIBERO-plus perturbation variants encode the perturbation in the filename
|
||||||
init_states_path = (
|
# but on disk only the base `.pruned_init` exists — strip the suffix to match
|
||||||
Path(get_libero_path("init_states"))
|
# LIBERO-plus's own suite.get_task_init_states() (we reimplement it here so we
|
||||||
/ task_suite.tasks[i].problem_folder
|
# can pass weights_only=False for PyTorch 2.6+ numpy pickles).
|
||||||
/ task_suite.tasks[i].init_states_file
|
_LIBERO_PERTURBATION_SUFFIX_RE = re.compile(r"_(?:language|view|light)_[^.]*|_(?:table|tb)_\d+")
|
||||||
)
|
|
||||||
init_states = torch.load(init_states_path, weights_only=False) # nosec B614
|
|
||||||
return init_states
|
def get_task_init_states(task_suite: Any, i: int, is_libero_plus: bool = False) -> np.ndarray:
|
||||||
|
task = task_suite.tasks[i]
|
||||||
|
filename = Path(task.init_states_file)
|
||||||
|
root = Path(get_libero_path("init_states"))
|
||||||
|
|
||||||
|
if not is_libero_plus:
|
||||||
|
init_states_path = root / task.problem_folder / filename.name
|
||||||
|
return torch.load(init_states_path, weights_only=False) # nosec B614
|
||||||
|
|
||||||
|
# LIBERO-plus: `_add_` / `_level` variants store extra-object layouts under
|
||||||
|
# libero_newobj/ as a flat array that must be reshaped to (1, -1).
|
||||||
|
if "_add_" in filename.name or "_level" in filename.name:
|
||||||
|
init_states_path = root / "libero_newobj" / task.problem_folder / filename.name
|
||||||
|
init_states = torch.load(init_states_path, weights_only=False) # nosec B614
|
||||||
|
return init_states.reshape(1, -1)
|
||||||
|
|
||||||
|
# LIBERO-plus perturbation variants encode the perturbation in the filename
|
||||||
|
# but on disk only the base `.pruned_init` exists — strip the suffix to match.
|
||||||
|
stripped = _LIBERO_PERTURBATION_SUFFIX_RE.sub("", filename.stem) + filename.suffix
|
||||||
|
init_states_path = root / task.problem_folder / stripped
|
||||||
|
return torch.load(init_states_path, weights_only=False) # nosec B614
|
||||||
|
|
||||||
|
|
||||||
def get_libero_dummy_action():
|
def get_libero_dummy_action():
|
||||||
@@ -105,9 +126,11 @@ class LiberoEnv(gym.Env):
|
|||||||
camera_name_mapping: dict[str, str] | None = None,
|
camera_name_mapping: dict[str, str] | None = None,
|
||||||
num_steps_wait: int = 10,
|
num_steps_wait: int = 10,
|
||||||
control_mode: str = "relative",
|
control_mode: str = "relative",
|
||||||
|
is_libero_plus: bool = False,
|
||||||
):
|
):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.task_id = task_id
|
self.task_id = task_id
|
||||||
|
self.is_libero_plus = is_libero_plus
|
||||||
self.obs_type = obs_type
|
self.obs_type = obs_type
|
||||||
self.render_mode = render_mode
|
self.render_mode = render_mode
|
||||||
self.observation_width = observation_width
|
self.observation_width = observation_width
|
||||||
@@ -134,7 +157,11 @@ class LiberoEnv(gym.Env):
|
|||||||
self.episode_index = episode_index
|
self.episode_index = episode_index
|
||||||
self.episode_length = episode_length
|
self.episode_length = episode_length
|
||||||
# Load once and keep
|
# Load once and keep
|
||||||
self._init_states = get_task_init_states(task_suite, self.task_id) if self.init_states else None
|
self._init_states = (
|
||||||
|
get_task_init_states(task_suite, self.task_id, is_libero_plus=self.is_libero_plus)
|
||||||
|
if self.init_states
|
||||||
|
else None
|
||||||
|
)
|
||||||
self._reset_stride = n_envs # when performing a reset, append `_reset_stride` to `init_state_id`.
|
self._reset_stride = n_envs # when performing a reset, append `_reset_stride` to `init_state_id`.
|
||||||
|
|
||||||
self.init_state_id = self.episode_index # tie each sub-env to a fixed init state
|
self.init_state_id = self.episode_index # tie each sub-env to a fixed init state
|
||||||
@@ -367,6 +394,7 @@ def _make_env_fns(
|
|||||||
gym_kwargs: Mapping[str, Any],
|
gym_kwargs: Mapping[str, Any],
|
||||||
control_mode: str,
|
control_mode: str,
|
||||||
camera_name_mapping: dict[str, str] | None = None,
|
camera_name_mapping: dict[str, str] | None = None,
|
||||||
|
is_libero_plus: bool = False,
|
||||||
) -> list[Callable[[], LiberoEnv]]:
|
) -> list[Callable[[], LiberoEnv]]:
|
||||||
"""Build n_envs factory callables for a single (suite, task_id)."""
|
"""Build n_envs factory callables for a single (suite, task_id)."""
|
||||||
|
|
||||||
@@ -383,6 +411,7 @@ def _make_env_fns(
|
|||||||
n_envs=n_envs,
|
n_envs=n_envs,
|
||||||
control_mode=control_mode,
|
control_mode=control_mode,
|
||||||
camera_name_mapping=camera_name_mapping,
|
camera_name_mapping=camera_name_mapping,
|
||||||
|
is_libero_plus=is_libero_plus,
|
||||||
**local_kwargs,
|
**local_kwargs,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -405,6 +434,7 @@ def create_libero_envs(
|
|||||||
control_mode: str = "relative",
|
control_mode: str = "relative",
|
||||||
episode_length: int | None = None,
|
episode_length: int | None = None,
|
||||||
camera_name_mapping: dict[str, str] | None = None,
|
camera_name_mapping: dict[str, str] | None = None,
|
||||||
|
is_libero_plus: bool = False,
|
||||||
) -> dict[str, dict[int, Any]]:
|
) -> dict[str, dict[int, Any]]:
|
||||||
"""
|
"""
|
||||||
Create vectorized LIBERO environments with a consistent return shape.
|
Create vectorized LIBERO environments with a consistent return shape.
|
||||||
@@ -463,6 +493,7 @@ def create_libero_envs(
|
|||||||
gym_kwargs=gym_kwargs,
|
gym_kwargs=gym_kwargs,
|
||||||
control_mode=control_mode,
|
control_mode=control_mode,
|
||||||
camera_name_mapping=camera_name_mapping,
|
camera_name_mapping=camera_name_mapping,
|
||||||
|
is_libero_plus=is_libero_plus,
|
||||||
)
|
)
|
||||||
if is_async:
|
if is_async:
|
||||||
lazy = _LazyAsyncVectorEnv(fns, cached_obs_space, cached_act_space, cached_metadata)
|
lazy = _LazyAsyncVectorEnv(fns, cached_obs_space, cached_act_space, cached_metadata)
|
||||||
|
|||||||
@@ -0,0 +1,245 @@
|
|||||||
|
"""RoboMME environment wrapper for LeRobot evaluation.
|
||||||
|
|
||||||
|
Wraps the RoboMME ``BenchmarkEnvBuilder`` into a Gymnasium-compatible
|
||||||
|
``VectorEnv`` suitable for ``lerobot_eval``.
|
||||||
|
|
||||||
|
RoboMME tasks:
|
||||||
|
Counting: BinFill, PickXtimes, SwingXtimes, StopCube
|
||||||
|
Permanence: VideoUnmask, VideoUnmaskSwap, ButtonUnmask, ButtonUnmaskSwap
|
||||||
|
Reference: PickHighlight, VideoRepick, VideoPlaceButton, VideoPlaceOrder
|
||||||
|
Imitation: MoveCube, InsertPeg, PatternLock, RouteStick
|
||||||
|
|
||||||
|
Dataset: lerobot/robomme (LeRobot v3.0, 1,600 episodes)
|
||||||
|
Install: see docker/Dockerfile.benchmark.robomme (Linux only — mani-skill vs numpy pin conflict)
|
||||||
|
Benchmark: https://github.com/RoboMME/robomme_benchmark
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from collections.abc import Callable, Sequence
|
||||||
|
from functools import partial
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import gymnasium as gym
|
||||||
|
import numpy as np
|
||||||
|
from gymnasium import spaces
|
||||||
|
|
||||||
|
from .utils import _LazyAsyncVectorEnv
|
||||||
|
|
||||||
|
ROBOMME_TASKS = [
|
||||||
|
"BinFill",
|
||||||
|
"PickXtimes",
|
||||||
|
"SwingXtimes",
|
||||||
|
"StopCube",
|
||||||
|
"VideoUnmask",
|
||||||
|
"VideoUnmaskSwap",
|
||||||
|
"ButtonUnmask",
|
||||||
|
"ButtonUnmaskSwap",
|
||||||
|
"PickHighlight",
|
||||||
|
"VideoRepick",
|
||||||
|
"VideoPlaceButton",
|
||||||
|
"VideoPlaceOrder",
|
||||||
|
"MoveCube",
|
||||||
|
"InsertPeg",
|
||||||
|
"PatternLock",
|
||||||
|
"RouteStick",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
class RoboMMEGymEnv(gym.Env):
|
||||||
|
"""Thin Gymnasium wrapper around a single RoboMME episode env."""
|
||||||
|
|
||||||
|
metadata = {"render_modes": ["rgb_array"], "render_fps": 10}
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
task: str = "PickXtimes",
|
||||||
|
action_space_type: str = "joint_angle",
|
||||||
|
dataset: str = "test",
|
||||||
|
episode_idx: int = 0,
|
||||||
|
max_steps: int = 300,
|
||||||
|
):
|
||||||
|
super().__init__()
|
||||||
|
from robomme.env_record_wrapper import BenchmarkEnvBuilder
|
||||||
|
|
||||||
|
self._task = task
|
||||||
|
self._action_space_type = action_space_type
|
||||||
|
self._dataset = dataset
|
||||||
|
self._episode_idx = episode_idx
|
||||||
|
self._max_steps = max_steps
|
||||||
|
self._max_episode_steps = max_steps
|
||||||
|
|
||||||
|
self._builder = BenchmarkEnvBuilder(
|
||||||
|
env_id=task,
|
||||||
|
dataset=dataset,
|
||||||
|
action_space=action_space_type,
|
||||||
|
gui_render=False,
|
||||||
|
max_steps=max_steps,
|
||||||
|
)
|
||||||
|
self._env = None
|
||||||
|
self._last_raw_obs: dict | None = None
|
||||||
|
|
||||||
|
action_dim = 8 if action_space_type == "joint_angle" else 7
|
||||||
|
self.action_space = spaces.Box(low=-1.0, high=1.0, shape=(action_dim,), dtype=np.float32)
|
||||||
|
# `pixels` must be a nested Dict so `preprocess_observation()` in
|
||||||
|
# envs/utils.py picks it up and maps each camera to
|
||||||
|
# `observation.images.<cam>`. A flat layout (`pixels/image`,
|
||||||
|
# `pixels/wrist_image`) silently drops every image from the batch.
|
||||||
|
self.observation_space = spaces.Dict(
|
||||||
|
{
|
||||||
|
"pixels": spaces.Dict(
|
||||||
|
{
|
||||||
|
"image": spaces.Box(0, 255, shape=(256, 256, 3), dtype=np.uint8),
|
||||||
|
"wrist_image": spaces.Box(0, 255, shape=(256, 256, 3), dtype=np.uint8),
|
||||||
|
}
|
||||||
|
),
|
||||||
|
"agent_pos": spaces.Box(-np.inf, np.inf, shape=(8,), dtype=np.float32),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
def reset(self, *, seed=None, options=None):
|
||||||
|
super().reset(seed=seed)
|
||||||
|
self._env = self._builder.make_env_for_episode(
|
||||||
|
episode_idx=self._episode_idx,
|
||||||
|
max_steps=self._max_steps,
|
||||||
|
)
|
||||||
|
obs, info = self._env.reset()
|
||||||
|
self._last_raw_obs = obs
|
||||||
|
return self._convert_obs(obs), self._convert_info(info)
|
||||||
|
|
||||||
|
def step(self, action):
|
||||||
|
obs, reward, terminated, truncated, info = self._env.step(action)
|
||||||
|
self._last_raw_obs = obs
|
||||||
|
|
||||||
|
terminated_bool = bool(terminated.item()) if hasattr(terminated, "item") else bool(terminated)
|
||||||
|
truncated_bool = bool(truncated.item()) if hasattr(truncated, "item") else bool(truncated)
|
||||||
|
|
||||||
|
status = info.get("status", "ongoing")
|
||||||
|
is_success = status == "success"
|
||||||
|
conv_info = self._convert_info(info)
|
||||||
|
conv_info["is_success"] = is_success
|
||||||
|
|
||||||
|
return self._convert_obs(obs), float(reward), terminated_bool, truncated_bool, conv_info
|
||||||
|
|
||||||
|
def render(self) -> np.ndarray | None:
|
||||||
|
"""Return the front camera image from the last observation for video recording."""
|
||||||
|
if self._last_raw_obs is None:
|
||||||
|
return np.zeros((256, 256, 3), dtype=np.uint8)
|
||||||
|
front = self._last_raw_obs.get("front_rgb_list")
|
||||||
|
if front is None:
|
||||||
|
return np.zeros((256, 256, 3), dtype=np.uint8)
|
||||||
|
frame = front[-1] if isinstance(front, list) else front
|
||||||
|
return np.asarray(frame, dtype=np.uint8)
|
||||||
|
|
||||||
|
def _convert_obs(self, obs: dict) -> dict:
|
||||||
|
front_rgb = (
|
||||||
|
obs["front_rgb_list"][-1] if isinstance(obs["front_rgb_list"], list) else obs["front_rgb_list"]
|
||||||
|
)
|
||||||
|
wrist_rgb = (
|
||||||
|
obs["wrist_rgb_list"][-1] if isinstance(obs["wrist_rgb_list"], list) else obs["wrist_rgb_list"]
|
||||||
|
)
|
||||||
|
joint_state = (
|
||||||
|
obs["joint_state_list"][-1]
|
||||||
|
if isinstance(obs["joint_state_list"], list)
|
||||||
|
else obs["joint_state_list"]
|
||||||
|
)
|
||||||
|
gripper_state = (
|
||||||
|
obs["gripper_state_list"][-1]
|
||||||
|
if isinstance(obs["gripper_state_list"], list)
|
||||||
|
else obs["gripper_state_list"]
|
||||||
|
)
|
||||||
|
|
||||||
|
front_rgb = np.asarray(front_rgb, dtype=np.uint8)
|
||||||
|
wrist_rgb = np.asarray(wrist_rgb, dtype=np.uint8)
|
||||||
|
joint = np.asarray(joint_state, dtype=np.float32).flatten()[:7]
|
||||||
|
gripper = np.asarray(gripper_state, dtype=np.float32).flatten()[:1]
|
||||||
|
state = np.concatenate([joint, gripper])
|
||||||
|
|
||||||
|
return {
|
||||||
|
"pixels": {"image": front_rgb, "wrist_image": wrist_rgb},
|
||||||
|
"agent_pos": state,
|
||||||
|
}
|
||||||
|
|
||||||
|
def _convert_info(self, info: dict) -> dict:
|
||||||
|
return {
|
||||||
|
"status": info.get("status", "ongoing"),
|
||||||
|
"task_goal": info.get("task_goal", ""),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _make_env_fns(
|
||||||
|
*,
|
||||||
|
task: str,
|
||||||
|
n_envs: int,
|
||||||
|
action_space_type: str,
|
||||||
|
dataset: str,
|
||||||
|
episode_length: int,
|
||||||
|
task_id: int,
|
||||||
|
) -> list[Callable[[], RoboMMEGymEnv]]:
|
||||||
|
"""Build n_envs factory callables for one RoboMME task id."""
|
||||||
|
|
||||||
|
def _make_one(episode_index: int) -> RoboMMEGymEnv:
|
||||||
|
return RoboMMEGymEnv(
|
||||||
|
task=task,
|
||||||
|
action_space_type=action_space_type,
|
||||||
|
dataset=dataset,
|
||||||
|
episode_idx=episode_index,
|
||||||
|
max_steps=episode_length,
|
||||||
|
)
|
||||||
|
|
||||||
|
return [partial(_make_one, task_id + i) for i in range(n_envs)]
|
||||||
|
|
||||||
|
|
||||||
|
def create_robomme_envs(
|
||||||
|
task: str,
|
||||||
|
n_envs: int = 1,
|
||||||
|
action_space_type: str = "joint_angle",
|
||||||
|
dataset: str = "test",
|
||||||
|
episode_length: int = 300,
|
||||||
|
task_ids: list[int] | None = None,
|
||||||
|
env_cls: Callable[[Sequence[Callable[[], Any]]], Any] | None = None,
|
||||||
|
) -> dict[str, dict[int, gym.vector.VectorEnv]]:
|
||||||
|
"""Create vectorized RoboMME environments for evaluation.
|
||||||
|
|
||||||
|
`task` may be a single RoboMME task name (e.g. "PickXtimes") or a
|
||||||
|
comma-separated list (e.g. "PickXtimes,BinFill,StopCube"). Each task
|
||||||
|
becomes its own suite in the returned mapping.
|
||||||
|
|
||||||
|
Returns {suite_name: {task_id: VectorEnv}} matching lerobot's expected format.
|
||||||
|
"""
|
||||||
|
if env_cls is None or not callable(env_cls):
|
||||||
|
raise ValueError("env_cls must be a callable that wraps a list of env factory callables.")
|
||||||
|
if not isinstance(n_envs, int) or n_envs <= 0:
|
||||||
|
raise ValueError(f"n_envs must be a positive int; got {n_envs}.")
|
||||||
|
|
||||||
|
if task_ids is None:
|
||||||
|
task_ids = [0]
|
||||||
|
|
||||||
|
task_names = [t.strip() for t in task.split(",") if t.strip()]
|
||||||
|
is_async = env_cls is gym.vector.AsyncVectorEnv
|
||||||
|
cached_obs_space: spaces.Space | None = None
|
||||||
|
cached_act_space: spaces.Space | None = None
|
||||||
|
cached_metadata: dict[str, Any] | None = None
|
||||||
|
out: dict[str, dict[int, gym.vector.VectorEnv]] = {}
|
||||||
|
for task_name in task_names:
|
||||||
|
envs_by_task: dict[int, gym.vector.VectorEnv] = {}
|
||||||
|
for task_id in task_ids:
|
||||||
|
fns = _make_env_fns(
|
||||||
|
task=task_name,
|
||||||
|
n_envs=n_envs,
|
||||||
|
action_space_type=action_space_type,
|
||||||
|
dataset=dataset,
|
||||||
|
episode_length=episode_length,
|
||||||
|
task_id=task_id,
|
||||||
|
)
|
||||||
|
if is_async:
|
||||||
|
lazy = _LazyAsyncVectorEnv(fns, cached_obs_space, cached_act_space, cached_metadata)
|
||||||
|
if cached_obs_space is None:
|
||||||
|
cached_obs_space = lazy.observation_space
|
||||||
|
cached_act_space = lazy.action_space
|
||||||
|
cached_metadata = lazy.metadata
|
||||||
|
envs_by_task[task_id] = lazy
|
||||||
|
else:
|
||||||
|
envs_by_task[task_id] = env_cls(fns)
|
||||||
|
out[task_name] = envs_by_task
|
||||||
|
return out
|
||||||
@@ -0,0 +1,232 @@
|
|||||||
|
# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
"""Unit tests for the RoboMME env wrapper and config.
|
||||||
|
|
||||||
|
RoboMME requires Linux + ManiSkill (Vulkan/SAPIEN), so tests that touch the
|
||||||
|
env wrapper mock the ``robomme`` package. Tests that only exercise the
|
||||||
|
dataclass config run without any mocking.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import sys
|
||||||
|
from types import ModuleType
|
||||||
|
from unittest.mock import MagicMock
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
|
||||||
|
def _install_robomme_stub():
|
||||||
|
"""Register a minimal stub for the ``robomme`` package on sys.modules."""
|
||||||
|
stub = ModuleType("robomme")
|
||||||
|
wrapper_stub = ModuleType("robomme.env_record_wrapper")
|
||||||
|
|
||||||
|
class FakeBuilder:
|
||||||
|
def __init__(self, **kwargs):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def make_env_for_episode(self, episode_idx: int, max_steps: int):
|
||||||
|
env = MagicMock()
|
||||||
|
obs = {
|
||||||
|
"front_rgb_list": [np.zeros((256, 256, 3), dtype=np.uint8)],
|
||||||
|
"wrist_rgb_list": [np.zeros((256, 256, 3), dtype=np.uint8)],
|
||||||
|
"joint_state_list": [np.zeros(7, dtype=np.float32)],
|
||||||
|
"gripper_state_list": [np.zeros(2, dtype=np.float32)],
|
||||||
|
}
|
||||||
|
env.reset.return_value = (obs, {"status": "ongoing", "task_goal": "pick the cube"})
|
||||||
|
env.step.return_value = (obs, 0.0, False, False, {"status": "ongoing", "task_goal": ""})
|
||||||
|
return env
|
||||||
|
|
||||||
|
wrapper_stub.BenchmarkEnvBuilder = FakeBuilder
|
||||||
|
stub.env_record_wrapper = wrapper_stub
|
||||||
|
sys.modules["robomme"] = stub
|
||||||
|
sys.modules["robomme.env_record_wrapper"] = wrapper_stub
|
||||||
|
|
||||||
|
|
||||||
|
def _uninstall_robomme_stub():
|
||||||
|
sys.modules.pop("robomme", None)
|
||||||
|
sys.modules.pop("robomme.env_record_wrapper", None)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Config tests (no sim required)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_robomme_env_config_defaults():
|
||||||
|
from lerobot.envs.configs import RoboMMEEnv
|
||||||
|
|
||||||
|
cfg = RoboMMEEnv()
|
||||||
|
assert cfg.task == "PickXtimes"
|
||||||
|
assert cfg.fps == 10
|
||||||
|
assert cfg.episode_length == 300
|
||||||
|
assert cfg.action_space == "joint_angle"
|
||||||
|
assert cfg.dataset_split == "test"
|
||||||
|
assert cfg.task_ids is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_robomme_env_config_type():
|
||||||
|
from lerobot.envs.configs import RoboMMEEnv
|
||||||
|
|
||||||
|
cfg = RoboMMEEnv()
|
||||||
|
assert cfg.type == "robomme"
|
||||||
|
|
||||||
|
|
||||||
|
def test_robomme_features_map():
|
||||||
|
from lerobot.envs.configs import RoboMMEEnv
|
||||||
|
from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE
|
||||||
|
|
||||||
|
cfg = RoboMMEEnv()
|
||||||
|
assert cfg.features_map[ACTION] == ACTION
|
||||||
|
assert cfg.features_map["pixels/image"] == f"{OBS_IMAGES}.image"
|
||||||
|
assert cfg.features_map["pixels/wrist_image"] == f"{OBS_IMAGES}.wrist_image"
|
||||||
|
assert cfg.features_map["agent_pos"] == OBS_STATE
|
||||||
|
|
||||||
|
|
||||||
|
def test_robomme_features_action_dim_joint_angle():
|
||||||
|
from lerobot.envs.configs import RoboMMEEnv
|
||||||
|
from lerobot.utils.constants import ACTION
|
||||||
|
|
||||||
|
cfg = RoboMMEEnv(action_space="joint_angle")
|
||||||
|
assert cfg.features[ACTION].shape == (8,)
|
||||||
|
|
||||||
|
|
||||||
|
def test_robomme_features_action_dim_ee_pose():
|
||||||
|
"""`ee_pose` uses a 7-D action; __post_init__ sets the correct shape."""
|
||||||
|
from lerobot.envs.configs import RoboMMEEnv
|
||||||
|
from lerobot.utils.constants import ACTION
|
||||||
|
|
||||||
|
cfg = RoboMMEEnv(action_space="ee_pose")
|
||||||
|
assert cfg.features[ACTION].shape == (7,)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Obs conversion (pure Python, no sim)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_convert_obs_list_format():
|
||||||
|
"""_convert_obs takes the last element from list-format obs fields and
|
||||||
|
emits a nested ``pixels`` dict (image, wrist_image) plus ``agent_pos``.
|
||||||
|
|
||||||
|
The nested layout is required so ``preprocess_observation()`` in
|
||||||
|
``envs/utils.py`` maps each camera to ``observation.images.<cam>``.
|
||||||
|
"""
|
||||||
|
_install_robomme_stub()
|
||||||
|
try:
|
||||||
|
from lerobot.envs.robomme import RoboMMEGymEnv
|
||||||
|
|
||||||
|
env = RoboMMEGymEnv.__new__(RoboMMEGymEnv)
|
||||||
|
|
||||||
|
front = np.full((256, 256, 3), 42, dtype=np.uint8)
|
||||||
|
wrist = np.full((256, 256, 3), 7, dtype=np.uint8)
|
||||||
|
joints = np.arange(7, dtype=np.float32)
|
||||||
|
gripper = np.array([0.5, 0.5], dtype=np.float32)
|
||||||
|
|
||||||
|
obs_raw = {
|
||||||
|
"front_rgb_list": [np.zeros_like(front), front],
|
||||||
|
"wrist_rgb_list": [np.zeros_like(wrist), wrist],
|
||||||
|
"joint_state_list": [np.zeros(7, dtype=np.float32), joints],
|
||||||
|
"gripper_state_list": [np.zeros(2, dtype=np.float32), gripper],
|
||||||
|
}
|
||||||
|
|
||||||
|
result = env._convert_obs(obs_raw)
|
||||||
|
np.testing.assert_array_equal(result["pixels"]["image"], front)
|
||||||
|
np.testing.assert_array_equal(result["pixels"]["wrist_image"], wrist)
|
||||||
|
assert result["agent_pos"].shape == (8,)
|
||||||
|
np.testing.assert_array_almost_equal(result["agent_pos"][:7], joints)
|
||||||
|
assert result["agent_pos"][7] == gripper[0]
|
||||||
|
finally:
|
||||||
|
_uninstall_robomme_stub()
|
||||||
|
|
||||||
|
|
||||||
|
def test_convert_obs_array_format():
|
||||||
|
"""_convert_obs also handles non-list (direct array) obs."""
|
||||||
|
_install_robomme_stub()
|
||||||
|
try:
|
||||||
|
from lerobot.envs.robomme import RoboMMEGymEnv
|
||||||
|
|
||||||
|
env = RoboMMEGymEnv.__new__(RoboMMEGymEnv)
|
||||||
|
|
||||||
|
front = np.zeros((256, 256, 3), dtype=np.uint8)
|
||||||
|
obs_raw = {
|
||||||
|
"front_rgb_list": front,
|
||||||
|
"wrist_rgb_list": front,
|
||||||
|
"joint_state_list": np.zeros(7, dtype=np.float32),
|
||||||
|
"gripper_state_list": np.zeros(2, dtype=np.float32),
|
||||||
|
}
|
||||||
|
result = env._convert_obs(obs_raw)
|
||||||
|
assert result["pixels"]["image"].shape == (256, 256, 3)
|
||||||
|
assert result["pixels"]["wrist_image"].shape == (256, 256, 3)
|
||||||
|
assert result["agent_pos"].shape == (8,)
|
||||||
|
finally:
|
||||||
|
_uninstall_robomme_stub()
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# create_robomme_envs (mocked sim)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_create_robomme_envs_returns_correct_structure():
|
||||||
|
"""Single task -> {task_name: {task_id: VectorEnv}} with one entry per task_id."""
|
||||||
|
_install_robomme_stub()
|
||||||
|
try:
|
||||||
|
from lerobot.envs.robomme import create_robomme_envs
|
||||||
|
|
||||||
|
env_cls = MagicMock(return_value=MagicMock())
|
||||||
|
result = create_robomme_envs(
|
||||||
|
task="PickXtimes",
|
||||||
|
n_envs=1,
|
||||||
|
task_ids=[0, 1],
|
||||||
|
env_cls=env_cls,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert "PickXtimes" in result
|
||||||
|
assert 0 in result["PickXtimes"]
|
||||||
|
assert 1 in result["PickXtimes"]
|
||||||
|
assert env_cls.call_count == 2
|
||||||
|
finally:
|
||||||
|
_uninstall_robomme_stub()
|
||||||
|
|
||||||
|
|
||||||
|
def test_create_robomme_envs_multi_task():
|
||||||
|
"""Comma-separated task list produces one suite per task."""
|
||||||
|
_install_robomme_stub()
|
||||||
|
try:
|
||||||
|
from lerobot.envs.robomme import create_robomme_envs
|
||||||
|
|
||||||
|
env_cls = MagicMock(return_value=MagicMock())
|
||||||
|
result = create_robomme_envs(
|
||||||
|
task="PickXtimes,BinFill,StopCube",
|
||||||
|
n_envs=1,
|
||||||
|
env_cls=env_cls,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert set(result.keys()) == {"PickXtimes", "BinFill", "StopCube"}
|
||||||
|
finally:
|
||||||
|
_uninstall_robomme_stub()
|
||||||
|
|
||||||
|
|
||||||
|
def test_create_robomme_envs_raises_on_invalid_env_cls():
|
||||||
|
_install_robomme_stub()
|
||||||
|
try:
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from lerobot.envs.robomme import create_robomme_envs
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="env_cls must be a callable"):
|
||||||
|
create_robomme_envs(task="PickXtimes", n_envs=1, env_cls=None)
|
||||||
|
finally:
|
||||||
|
_uninstall_robomme_stub()
|
||||||
Reference in New Issue
Block a user