mirror of
https://github.com/huggingface/lerobot.git
synced 2026-07-27 11:46:04 +00:00
Compare commits
3 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 129537068a | |||
| 1205bb086d | |||
| 501b916601 |
@@ -128,6 +128,9 @@ jobs:
|
|||||||
'--env.camera_name_mapping={\"agentview_image\": \"camera1\", \"robot0_eye_in_hand_image\": \"camera2\"}' \
|
'--env.camera_name_mapping={\"agentview_image\": \"camera1\", \"robot0_eye_in_hand_image\": \"camera2\"}' \
|
||||||
--policy.empty_cameras=1 \
|
--policy.empty_cameras=1 \
|
||||||
--output_dir=/tmp/eval-artifacts
|
--output_dir=/tmp/eval-artifacts
|
||||||
|
python3 /lerobot/scripts/ci/extract_task_descriptions.py \
|
||||||
|
--env libero --task libero_spatial \
|
||||||
|
--output /tmp/eval-artifacts/task_descriptions.json 2>/dev/null || true
|
||||||
"
|
"
|
||||||
|
|
||||||
- name: Copy Libero artifacts from container
|
- name: Copy Libero artifacts from container
|
||||||
@@ -162,6 +165,60 @@ jobs:
|
|||||||
path: /tmp/libero-artifacts/metrics.json
|
path: /tmp/libero-artifacts/metrics.json
|
||||||
if-no-files-found: warn
|
if-no-files-found: warn
|
||||||
|
|
||||||
|
# ── LIBERO TRAIN+EVAL SMOKE ──────────────────────────────────────────────
|
||||||
|
# Train SmolVLA for 1 step (batch_size=1, dataset episode 0 only) then
|
||||||
|
# immediately runs eval inside the training loop (eval_freq=1, 1 episode).
|
||||||
|
# Tests the full train→eval-within-training pipeline end-to-end.
|
||||||
|
- name: Run Libero train+eval smoke (1 step, eval_freq=1)
|
||||||
|
run: |
|
||||||
|
docker run --name libero-train-smoke --gpus all \
|
||||||
|
--shm-size=4g \
|
||||||
|
-e HF_HOME=/tmp/hf \
|
||||||
|
-e HF_USER_TOKEN="${HF_USER_TOKEN}" \
|
||||||
|
-e HF_HUB_DOWNLOAD_TIMEOUT=300 \
|
||||||
|
lerobot-benchmark-libero:ci \
|
||||||
|
bash -c "
|
||||||
|
hf auth login --token \"\$HF_USER_TOKEN\" --add-to-git-credential 2>/dev/null || true
|
||||||
|
accelerate launch --num_processes=1 \$(which lerobot-train) \
|
||||||
|
--policy.path=lerobot/smolvla_base \
|
||||||
|
--policy.load_vlm_weights=true \
|
||||||
|
--policy.scheduler_decay_steps=25000 \
|
||||||
|
--policy.freeze_vision_encoder=false \
|
||||||
|
--policy.train_expert_only=false \
|
||||||
|
--dataset.repo_id=lerobot/libero \
|
||||||
|
--dataset.episodes=[0] \
|
||||||
|
--dataset.use_imagenet_stats=false \
|
||||||
|
--env.type=libero \
|
||||||
|
--env.task=libero_spatial \
|
||||||
|
'--env.camera_name_mapping={\"agentview_image\": \"camera1\", \"robot0_eye_in_hand_image\": \"camera2\"}' \
|
||||||
|
--policy.empty_cameras=1 \
|
||||||
|
--output_dir=/tmp/train-smoke \
|
||||||
|
--steps=1 \
|
||||||
|
--batch_size=1 \
|
||||||
|
--eval_freq=1 \
|
||||||
|
--eval.n_episodes=1 \
|
||||||
|
--eval.batch_size=1 \
|
||||||
|
--eval.use_async_envs=false \
|
||||||
|
--save_freq=1 \
|
||||||
|
--policy.push_to_hub=false \
|
||||||
|
'--rename_map={\"observation.images.image\": \"observation.images.camera1\", \"observation.images.image2\": \"observation.images.camera2\"}'
|
||||||
|
"
|
||||||
|
|
||||||
|
- name: Copy Libero train-smoke artifacts from container
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
mkdir -p /tmp/libero-train-smoke-artifacts
|
||||||
|
docker cp libero-train-smoke:/tmp/train-smoke/. /tmp/libero-train-smoke-artifacts/ 2>/dev/null || true
|
||||||
|
docker rm -f libero-train-smoke || true
|
||||||
|
|
||||||
|
- name: Upload Libero train-smoke eval video
|
||||||
|
if: always()
|
||||||
|
uses: actions/upload-artifact@v4
|
||||||
|
with:
|
||||||
|
name: libero-train-smoke-video
|
||||||
|
path: /tmp/libero-train-smoke-artifacts/eval/
|
||||||
|
if-no-files-found: warn
|
||||||
|
|
||||||
# ── METAWORLD ─────────────────────────────────────────────────────────────
|
# ── METAWORLD ─────────────────────────────────────────────────────────────
|
||||||
# Isolated image: lerobot[metaworld] only (metaworld==3.0.0, mujoco>=3 chain)
|
# Isolated image: lerobot[metaworld] only (metaworld==3.0.0, mujoco>=3 chain)
|
||||||
metaworld-integration-test:
|
metaworld-integration-test:
|
||||||
@@ -214,6 +271,9 @@ jobs:
|
|||||||
'--rename_map={\"observation.image\": \"observation.images.camera1\"}' \
|
'--rename_map={\"observation.image\": \"observation.images.camera1\"}' \
|
||||||
--policy.empty_cameras=2 \
|
--policy.empty_cameras=2 \
|
||||||
--output_dir=/tmp/eval-artifacts
|
--output_dir=/tmp/eval-artifacts
|
||||||
|
python3 /lerobot/scripts/ci/extract_task_descriptions.py \
|
||||||
|
--env metaworld --task metaworld-push-v3 \
|
||||||
|
--output /tmp/eval-artifacts/task_descriptions.json 2>/dev/null || true
|
||||||
"
|
"
|
||||||
|
|
||||||
- name: Copy MetaWorld artifacts from container
|
- name: Copy MetaWorld artifacts from container
|
||||||
|
|||||||
@@ -0,0 +1,89 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
"""Extract natural-language task descriptions for a benchmark suite.
|
||||||
|
|
||||||
|
Runs inside the benchmark Docker container (where the env library is installed)
|
||||||
|
immediately after lerobot-eval, writing a JSON file that parse_eval_metrics.py
|
||||||
|
picks up and embeds in metrics.json.
|
||||||
|
|
||||||
|
Output format: {"<suite>_<task_idx>": "<nl instruction>", ...}
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python scripts/ci/extract_task_descriptions.py \\
|
||||||
|
--env libero --task libero_spatial \\
|
||||||
|
--output /tmp/eval-artifacts/task_descriptions.json
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
def _libero_descriptions(task_suite: str) -> dict[str, str]:
|
||||||
|
from libero.libero import benchmark # type: ignore[import-untyped]
|
||||||
|
|
||||||
|
suite_dict = benchmark.get_benchmark_dict()
|
||||||
|
if task_suite not in suite_dict:
|
||||||
|
print(
|
||||||
|
f"[extract_task_descriptions] Unknown LIBERO suite '{task_suite}'. "
|
||||||
|
f"Available: {list(suite_dict.keys())}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
return {}
|
||||||
|
suite = suite_dict[task_suite]()
|
||||||
|
return {f"{task_suite}_{i}": suite.get_task(i).language for i in range(suite.n_tasks)}
|
||||||
|
|
||||||
|
|
||||||
|
def _metaworld_descriptions(task_name: str) -> dict[str, str]:
|
||||||
|
# MetaWorld tasks don't expose a separate NL description attribute;
|
||||||
|
# use a cleaned version of the task name as the description.
|
||||||
|
label = task_name.removeprefix("metaworld-").replace("-", " ").strip()
|
||||||
|
return {f"{task_name}_0": label}
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--env", required=True, help="Environment family (libero, metaworld, ...)")
|
||||||
|
parser.add_argument("--task", required=True, help="Task/suite name (e.g. libero_spatial)")
|
||||||
|
parser.add_argument("--output", required=True, help="Path to write task_descriptions.json")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
descriptions: dict[str, str] = {}
|
||||||
|
try:
|
||||||
|
if args.env == "libero":
|
||||||
|
descriptions = _libero_descriptions(args.task)
|
||||||
|
elif args.env == "metaworld":
|
||||||
|
descriptions = _metaworld_descriptions(args.task)
|
||||||
|
else:
|
||||||
|
print(
|
||||||
|
f"[extract_task_descriptions] No description extractor for env '{args.env}'.",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
except Exception as exc:
|
||||||
|
print(f"[extract_task_descriptions] Warning: {exc}", file=sys.stderr)
|
||||||
|
|
||||||
|
out_path = Path(args.output)
|
||||||
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
out_path.write_text(json.dumps(descriptions, indent=2))
|
||||||
|
print(f"[extract_task_descriptions] {len(descriptions)} descriptions → {out_path}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -39,30 +39,30 @@ import sys
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
def _extract_pc_success(info: dict) -> tuple[float | None, int | None]:
|
def _extract_metrics(info: dict) -> tuple[float | None, int | None, float | None, float | None]:
|
||||||
"""Extract (pc_success, n_episodes) from eval_info.json.
|
"""Extract (pc_success, n_episodes, avg_sum_reward, eval_s) from eval_info.json.
|
||||||
|
|
||||||
Handles two output shapes:
|
Handles two output shapes:
|
||||||
- Single-task: {"aggregated": {"pc_success": 80.0, ...}}
|
- Single-task: {"aggregated": {"pc_success": 80.0, ...}}
|
||||||
- Multi-task: {"overall": {"pc_success": 80.0, "n_episodes": 5, ...}}
|
- Multi-task: {"overall": {"pc_success": 80.0, "n_episodes": 5, ...}}
|
||||||
"""
|
"""
|
||||||
# Single-task path
|
for key in ("aggregated", "overall"):
|
||||||
if "aggregated" in info:
|
if key not in info:
|
||||||
agg = info["aggregated"]
|
continue
|
||||||
|
agg = info[key]
|
||||||
pc = agg.get("pc_success")
|
pc = agg.get("pc_success")
|
||||||
n = agg.get("n_episodes") # may be absent in older format
|
n = agg.get("n_episodes")
|
||||||
|
reward = agg.get("avg_sum_reward")
|
||||||
|
eval_s = agg.get("eval_s")
|
||||||
if pc is not None and not math.isnan(pc):
|
if pc is not None and not math.isnan(pc):
|
||||||
return float(pc), int(n) if n is not None else None
|
return (
|
||||||
|
float(pc),
|
||||||
|
int(n) if n is not None else None,
|
||||||
|
float(reward) if reward is not None else None,
|
||||||
|
float(eval_s) if eval_s is not None else None,
|
||||||
|
)
|
||||||
|
|
||||||
# Multi-task path
|
return None, None, None, None
|
||||||
if "overall" in info:
|
|
||||||
overall = info["overall"]
|
|
||||||
pc = overall.get("pc_success")
|
|
||||||
n = overall.get("n_episodes")
|
|
||||||
if pc is not None and not math.isnan(pc):
|
|
||||||
return float(pc), int(n) if n is not None else None
|
|
||||||
|
|
||||||
return None, None
|
|
||||||
|
|
||||||
|
|
||||||
def main() -> int:
|
def main() -> int:
|
||||||
@@ -80,11 +80,13 @@ def main() -> int:
|
|||||||
|
|
||||||
pc_success: float | None = None
|
pc_success: float | None = None
|
||||||
n_episodes: int | None = None
|
n_episodes: int | None = None
|
||||||
|
avg_sum_reward: float | None = None
|
||||||
|
eval_s: float | None = None
|
||||||
|
|
||||||
if eval_info_path.exists():
|
if eval_info_path.exists():
|
||||||
try:
|
try:
|
||||||
info = json.loads(eval_info_path.read_text())
|
info = json.loads(eval_info_path.read_text())
|
||||||
pc_success, n_episodes = _extract_pc_success(info)
|
pc_success, n_episodes, avg_sum_reward, eval_s = _extract_metrics(info)
|
||||||
except (json.JSONDecodeError, KeyError, TypeError) as exc:
|
except (json.JSONDecodeError, KeyError, TypeError) as exc:
|
||||||
print(f"[parse_eval_metrics] Warning: could not parse eval_info.json: {exc}", file=sys.stderr)
|
print(f"[parse_eval_metrics] Warning: could not parse eval_info.json: {exc}", file=sys.stderr)
|
||||||
else:
|
else:
|
||||||
@@ -93,12 +95,26 @@ def main() -> int:
|
|||||||
file=sys.stderr,
|
file=sys.stderr,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
task_descriptions: dict[str, str] = {}
|
||||||
|
task_desc_path = artifacts_dir / "task_descriptions.json"
|
||||||
|
if task_desc_path.exists():
|
||||||
|
try:
|
||||||
|
task_descriptions = json.loads(task_desc_path.read_text())
|
||||||
|
except json.JSONDecodeError as exc:
|
||||||
|
print(
|
||||||
|
f"[parse_eval_metrics] Warning: could not parse task_descriptions.json: {exc}",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
|
||||||
metrics = {
|
metrics = {
|
||||||
"env": args.env,
|
"env": args.env,
|
||||||
"task": args.task,
|
"task": args.task,
|
||||||
"policy": args.policy,
|
"policy": args.policy,
|
||||||
"pc_success": pc_success,
|
"pc_success": pc_success,
|
||||||
"n_episodes": n_episodes,
|
"n_episodes": n_episodes,
|
||||||
|
"avg_sum_reward": avg_sum_reward,
|
||||||
|
"eval_s": eval_s,
|
||||||
|
"task_descriptions": task_descriptions,
|
||||||
}
|
}
|
||||||
|
|
||||||
out_path = artifacts_dir / "metrics.json"
|
out_path = artifacts_dir / "metrics.json"
|
||||||
|
|||||||
Reference in New Issue
Block a user