mirror of
https://github.com/huggingface/lerobot.git
synced 2026-07-29 12:39:41 +00:00
Compare commits
47 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 77259f436e | |||
| 85f5c3606d | |||
| b587e81587 | |||
| 4658dada9b | |||
| 57ea6f4106 | |||
| 4209639f33 | |||
| fc7a0bc2fd | |||
| 5f6513551c | |||
| 70e157e00f | |||
| 1837be51bf | |||
| bedd56eed9 | |||
| c165e4df68 | |||
| 5e24da483a | |||
| 9c54665a76 | |||
| f6a845c30c | |||
| 45e8336854 | |||
| 5046e2df32 | |||
| 1c88e26c6d | |||
| 69a3edfa33 | |||
| 2492ce2c29 | |||
| c8e75da55f | |||
| 2eae31ea2b | |||
| c997abe739 | |||
| c73579055e | |||
| 4be438161b | |||
| 806d28a883 | |||
| 573b65ff6b | |||
| bc55713e7c | |||
| 4f53c42583 | |||
| bfced3d149 | |||
| 4969813d4e | |||
| 1c87ca31a3 | |||
| 4bcde762cc | |||
| 943ae78cfe | |||
| 3363688f1e | |||
| 0876629e72 | |||
| 305614b8c6 | |||
| 02d3202c4f | |||
| 3b6de2fdf8 | |||
| 744f3667c0 | |||
| fdde436776 | |||
| 5c683c65c6 | |||
| dfbc25c58f | |||
| 804c76bcc2 | |||
| e6afa69be9 | |||
| 31d1439e29 | |||
| 1c118c6359 |
+5
-1
@@ -374,7 +374,11 @@ torch = [{ index = "pytorch-cu128", marker = "sys_platform == 'linux'" }]
|
|||||||
torchvision = [{ index = "pytorch-cu128", marker = "sys_platform == 'linux'" }]
|
torchvision = [{ index = "pytorch-cu128", marker = "sys_platform == 'linux'" }]
|
||||||
|
|
||||||
[tool.setuptools.package-data]
|
[tool.setuptools.package-data]
|
||||||
lerobot = ["envs/*.json", "annotations/steerable_pipeline/prompts/*.txt"]
|
lerobot = [
|
||||||
|
"envs/*.json",
|
||||||
|
"annotations/steerable_pipeline/prompts/*.txt",
|
||||||
|
"teleoperators/pico_headset/assets/*.npz",
|
||||||
|
]
|
||||||
|
|
||||||
[tool.setuptools.packages.find]
|
[tool.setuptools.packages.find]
|
||||||
where = ["src"]
|
where = ["src"]
|
||||||
|
|||||||
@@ -0,0 +1,89 @@
|
|||||||
|
# Unitree G1 — SONIC encoder/decoder whole-body control
|
||||||
|
|
||||||
|
This package runs NVIDIA's **SONIC** encoder/decoder on the Unitree G1, in MuJoCo
|
||||||
|
simulation or on real hardware, driven by a dense **34-D whole-body command** (the
|
||||||
|
OpenHLM / pi0.5 action layout). It is a pure-Python/ONNX reimplementation of the
|
||||||
|
reference-tracking half of the SONIC deploy stack (no `gear_sonic`/torch dependency, and
|
||||||
|
no motion planner): the encoder compresses a reference motion window into a latent token
|
||||||
|
and the decoder maps that token + proprioception history into 50 Hz joint-position
|
||||||
|
targets for the robot's PD controller.
|
||||||
|
|
||||||
|
## Controllers
|
||||||
|
|
||||||
|
Selected with `--robot.controller=<ClassName>`:
|
||||||
|
|
||||||
|
| Controller | Purpose |
|
||||||
|
| ------------------------------ | ------------------------------------------------------------ |
|
||||||
|
| `SonicWholeBodyController` | SONIC encoder/decoder driven by a 34-D OpenHLM/pi0.5 command |
|
||||||
|
| `GrootLocomotionController` | GR00T locomotion policy |
|
||||||
|
| `HolosomaLocomotionController` | Holosoma locomotion policy |
|
||||||
|
|
||||||
|
The rest of this document covers the SONIC whole-body path.
|
||||||
|
|
||||||
|
Each tick the `SonicWholeBodyController` takes a 34-D command (`wb.0.pos … wb.33.pos`) in the OpenHLM
|
||||||
|
layout:
|
||||||
|
|
||||||
|
```
|
||||||
|
[L-arm(7), L-grip(1), R-arm(7), R-grip(1), L-leg(6), R-leg(6), waist(3),
|
||||||
|
root roll/pitch + yaw-rate(3)]
|
||||||
|
```
|
||||||
|
|
||||||
|
The 29 joint targets become the SONIC encode-mode-0 reference (accumulated into a rolling
|
||||||
|
50-frame trajectory with finite-difference velocities so the encoder's lookahead sees a
|
||||||
|
real motion sequence), the root roll/pitch set the anchor orientation, and the two grip
|
||||||
|
scalars can drive the Dex3 hands (see below). On startup the controller **interpolates**
|
||||||
|
from the robot's measured pose into the policy's commanded target over ~3 s (no snap).
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
- `onnxruntime` (CPU) **or** `onnxruntime-gpu` (recommended). Verify with:
|
||||||
|
```bash
|
||||||
|
python -c "import onnxruntime as ort; print(ort.get_available_providers())"
|
||||||
|
```
|
||||||
|
- `mujoco` for simulation (`is_simulation=True`).
|
||||||
|
- The SONIC encoder/decoder ONNX models download automatically from the
|
||||||
|
`nvidia/GEAR-SONIC` Hub repo.
|
||||||
|
|
||||||
|
## Running a rollout
|
||||||
|
|
||||||
|
Drive the G1 with a 34-D VLA policy (OpenHLM / pi0.5) via `lerobot-rollout`:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
lerobot-rollout \
|
||||||
|
--strategy.type=base \
|
||||||
|
--policy.path=<pi05_openhlm_dir> \
|
||||||
|
--robot.type=unitree_g1 \
|
||||||
|
--robot.controller=SonicWholeBodyController \
|
||||||
|
--robot.is_simulation=true \
|
||||||
|
--robot.publish_hands=true \
|
||||||
|
--task="<language instruction>" \
|
||||||
|
--duration=45 --device=cuda
|
||||||
|
```
|
||||||
|
|
||||||
|
### Cameras
|
||||||
|
|
||||||
|
Image-conditioned policies need camera frames. Two options are available without live
|
||||||
|
cameras:
|
||||||
|
|
||||||
|
- **Black frames**: `--robot.empty_cameras='[base, left_wrist, right_wrist]'`.
|
||||||
|
- **Replay a recorded episode** as the camera feed:
|
||||||
|
```bash
|
||||||
|
--robot.replay_camera_parquet=<episode.parquet> \
|
||||||
|
--robot.replay_camera_map='{base: head_image_left, left_wrist: left_wrist_image, right_wrist: right_wrist_image}'
|
||||||
|
```
|
||||||
|
|
||||||
|
### Hands (Dex3)
|
||||||
|
|
||||||
|
`--robot.publish_hands=true` publishes `rt/dex3/{left,right}/cmd` from the two grip
|
||||||
|
scalars (`wb.7.pos` left, `wb.15.pos` right). The scalar is interpolated between
|
||||||
|
`hand_open_grip_value` (default 1.0 = open) and `hand_closed_grip_value` (default 0.0 =
|
||||||
|
closed) and scaled onto `hand_closed_pose` (7 joints:
|
||||||
|
`thumb_0, thumb_1, thumb_2, middle_0, middle_1, index_0, index_1`). Flip the signs in
|
||||||
|
`hand_closed_pose` if the fingers curl the wrong way, or raise `hand_kp` for a firmer
|
||||||
|
grip.
|
||||||
|
|
||||||
|
## Observation state
|
||||||
|
|
||||||
|
When the whole-body controller is active the robot exposes a 34-D proprio state
|
||||||
|
(`wb_state.0.pos … wb_state.33.pos`) in the same OpenHLM layout as the action, which the
|
||||||
|
rollout aggregates into `observation.state` for the policy.
|
||||||
@@ -62,12 +62,76 @@ class UnitreeG1Config(RobotConfig):
|
|||||||
# Socket config for ZMQ bridge
|
# Socket config for ZMQ bridge
|
||||||
robot_ip: str = "192.168.123.164" # default G1 IP
|
robot_ip: str = "192.168.123.164" # default G1 IP
|
||||||
|
|
||||||
|
# Run the locomotion / whole-body controller ONBOARD the robot (policy on the G1
|
||||||
|
# itself, against local DDS at full rate) instead of on the laptop over the ZMQ
|
||||||
|
# socket bridge. In this mode the robot object uses the real Unitree SDK channels
|
||||||
|
# and expects high-level actions (arm targets + joystick axes, or 64-D SONIC
|
||||||
|
# tokens) fed via send_action -- e.g. by run_g1_onboard.py, which receives them
|
||||||
|
# from the laptop over ZMQ. Mutually exclusive with is_simulation.
|
||||||
|
onboard: bool = False
|
||||||
|
# DDS network interface for onboard mode (None = SDK default, matching
|
||||||
|
# run_g1_server.py's ChannelFactoryInitialize(0)).
|
||||||
|
dds_interface: str | None = None
|
||||||
|
# Onboard sub-flags. On a real G1 both are True: the built-in motion services
|
||||||
|
# must be released before we can write lowcmd, and locomotion axes are read from
|
||||||
|
# the physical wireless remote. Against a DDS sim neither applies (no
|
||||||
|
# MotionSwitcher, no physical remote), so set both False so the controller takes
|
||||||
|
# its locomotion axes purely from send_action (ZMQ) input.
|
||||||
|
release_motion_control: bool = True
|
||||||
|
physical_remote: bool = True
|
||||||
|
|
||||||
# Cameras (ZMQ-based remote cameras)
|
# Cameras (ZMQ-based remote cameras)
|
||||||
cameras: dict[str, CameraConfig] = field(default_factory=dict)
|
cameras: dict[str, CameraConfig] = field(default_factory=dict)
|
||||||
|
|
||||||
|
# Synthetic zero-image cameras exposed as ``observation.images.{name}`` (H×W×3
|
||||||
|
# black frames). Lets image-conditioned policies (e.g. pi0.5 / OpenHLM) run in
|
||||||
|
# sim before real cameras are wired. Empty = disabled.
|
||||||
|
empty_cameras: list[str] = field(default_factory=list)
|
||||||
|
empty_camera_hw: tuple[int, int] = (224, 224)
|
||||||
|
|
||||||
|
# Publish Dex3 hand commands (``rt/dex3/{left,right}/cmd``) driven by the OpenHLM
|
||||||
|
# gripper scalars (``wb.7.pos`` left, ``wb.15.pos`` right). Lets the 43-DoF sim
|
||||||
|
# (or a real Dex3-equipped G1) show grasping. The scalar in [0, 1] is remapped to
|
||||||
|
# a curl amount (``hand_open_grip_value`` -> open) and scaled onto
|
||||||
|
# ``hand_closed_pose`` (7 joints: thumb_0/1/2, middle_0/1, index_0/1). Flip signs
|
||||||
|
# in ``hand_closed_pose`` if fingers curl the wrong way.
|
||||||
|
publish_hands: bool = False
|
||||||
|
# When False, connect() does not start the background controller thread, so a
|
||||||
|
# caller can drive the controller synchronously (one decode per fed action),
|
||||||
|
# reproducing the deploy's single 50Hz control clock for faithful replay.
|
||||||
|
run_controller_thread: bool = True
|
||||||
|
hand_open_grip_value: float = 1.0
|
||||||
|
hand_closed_grip_value: float = 0.0
|
||||||
|
hand_closed_pose: list[float] = field(default_factory=lambda: [1.0, 0.9, 0.9, 1.3, 1.3, 1.3, 1.3])
|
||||||
|
hand_kp: float = 1.5
|
||||||
|
hand_kd: float = 0.1
|
||||||
|
|
||||||
|
# Replay recorded camera frames from a LeRobot parquet episode as the camera
|
||||||
|
# feed (e.g. OpenHLM-data episode). Maps a robot camera name to a parquet image
|
||||||
|
# column; frames advance one per observation and loop. Lets a VLA see the real
|
||||||
|
# task video in sim without live cameras. Empty map = disabled.
|
||||||
|
replay_camera_parquet: str | None = None
|
||||||
|
replay_camera_map: dict[str, str] = field(default_factory=dict)
|
||||||
|
replay_camera_loop: bool = True
|
||||||
|
|
||||||
|
# Token-output VLA interface for the SONIC decoder. When True (and the controller
|
||||||
|
# is ``SonicWholeBodyController``), the robot advertises a 64-D latent-token action
|
||||||
|
# space (``motion_token.{i}.pos``) instead of the 34-D whole-body command, and
|
||||||
|
# exposes the last commanded token as a 64-D ``observation.state``
|
||||||
|
# (``motion_token_state.{i}.pos``). This lets ``lerobot-rollout`` drive a policy
|
||||||
|
# that was trained with 64-D SONIC motion tokens as both state and action
|
||||||
|
# (e.g. nepyope/sonic_walk): the decoder consumes the token directly, encoder
|
||||||
|
# bypassed. Ignored unless a SONIC whole-body controller is active.
|
||||||
|
sonic_token_action: bool = False
|
||||||
|
|
||||||
# Compensates for gravity on the unitree's arms using the arm ik solver
|
# Compensates for gravity on the unitree's arms using the arm ik solver
|
||||||
gravity_compensation: bool = False
|
gravity_compensation: bool = False
|
||||||
|
|
||||||
# Lower-body controller class name, e.g. "GrootLocomotionController" or
|
# Locomotion controller class name, e.g. "GrootLocomotionController",
|
||||||
# "HolosomaLocomotionController". None disables it.
|
# "HolosomaLocomotionController", or "SonicWholeBodyController". None disables it.
|
||||||
controller: str | None = None
|
controller: str | None = None
|
||||||
|
|
||||||
|
# On disconnect (e.g. Ctrl-C), seconds to hold the current pose while ramping joint
|
||||||
|
# stiffness (kp) to zero — a soft, damped settle instead of an instant limp /
|
||||||
|
# free-fall. 0 disables it (immediate zero-torque). Real robot only.
|
||||||
|
graceful_stop_s: float = 1.5
|
||||||
|
|||||||
@@ -0,0 +1,28 @@
|
|||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
|
# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
"""Unitree G1 locomotion controllers (Groot, Holosoma, SONIC)."""
|
||||||
|
|
||||||
|
from .gr00t_locomotion import GrootLocomotionController
|
||||||
|
from .holosoma_locomotion import HolosomaLocomotionController
|
||||||
|
from .sonic_whole_body import SonicRuntime, SonicWholeBodyController
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"GrootLocomotionController",
|
||||||
|
"HolosomaLocomotionController",
|
||||||
|
"SonicRuntime",
|
||||||
|
"SonicWholeBodyController",
|
||||||
|
]
|
||||||
+31
-5
@@ -14,20 +14,29 @@
|
|||||||
# See the License for the specific language governing permissions and
|
# See the License for the specific language governing permissions and
|
||||||
# limitations under the License.
|
# limitations under the License.
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
from collections import deque
|
from collections import deque
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import onnxruntime as ort
|
|
||||||
from huggingface_hub import hf_hub_download
|
from huggingface_hub import hf_hub_download
|
||||||
|
|
||||||
from .g1_utils import (
|
from lerobot.utils.import_utils import _onnxruntime_available, require_package
|
||||||
|
|
||||||
|
from ..g1_utils import (
|
||||||
REMOTE_AXES,
|
REMOTE_AXES,
|
||||||
REMOTE_BUTTONS,
|
REMOTE_BUTTONS,
|
||||||
G1_29_JointIndex,
|
G1_29_JointIndex,
|
||||||
get_gravity_orientation,
|
get_gravity_orientation,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING or _onnxruntime_available:
|
||||||
|
import onnxruntime as ort
|
||||||
|
else:
|
||||||
|
ort = None
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
@@ -68,9 +77,15 @@ def load_groot_policies(
|
|||||||
filename="GR00T-WholeBodyControl-Walk.onnx",
|
filename="GR00T-WholeBodyControl-Walk.onnx",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Load ONNX policies
|
# Load ONNX policies with a capped thread pool. GR00T runs at 50 Hz in a
|
||||||
policy_balance = ort.InferenceSession(balance_path)
|
# background thread alongside the (torch) upper-body policy, IK and sim; letting
|
||||||
policy_walk = ort.InferenceSession(walk_path)
|
# ORT grab every core starves those and makes the whole rollout stutter. These
|
||||||
|
# are small MLPs, so 1 thread is both enough and lowest-latency.
|
||||||
|
from .sonic_pipeline import make_ort_session_options
|
||||||
|
|
||||||
|
so = make_ort_session_options(intra_op_num_threads=1, inter_op_num_threads=1)
|
||||||
|
policy_balance = ort.InferenceSession(balance_path, sess_options=so)
|
||||||
|
policy_walk = ort.InferenceSession(walk_path, sess_options=so)
|
||||||
|
|
||||||
logger.info("GR00T policies loaded successfully")
|
logger.info("GR00T policies loaded successfully")
|
||||||
|
|
||||||
@@ -83,6 +98,7 @@ class GrootLocomotionController:
|
|||||||
control_dt = CONTROL_DT # Expose for unitree_g1.py
|
control_dt = CONTROL_DT # Expose for unitree_g1.py
|
||||||
|
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
|
require_package("onnxruntime", extra="unitree_g1")
|
||||||
# Load policies
|
# Load policies
|
||||||
self.policy_balance, self.policy_walk = load_groot_policies()
|
self.policy_balance, self.policy_walk = load_groot_policies()
|
||||||
|
|
||||||
@@ -196,6 +212,16 @@ class GrootLocomotionController:
|
|||||||
# Transform action back to target joint positions
|
# Transform action back to target joint positions
|
||||||
target_dof_pos_15 = GROOT_DEFAULT_ANGLES[:15] + self.groot_action * ACTION_SCALE
|
target_dof_pos_15 = GROOT_DEFAULT_ANGLES[:15] + self.groot_action * ACTION_SCALE
|
||||||
|
|
||||||
|
# Waist override: an external upper-body IK can command the 3 waist joints
|
||||||
|
# (indices 12/13/14) via ``kWaist{Yaw,Roll,Pitch}.q`` in the action dict. When
|
||||||
|
# present, we substitute the balance policy's waist target so the torso tracks
|
||||||
|
# the IK while the policy keeps only the legs balanced. Single-publisher stays
|
||||||
|
# intact (this thread still owns joints 0-14).
|
||||||
|
for idx in (G1_29_JointIndex.kWaistYaw, G1_29_JointIndex.kWaistRoll, G1_29_JointIndex.kWaistPitch):
|
||||||
|
key = f"{idx.name}.q"
|
||||||
|
if key in action and action[key] is not None:
|
||||||
|
target_dof_pos_15[idx.value] = float(action[key])
|
||||||
|
|
||||||
# Build action dict
|
# Build action dict
|
||||||
action_dict = {}
|
action_dict = {}
|
||||||
for i in range(15):
|
for i in range(15):
|
||||||
+18
-3
@@ -14,21 +14,34 @@
|
|||||||
# See the License for the specific language governing permissions and
|
# See the License for the specific language governing permissions and
|
||||||
# limitations under the License.
|
# limitations under the License.
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import onnx
|
|
||||||
import onnxruntime as ort
|
|
||||||
from huggingface_hub import hf_hub_download
|
from huggingface_hub import hf_hub_download
|
||||||
|
|
||||||
from .g1_utils import (
|
from lerobot.utils.import_utils import _onnx_available, _onnxruntime_available, require_package
|
||||||
|
|
||||||
|
from ..g1_utils import (
|
||||||
REMOTE_AXES,
|
REMOTE_AXES,
|
||||||
G1_29_JointArmIndex,
|
G1_29_JointArmIndex,
|
||||||
G1_29_JointIndex,
|
G1_29_JointIndex,
|
||||||
get_gravity_orientation,
|
get_gravity_orientation,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING or _onnxruntime_available:
|
||||||
|
import onnxruntime as ort
|
||||||
|
else:
|
||||||
|
ort = None
|
||||||
|
|
||||||
|
if TYPE_CHECKING or _onnx_available:
|
||||||
|
import onnx
|
||||||
|
else:
|
||||||
|
onnx = None
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
DEFAULT_ANGLES = np.zeros(29, dtype=np.float32)
|
DEFAULT_ANGLES = np.zeros(29, dtype=np.float32)
|
||||||
@@ -101,6 +114,8 @@ class HolosomaLocomotionController:
|
|||||||
control_dt = CONTROL_DT # Expose for unitree_g1.py
|
control_dt = CONTROL_DT # Expose for unitree_g1.py
|
||||||
|
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
|
require_package("onnxruntime", extra="unitree_g1")
|
||||||
|
require_package("onnx", extra="unitree_g1")
|
||||||
# Load policy and gains
|
# Load policy and gains
|
||||||
self.policy, self.kp, self.kd = load_policy()
|
self.policy, self.kp, self.kd = load_policy()
|
||||||
|
|
||||||
@@ -0,0 +1,670 @@
|
|||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
|
# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
"""SONIC encoder/decoder pipeline for the Unitree G1 whole-body controller.
|
||||||
|
|
||||||
|
Pure-Python/ONNX re-implementation of the reference-tracking half of NVIDIA's SONIC
|
||||||
|
deploy stack (mirrors ``g1_deploy_onnx_ref.cpp``). Given a reference motion buffer
|
||||||
|
(joint targets + body orientation per frame) it produces 50 Hz joint-position targets
|
||||||
|
for the robot's PD controller. The upstream *motion planner* is intentionally absent:
|
||||||
|
here the reference is supplied directly by the caller (e.g. a 34-D OpenHLM / pi0.5 VLA
|
||||||
|
command per tick, in ``sonic_whole_body.py``).
|
||||||
|
|
||||||
|
Two cooperating ONNX models:
|
||||||
|
* **encoder** – compresses the reference window into a 64-D latent ``token``
|
||||||
|
(refreshed every ``ENCODER_UPDATE_EVERY`` ticks).
|
||||||
|
* **decoder** – every tick, maps the token + recent proprioception history to a
|
||||||
|
residual action that is scaled and added to ``DEFAULT_ANGLES``.
|
||||||
|
|
||||||
|
Index spaces: joints exist in two orderings — **IsaacLab** (policy/training order)
|
||||||
|
and **MuJoCo** (deploy order). ``ISAACLAB_TO_MUJOCO`` / ``MUJOCO_TO_ISAACLAB`` convert
|
||||||
|
between them. Quaternions are scalar-first ``(w, x, y, z)``.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import threading
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
from lerobot.utils.import_utils import _onnxruntime_available
|
||||||
|
|
||||||
|
from ..g1_utils import (
|
||||||
|
ISAACLAB_TO_MUJOCO,
|
||||||
|
MUJOCO_TO_ISAACLAB,
|
||||||
|
G1_29_JointIndex,
|
||||||
|
get_gravity_orientation,
|
||||||
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING or _onnxruntime_available:
|
||||||
|
import onnxruntime as ort
|
||||||
|
else:
|
||||||
|
ort = None
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# ── Constants ────────────────────────────────────────────────────────────────
|
||||||
|
# Robot/motor physical constants and the joint-order permutation tables. All
|
||||||
|
# 29-vectors are in IsaacLab joint order unless the name says ``_MUJOCO``.
|
||||||
|
|
||||||
|
# Nominal standing pose (rad), 29 joints in IsaacLab order. Actions are residuals
|
||||||
|
# added on top of this; also used as the planner/encoder standing reference.
|
||||||
|
DEFAULT_ANGLES = np.array(
|
||||||
|
[
|
||||||
|
-0.312,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
0.669,
|
||||||
|
-0.363,
|
||||||
|
0.0,
|
||||||
|
-0.312,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
0.669,
|
||||||
|
-0.363,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
0.2,
|
||||||
|
0.2,
|
||||||
|
0.0,
|
||||||
|
0.6,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
0.2,
|
||||||
|
-0.2,
|
||||||
|
0.0,
|
||||||
|
0.6,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
0.0,
|
||||||
|
],
|
||||||
|
dtype=np.float32,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Per-motor-type parameters used to derive action scaling and PD gains. Keys are
|
||||||
|
# Unitree motor model names; ARMATURE = rotor inertia, EFFORT = torque limit (N·m).
|
||||||
|
NATURAL_FREQ = 10.0 * 2.0 * np.pi # target closed-loop stiffness bandwidth (rad/s)
|
||||||
|
ARMATURE = {"5020": 0.003609725, "7520_14": 0.010177520, "7520_22": 0.025101925, "4010": 0.00425}
|
||||||
|
EFFORT = {"5020": 25.0, "7520_14": 88.0, "7520_22": 139.0, "4010": 5.0}
|
||||||
|
|
||||||
|
|
||||||
|
def _action_scale(k):
|
||||||
|
"""Per-motor residual-action scale (maps policy output to joint-angle delta)."""
|
||||||
|
return 0.25 * EFFORT[k] / (ARMATURE[k] * NATURAL_FREQ**2)
|
||||||
|
|
||||||
|
|
||||||
|
# Per-joint motor model (IsaacLab order): legs, waist, then arms. Single source of
|
||||||
|
# truth for both ACTION_SCALE and compute_kp_kd().
|
||||||
|
MOTOR_MODELS = (
|
||||||
|
["7520_22", "7520_22", "7520_14", "7520_22", "5020", "5020"] * 2
|
||||||
|
+ ["7520_14", "5020", "5020"]
|
||||||
|
+ ["5020", "5020", "5020", "5020", "5020", "4010", "4010"] * 2
|
||||||
|
)
|
||||||
|
ACTION_SCALE = np.array([_action_scale(k) for k in MOTOR_MODELS], dtype=np.float32) # (29,) IsaacLab order
|
||||||
|
|
||||||
|
CONTROL_DT = 0.02 # 50 Hz control period (s)
|
||||||
|
DEFAULT_HEIGHT = 0.788740 # nominal pelvis height (m)
|
||||||
|
TOKEN_DIM = 64 # encoder latent size
|
||||||
|
ENCODER_UPDATE_EVERY = 5 # refresh the encoder token every N ticks (decoder runs every tick)
|
||||||
|
DEBUG_PRINT_EVERY = 100 # ticks between debug prints
|
||||||
|
|
||||||
|
|
||||||
|
def _to_mujoco(a):
|
||||||
|
"""Apply the ``MUJOCO_TO_ISAACLAB`` gather to a 29-vector (deploy-order reorder).
|
||||||
|
|
||||||
|
NOTE: this returns ``a[MUJOCO_TO_ISAACLAB]``. The ``_mj`` suffixes and the exact
|
||||||
|
permutation direction throughout this module are a fixed convention validated
|
||||||
|
against the deployed SONIC ONNX policy (the encoder/decoder consume vectors in
|
||||||
|
this order). Do not "correct" the table or rename toward the opposite direction
|
||||||
|
without re-validating on hardware — the labels are historical, the ordering is
|
||||||
|
load-bearing.
|
||||||
|
"""
|
||||||
|
return a[MUJOCO_TO_ISAACLAB]
|
||||||
|
|
||||||
|
|
||||||
|
DEFAULT_ANGLES_MUJOCO = _to_mujoco(DEFAULT_ANGLES)
|
||||||
|
ENCODER_STANDING_REF = DEFAULT_ANGLES.copy()
|
||||||
|
|
||||||
|
# Joint-index subsets (IsaacLab order) used to slice encoder observations.
|
||||||
|
LOWER_BODY_IL = np.array([0, 3, 6, 9, 13, 17, 1, 4, 7, 10, 14, 18], dtype=np.int32) # 12 leg joints
|
||||||
|
WRIST_IL = np.array([23, 24, 25, 26, 27, 28], dtype=np.int32) # 6 wrist joints
|
||||||
|
VR_TARGET_DEF = np.zeros(9, dtype=np.float32) # 3-point VR position targets (mode 1)
|
||||||
|
VR_ORN_DEF = np.array([1, 0, 0, 0, 1, 0, 0, 0, 1, 0, 0, 0], dtype=np.float32) # VR orn targets (mode 1)
|
||||||
|
SMPL_DEF = np.zeros(720, dtype=np.float32) # SMPL whole-body window default (mode 2)
|
||||||
|
|
||||||
|
# ── PD gains ─────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def compute_kp_kd():
|
||||||
|
"""Derive per-joint PD gains (kp, kd) from motor armature and target bandwidth.
|
||||||
|
|
||||||
|
Ankle and waist joints get a x2 factor for extra stiffness. Returns two
|
||||||
|
(29,) float32 arrays in IsaacLab joint order.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def s(k):
|
||||||
|
return ARMATURE[k] * NATURAL_FREQ**2
|
||||||
|
|
||||||
|
def d(k):
|
||||||
|
return 2.0 * 2.0 * ARMATURE[k] * NATURAL_FREQ
|
||||||
|
|
||||||
|
_double = {4, 5, 10, 11, 13, 14} # ankle + waist indices with factor 2
|
||||||
|
kp = np.array([2 * s(k) if i in _double else s(k) for i, k in enumerate(MOTOR_MODELS)], dtype=np.float32)
|
||||||
|
kd = np.array([2 * d(k) if i in _double else d(k) for i, k in enumerate(MOTOR_MODELS)], dtype=np.float32)
|
||||||
|
return kp, kd
|
||||||
|
|
||||||
|
|
||||||
|
_kp_kd = compute_kp_kd # backward-compatible alias
|
||||||
|
|
||||||
|
|
||||||
|
# ── Quaternion helpers ────────────────────────────────────────────────────────
|
||||||
|
# All quaternions are scalar-first (w, x, y, z). "heading" = yaw-only quaternion.
|
||||||
|
|
||||||
|
|
||||||
|
def quat_conj(q):
|
||||||
|
"""Quaternion conjugate (inverse for unit quaternions)."""
|
||||||
|
return np.array([q[0], -q[1], -q[2], -q[3]], dtype=np.float32)
|
||||||
|
|
||||||
|
|
||||||
|
def quat_mul(q1, q2):
|
||||||
|
"""Hamilton product ``q1 ⊗ q2``."""
|
||||||
|
w1, x1, y1, z1 = q1
|
||||||
|
w2, x2, y2, z2 = q2
|
||||||
|
return np.array(
|
||||||
|
[
|
||||||
|
w1 * w2 - x1 * x2 - y1 * y2 - z1 * z2,
|
||||||
|
w1 * x2 + x1 * w2 + y1 * z2 - z1 * y2,
|
||||||
|
w1 * y2 - x1 * z2 + y1 * w2 + z1 * x2,
|
||||||
|
w1 * z2 + x1 * y2 - y1 * x2 + z1 * w2,
|
||||||
|
],
|
||||||
|
dtype=np.float32,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def quat_to_6d(q):
|
||||||
|
"""Quaternion → 6-D rotation representation (first two rotated basis rows)."""
|
||||||
|
w, x, y, z = q
|
||||||
|
return np.array(
|
||||||
|
[
|
||||||
|
1 - 2 * (y * y + z * z),
|
||||||
|
2 * (x * y - z * w),
|
||||||
|
2 * (x * y + z * w),
|
||||||
|
1 - 2 * (x * x + z * z),
|
||||||
|
2 * (x * z - y * w),
|
||||||
|
2 * (y * z + x * w),
|
||||||
|
],
|
||||||
|
dtype=np.float32,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def calc_heading(q):
|
||||||
|
"""Extract the yaw (heading) angle in radians from a quaternion."""
|
||||||
|
w, x, y, z = q
|
||||||
|
return float(np.arctan2(2 * (x * y + w * z), 1 - 2 * (y * y + z * z)))
|
||||||
|
|
||||||
|
|
||||||
|
def heading_quat(q, sign=1.0):
|
||||||
|
"""Yaw-only quaternion for ``q``'s heading (``sign=-1`` gives its inverse)."""
|
||||||
|
a = sign * calc_heading(q) / 2.0
|
||||||
|
return np.array([np.cos(a), 0, 0, np.sin(a)], dtype=np.float64)
|
||||||
|
|
||||||
|
|
||||||
|
def heading_quat_inv(q):
|
||||||
|
"""Inverse yaw-only quaternion for ``q``'s heading."""
|
||||||
|
return heading_quat(q, -1.0)
|
||||||
|
|
||||||
|
|
||||||
|
def quat_slerp(q0, q1, t):
|
||||||
|
"""Spherical linear interpolation between two quaternions (scalar ``t``)."""
|
||||||
|
q0 = q0 / (np.linalg.norm(q0) + 1e-12)
|
||||||
|
q1 = q1 / (np.linalg.norm(q1) + 1e-12)
|
||||||
|
dot = float(np.dot(q0, q1))
|
||||||
|
if dot < 0:
|
||||||
|
q1, dot = -q1, -dot
|
||||||
|
dot = min(dot, 1.0)
|
||||||
|
if dot > 0.9995:
|
||||||
|
r = q0 + t * (q1 - q0)
|
||||||
|
return r / (np.linalg.norm(r) + 1e-12)
|
||||||
|
th = np.arccos(dot)
|
||||||
|
st = np.sin(th)
|
||||||
|
return (np.sin((1 - t) * th) / st) * q0 + (np.sin(t * th) / st) * q1
|
||||||
|
|
||||||
|
|
||||||
|
def quat_slerp_batch(q0, q1, t):
|
||||||
|
"""Vectorized slerp over arrays of quaternions with a per-row parameter ``t``."""
|
||||||
|
q0 = q0 / (np.linalg.norm(q0, axis=1, keepdims=True) + 1e-12)
|
||||||
|
q1 = q1 / (np.linalg.norm(q1, axis=1, keepdims=True) + 1e-12)
|
||||||
|
dot = np.sum(q0 * q1, axis=1)
|
||||||
|
neg = dot < 0
|
||||||
|
q1 = q1.copy()
|
||||||
|
q1[neg] = -q1[neg]
|
||||||
|
dot[neg] = -dot[neg]
|
||||||
|
dot = np.clip(dot, -1, 1)
|
||||||
|
lin = dot > 0.9995
|
||||||
|
th = np.arccos(dot)
|
||||||
|
st = np.where(np.sin(th) == 0, 1, np.sin(th))
|
||||||
|
c0 = np.sin((1 - t) * th) / st
|
||||||
|
c1 = np.sin(t * th) / st
|
||||||
|
c0[lin] = 1 - t[lin]
|
||||||
|
c1[lin] = t[lin]
|
||||||
|
r = c0[:, None] * q0 + c1[:, None] * q1
|
||||||
|
return r / (np.linalg.norm(r, axis=1, keepdims=True) + 1e-12)
|
||||||
|
|
||||||
|
|
||||||
|
def ort_providers(force_cpu: bool = False) -> list[str]:
|
||||||
|
"""Prefer CUDA for enc/dec/planner (matches deploy when onnxruntime-gpu is installed)."""
|
||||||
|
avail = ort.get_available_providers()
|
||||||
|
if not force_cpu and "CUDAExecutionProvider" in avail:
|
||||||
|
return ["CUDAExecutionProvider", "CPUExecutionProvider"]
|
||||||
|
return ["CPUExecutionProvider"]
|
||||||
|
|
||||||
|
|
||||||
|
def make_ort_session_options(intra_op_num_threads: int | None = None,
|
||||||
|
inter_op_num_threads: int | None = None):
|
||||||
|
"""Build ONNX Runtime SessionOptions (quiet logging).
|
||||||
|
|
||||||
|
Pass thread counts to cap ORT's CPU pool. These tiny MLP policies are latency-
|
||||||
|
bound, not throughput-bound, so letting ORT grab every core just starves the
|
||||||
|
real-time control loop / torch policy / IK solver and causes stutter. 1 intra +
|
||||||
|
1 inter thread is plenty and lowest-latency for a per-step MLP inference.
|
||||||
|
"""
|
||||||
|
so = ort.SessionOptions()
|
||||||
|
so.log_severity_level = 3
|
||||||
|
if intra_op_num_threads is not None:
|
||||||
|
so.intra_op_num_threads = intra_op_num_threads
|
||||||
|
if inter_op_num_threads is not None:
|
||||||
|
so.inter_op_num_threads = inter_op_num_threads
|
||||||
|
return so
|
||||||
|
|
||||||
|
|
||||||
|
# ── Encoder / Decoder ─────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class StandingEncoderDecoder:
|
||||||
|
"""Runs the encoder + decoder ONNX models and owns the proprioception history.
|
||||||
|
|
||||||
|
Each tick it appends the latest robot state to 10-frame history buffers, builds
|
||||||
|
the encoder observation (1762-D, layout depends on ``encode_mode``) to refresh
|
||||||
|
the 64-D ``token``, then builds the decoder observation (994-D) and maps
|
||||||
|
``token + history`` to a residual action added onto ``DEFAULT_ANGLES``.
|
||||||
|
|
||||||
|
``PlannerController`` subclasses this to source the reference from a live,
|
||||||
|
planner-generated motion buffer instead of a fixed standing pose.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, encoder, decoder):
|
||||||
|
self.encoder, self.decoder = encoder, decoder
|
||||||
|
self.encoder_input = encoder.get_inputs()[0].name
|
||||||
|
self.decoder_input = decoder.get_inputs()[0].name
|
||||||
|
enc_dim = int(encoder.get_inputs()[0].shape[1])
|
||||||
|
dec_dim = int(decoder.get_inputs()[0].shape[1])
|
||||||
|
if enc_dim != 1762 or dec_dim != 994:
|
||||||
|
raise RuntimeError(f"Unexpected dims encoder={enc_dim}, decoder={dec_dim}")
|
||||||
|
self.token = np.zeros(TOKEN_DIM, np.float32)
|
||||||
|
self.last_action_mj = np.zeros(29, np.float32)
|
||||||
|
self.h_q_mj = [np.zeros(29, np.float32)] * 10
|
||||||
|
self.h_dq_mj = [np.zeros(29, np.float32)] * 10
|
||||||
|
self.h_ang = [np.zeros(3, np.float32)] * 10
|
||||||
|
self.h_act_mj = [np.zeros(29, np.float32)] * 10
|
||||||
|
self.h_quat = [np.array([1, 0, 0, 0], np.float32)] * 10
|
||||||
|
self.init_base_quat = np.array([1, 0, 0, 0], np.float32)
|
||||||
|
self.init_ref_quat = np.array([1, 0, 0, 0], np.float32)
|
||||||
|
self._heading_init = False
|
||||||
|
self.encode_mode = 0
|
||||||
|
self.vr_3point_local_target = VR_TARGET_DEF.copy()
|
||||||
|
self.vr_3point_local_orn_target = VR_ORN_DEF.copy()
|
||||||
|
self.smpl_joints_10frame_step1 = SMPL_DEF.copy()
|
||||||
|
# Optional per-frame SMPL root orientation (wxyz) for the mode-2 anchor.
|
||||||
|
# When None, the anchor falls back to the planner reference body quat.
|
||||||
|
self.smpl_root_quat = None
|
||||||
|
self.set_zero_reference()
|
||||||
|
|
||||||
|
def reset(self):
|
||||||
|
"""Clear the token, 10-frame proprioception history and heading init.
|
||||||
|
|
||||||
|
``UnitreeG1.reset()`` relies on this so the first decoder outputs of a new
|
||||||
|
episode are not contaminated by the previous episode's state.
|
||||||
|
"""
|
||||||
|
self.token = np.zeros(TOKEN_DIM, np.float32)
|
||||||
|
self.last_action_mj = np.zeros(29, np.float32)
|
||||||
|
self.h_q_mj = [np.zeros(29, np.float32)] * 10
|
||||||
|
self.h_dq_mj = [np.zeros(29, np.float32)] * 10
|
||||||
|
self.h_ang = [np.zeros(3, np.float32)] * 10
|
||||||
|
self.h_act_mj = [np.zeros(29, np.float32)] * 10
|
||||||
|
self.h_quat = [np.array([1, 0, 0, 0], np.float32)] * 10
|
||||||
|
self.init_base_quat = np.array([1, 0, 0, 0], np.float32)
|
||||||
|
self.init_ref_quat = np.array([1, 0, 0, 0], np.float32)
|
||||||
|
self._heading_init = False
|
||||||
|
|
||||||
|
def update_history(self, q, dq, ang, quat):
|
||||||
|
"""Push the latest proprioception (pos/vel/gyro/orientation) into the 10-frame buffers."""
|
||||||
|
quat = quat / (np.linalg.norm(quat) + 1e-8)
|
||||||
|
q_mj = _to_mujoco(q)
|
||||||
|
dq_mj = _to_mujoco(dq)
|
||||||
|
self.h_q_mj = [q_mj - DEFAULT_ANGLES_MUJOCO] + self.h_q_mj[:-1]
|
||||||
|
self.h_dq_mj = [dq_mj] + self.h_dq_mj[:-1]
|
||||||
|
self.h_ang = [ang.copy()] + self.h_ang[:-1]
|
||||||
|
self.h_act_mj = [self.last_action_mj.copy()] + self.h_act_mj[:-1]
|
||||||
|
self.h_quat = [quat.copy()] + self.h_quat[:-1]
|
||||||
|
if not self._heading_init:
|
||||||
|
self.init_base_quat = quat.copy()
|
||||||
|
self._heading_init = True
|
||||||
|
|
||||||
|
def _heading_quat(self, q):
|
||||||
|
h = calc_heading(q) / 2.0
|
||||||
|
return np.array([np.cos(h), 0, 0, np.sin(h)], np.float32)
|
||||||
|
|
||||||
|
def _heading_quat_inv(self, q):
|
||||||
|
h = calc_heading(q) / 2.0
|
||||||
|
return np.array([np.cos(-h), 0, 0, np.sin(-h)], np.float32)
|
||||||
|
|
||||||
|
def _anchor_6d(self, base_quat, ref_quat=None):
|
||||||
|
"""6-D orientation error between the robot base and the (heading-aligned) reference."""
|
||||||
|
if ref_quat is None:
|
||||||
|
ref_quat = self.init_ref_quat
|
||||||
|
delta = quat_mul(self._heading_quat(self.init_base_quat), self._heading_quat_inv(self.init_ref_quat))
|
||||||
|
new_ref = quat_mul(delta, ref_quat)
|
||||||
|
return quat_to_6d(quat_mul(quat_conj(base_quat), new_ref))
|
||||||
|
|
||||||
|
def set_zero_reference(self):
|
||||||
|
"""Initialize the reference to a single standing frame (used before a plan exists)."""
|
||||||
|
self.motion_joint_positions = [ENCODER_STANDING_REF.copy()]
|
||||||
|
self.motion_joint_velocities = [np.zeros(29, np.float32)]
|
||||||
|
self.motion_body_quats = [np.array([1, 0, 0, 0], np.float32)]
|
||||||
|
self.motion_body_z = [DEFAULT_HEIGHT]
|
||||||
|
self.motion_timesteps = 1
|
||||||
|
self.freeze_ref_frame = 0
|
||||||
|
self.init_ref_quat = self.motion_body_quats[0].copy()
|
||||||
|
|
||||||
|
def build_encoder_obs(self):
|
||||||
|
"""Assemble the 1762-D encoder input; slot layout depends on ``encode_mode``.
|
||||||
|
|
||||||
|
mode 0 = locomotion (ref joint pos + anchor), 1 = 3-point VR teleop
|
||||||
|
(lower-body ref + VR targets), 2 = SMPL whole-body window + anchor/wrist.
|
||||||
|
"""
|
||||||
|
obs = np.zeros(1762, np.float32)
|
||||||
|
obs[0] = float(self.encode_mode)
|
||||||
|
rf = min(self.freeze_ref_frame, self.motion_timesteps - 1)
|
||||||
|
ref_pos, ref_quat = self.motion_joint_positions[rf], self.motion_body_quats[rf]
|
||||||
|
if self.encode_mode == 0:
|
||||||
|
for f in range(10):
|
||||||
|
obs[4 + 29 * f : 4 + 29 * (f + 1)] = ref_pos
|
||||||
|
obs[601 + 6 * f : 601 + 6 * (f + 1)] = self._anchor_6d(self.h_quat[0], ref_quat)
|
||||||
|
elif self.encode_mode == 1:
|
||||||
|
ref_lower = ref_pos[LOWER_BODY_IL]
|
||||||
|
for f in range(10):
|
||||||
|
obs[661 + 12 * f : 661 + 12 * (f + 1)] = ref_lower
|
||||||
|
obs[901:910] = self.vr_3point_local_target
|
||||||
|
obs[910:922] = self.vr_3point_local_orn_target
|
||||||
|
obs[595:601] = self._anchor_6d(self.h_quat[0], ref_quat)
|
||||||
|
elif self.encode_mode == 2:
|
||||||
|
# Prefer the SMPL clip/stream root orientation for the anchor; fall
|
||||||
|
# back to the planner reference body quat when no root is provided.
|
||||||
|
anchor_ref = self.smpl_root_quat if self.smpl_root_quat is not None else ref_quat
|
||||||
|
obs[922:1642] = self.smpl_joints_10frame_step1
|
||||||
|
for f in range(10):
|
||||||
|
obs[1642 + 6 * f : 1642 + 6 * (f + 1)] = self._anchor_6d(self.h_quat[0], anchor_ref)
|
||||||
|
obs[1702 + 6 * f : 1702 + 6 * (f + 1)] = ref_pos[WRIST_IL]
|
||||||
|
else:
|
||||||
|
raise RuntimeError(f"Unsupported encoder mode: {self.encode_mode}")
|
||||||
|
return obs
|
||||||
|
|
||||||
|
def build_decoder_obs(self):
|
||||||
|
"""Assemble the 994-D decoder input: token + 10-frame proprioception history + gravity."""
|
||||||
|
obs = np.zeros(994, np.float32)
|
||||||
|
off = 0
|
||||||
|
obs[off : off + 64] = self.token
|
||||||
|
off += 64
|
||||||
|
for h, sz in [
|
||||||
|
(list(reversed(self.h_ang)), 3),
|
||||||
|
(list(reversed(self.h_q_mj)), 29),
|
||||||
|
(list(reversed(self.h_dq_mj)), 29),
|
||||||
|
(list(reversed(self.h_act_mj)), 29),
|
||||||
|
]:
|
||||||
|
for f in range(10):
|
||||||
|
obs[off : off + sz] = h[f]
|
||||||
|
off += sz
|
||||||
|
for q in reversed(self.h_quat):
|
||||||
|
obs[off : off + 3] = get_gravity_orientation(q)
|
||||||
|
off += 3
|
||||||
|
assert off == 994, f"Decoder obs mismatch: {off}"
|
||||||
|
return obs
|
||||||
|
|
||||||
|
def run_encoder(self):
|
||||||
|
"""Run the encoder ONNX model and return the fresh 64-D token."""
|
||||||
|
return (
|
||||||
|
self.encoder.run(None, {self.encoder_input: self.build_encoder_obs().reshape(1, -1)})[0]
|
||||||
|
.squeeze()
|
||||||
|
.astype(np.float32)
|
||||||
|
)
|
||||||
|
|
||||||
|
def step(self, robot_obs, update_encoder, debug=False):
|
||||||
|
"""One control tick: read robot obs, (optionally) re-encode, decode → joint targets.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
robot_obs: dict with ``<joint>.q``/``.dq`` and ``imu.*`` fields.
|
||||||
|
update_encoder: refresh the token this tick (else reuse the cached one).
|
||||||
|
debug: print action/delta norms.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict of ``<joint>.q`` target positions (rad) in IsaacLab joint order.
|
||||||
|
"""
|
||||||
|
jnames = [m.name for m in G1_29_JointIndex]
|
||||||
|
q = np.array(
|
||||||
|
[
|
||||||
|
robot_obs.get(f"{n}.q", DEFAULT_ANGLES[m.value])
|
||||||
|
for m, n in zip(G1_29_JointIndex, jnames, strict=False)
|
||||||
|
],
|
||||||
|
np.float32,
|
||||||
|
)
|
||||||
|
dq = np.array([robot_obs.get(f"{n}.dq", 0.0) for n in jnames], np.float32)
|
||||||
|
quat = np.array(
|
||||||
|
[
|
||||||
|
robot_obs.get("imu.quat.w", 1),
|
||||||
|
robot_obs.get("imu.quat.x", 0),
|
||||||
|
robot_obs.get("imu.quat.y", 0),
|
||||||
|
robot_obs.get("imu.quat.z", 0),
|
||||||
|
],
|
||||||
|
np.float32,
|
||||||
|
)
|
||||||
|
ang = np.array([robot_obs.get(f"imu.gyro.{a}", 0) for a in "xyz"], np.float32)
|
||||||
|
self.update_history(q, dq, ang, quat)
|
||||||
|
if update_encoder:
|
||||||
|
self.token = self.run_encoder()
|
||||||
|
action_mj = (
|
||||||
|
self.decoder.run(None, {self.decoder_input: self.build_decoder_obs().reshape(1, -1)})[0]
|
||||||
|
.squeeze()
|
||||||
|
.astype(np.float32)
|
||||||
|
)
|
||||||
|
self.last_action_mj = action_mj.copy()
|
||||||
|
target = DEFAULT_ANGLES + action_mj[ISAACLAB_TO_MUJOCO] * ACTION_SCALE
|
||||||
|
if debug:
|
||||||
|
delta = target - q
|
||||||
|
logger.debug(
|
||||||
|
"token_norm=%.4f action_norm=%.4f delta_max=%.4f delta_rms=%.4f",
|
||||||
|
np.linalg.norm(self.token),
|
||||||
|
np.linalg.norm(action_mj),
|
||||||
|
np.max(np.abs(delta)),
|
||||||
|
np.sqrt(np.mean(delta**2)),
|
||||||
|
)
|
||||||
|
return {f"{m.name}.q": float(target[m.value]) for m in G1_29_JointIndex}
|
||||||
|
|
||||||
|
|
||||||
|
class PlannerController(StandingEncoderDecoder):
|
||||||
|
"""Encoder/decoder driven by a caller-supplied, rolling motion buffer.
|
||||||
|
|
||||||
|
Extends ``StandingEncoderDecoder`` so the reference comes from a motion buffer
|
||||||
|
(a lookahead window with per-frame velocities) instead of a single fixed pose,
|
||||||
|
and handles heading re-initialization on the first frame / after a reset.
|
||||||
|
``motion_lock`` guards the buffer, which the whole-body controller rewrites each
|
||||||
|
tick from the incoming command. The class name is retained for continuity with
|
||||||
|
the SONIC reference; no motion planner is involved.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, encoder, decoder):
|
||||||
|
super().__init__(encoder, decoder)
|
||||||
|
self.ref_cursor = 0
|
||||||
|
self.motion_timesteps = 0
|
||||||
|
self.motion_joint_positions = np.zeros((1500, 29), np.float64)
|
||||||
|
self.motion_joint_velocities = np.zeros((1500, 29), np.float64)
|
||||||
|
self.motion_body_quats = np.zeros((1500, 4), np.float64)
|
||||||
|
self.motion_body_quats[:, 0] = 1.0
|
||||||
|
self.motion_body_pos = np.zeros((1500, 3), np.float64)
|
||||||
|
self.init_ref_quat = np.array([1, 0, 0, 0], np.float64)
|
||||||
|
self.heading_init_base_quat = np.array([1, 0, 0, 0], np.float64)
|
||||||
|
self.delta_heading = 0.0
|
||||||
|
self.reinit_heading = False
|
||||||
|
self.playing = self.first_motion = False
|
||||||
|
self.motion_lock = threading.Lock()
|
||||||
|
|
||||||
|
def reset(self):
|
||||||
|
"""Full reset: clear enc/dec state (super) plus the motion buffer and heading.
|
||||||
|
|
||||||
|
Forces a heading re-init on the next ``step`` so the reference frame is
|
||||||
|
re-latched to the post-reset robot orientation.
|
||||||
|
"""
|
||||||
|
super().reset()
|
||||||
|
with self.motion_lock:
|
||||||
|
self.ref_cursor = 0
|
||||||
|
self.motion_timesteps = 0
|
||||||
|
self.motion_joint_positions[:] = 0.0
|
||||||
|
self.motion_joint_velocities[:] = 0.0
|
||||||
|
self.motion_body_quats[:] = 0.0
|
||||||
|
self.motion_body_quats[:, 0] = 1.0
|
||||||
|
self.motion_body_pos[:] = 0.0
|
||||||
|
self.init_ref_quat = np.array([1, 0, 0, 0], np.float64)
|
||||||
|
self.heading_init_base_quat = np.array([1, 0, 0, 0], np.float64)
|
||||||
|
self.delta_heading = 0.0
|
||||||
|
self.first_motion = False
|
||||||
|
self.playing = False
|
||||||
|
self.reinit_heading = True
|
||||||
|
|
||||||
|
def _heading_apply_delta(self):
|
||||||
|
"""Heading correction quaternion (init base-vs-ref heading + operator ``delta_heading``)."""
|
||||||
|
delta = quat_mul(
|
||||||
|
heading_quat(self.heading_init_base_quat).astype(np.float32),
|
||||||
|
heading_quat_inv(self.init_ref_quat).astype(np.float32),
|
||||||
|
)
|
||||||
|
if self.delta_heading:
|
||||||
|
h = self.delta_heading / 2.0
|
||||||
|
delta = quat_mul(np.array([np.cos(h), 0, 0, np.sin(h)], np.float32), delta)
|
||||||
|
return delta
|
||||||
|
|
||||||
|
def _anchor_6d(self, base_quat, ref_quat=None):
|
||||||
|
"""6-D base-vs-reference orientation error, including the operator heading delta."""
|
||||||
|
if ref_quat is None:
|
||||||
|
ref_quat = self.init_ref_quat
|
||||||
|
new_ref = quat_mul(self._heading_apply_delta(), ref_quat.astype(np.float32))
|
||||||
|
return quat_to_6d(quat_mul(quat_conj(base_quat.astype(np.float32)), new_ref))
|
||||||
|
|
||||||
|
def build_encoder_obs(self):
|
||||||
|
"""Encoder input sourced from the live motion buffer (mode 0/2), lock-protected."""
|
||||||
|
obs = np.zeros(1762, np.float32)
|
||||||
|
obs[0] = float(self.encode_mode)
|
||||||
|
with self.motion_lock:
|
||||||
|
if self.encode_mode == 2:
|
||||||
|
# SMPL whole-body imitation: the 720-dim SMPL window carries the
|
||||||
|
# target pose; the planner reference frame supplies anchor + wrist.
|
||||||
|
rf = min(self.ref_cursor, self.motion_timesteps - 1)
|
||||||
|
ref_pos = self.motion_joint_positions[rf].astype(np.float32)
|
||||||
|
ref_quat = self.motion_body_quats[rf].astype(np.float32)
|
||||||
|
# Prefer the SMPL clip/stream root orientation (if provided) so the
|
||||||
|
# anchor tracks the operator's/clip's heading; else planner ref.
|
||||||
|
if self.smpl_root_quat is not None:
|
||||||
|
ref_quat = np.asarray(self.smpl_root_quat, np.float32)
|
||||||
|
anchor = self._anchor_6d(self.h_quat[0], ref_quat)
|
||||||
|
wrist = ref_pos[WRIST_IL]
|
||||||
|
obs[922:1642] = self.smpl_joints_10frame_step1
|
||||||
|
for f in range(10):
|
||||||
|
obs[1642 + 6 * f : 1642 + 6 * (f + 1)] = anchor
|
||||||
|
obs[1702 + 6 * f : 1702 + 6 * (f + 1)] = wrist
|
||||||
|
return obs
|
||||||
|
if self.encode_mode == 1:
|
||||||
|
# 3-point VR teleop: the upper body tracks the VR wrist/neck targets
|
||||||
|
# while the planner reference supplies the lower body + anchor. Lower
|
||||||
|
# body is per-frame (step 5) like mode 0; the VR targets are current.
|
||||||
|
rf = min(self.ref_cursor, self.motion_timesteps - 1)
|
||||||
|
obs[595:601] = self._anchor_6d(self.h_quat[0], self.motion_body_quats[rf].astype(np.float32))
|
||||||
|
for f in range(10):
|
||||||
|
tf = min(
|
||||||
|
self.ref_cursor + f * 5 if self.playing else self.ref_cursor,
|
||||||
|
self.motion_timesteps - 1,
|
||||||
|
)
|
||||||
|
ref_lower = self.motion_joint_positions[tf].astype(np.float32)[LOWER_BODY_IL]
|
||||||
|
obs[661 + 12 * f : 661 + 12 * (f + 1)] = ref_lower
|
||||||
|
obs[901:910] = self.vr_3point_local_target
|
||||||
|
obs[910:922] = self.vr_3point_local_orn_target
|
||||||
|
return obs
|
||||||
|
for f in range(10):
|
||||||
|
tf = min(
|
||||||
|
self.ref_cursor + f * 5 if self.playing else self.ref_cursor, self.motion_timesteps - 1
|
||||||
|
)
|
||||||
|
obs[4 + 29 * f : 4 + 29 * (f + 1)] = self.motion_joint_positions[tf].astype(np.float32)
|
||||||
|
if self.playing:
|
||||||
|
obs[294 + 29 * f : 294 + 29 * (f + 1)] = self.motion_joint_velocities[tf].astype(
|
||||||
|
np.float32
|
||||||
|
)
|
||||||
|
obs[601 + 6 * f : 601 + 6 * (f + 1)] = self._anchor_6d(
|
||||||
|
self.h_quat[0], self.motion_body_quats[tf].astype(np.float32)
|
||||||
|
)
|
||||||
|
return obs
|
||||||
|
|
||||||
|
def step(self, robot_obs, update_encoder, debug=False):
|
||||||
|
"""Re-init the heading reference on first frame / after a reset, then run the base step."""
|
||||||
|
if robot_obs and (self.first_motion or self.reinit_heading):
|
||||||
|
q = None
|
||||||
|
if "imu.quat.w" in robot_obs:
|
||||||
|
q = np.array(
|
||||||
|
[
|
||||||
|
robot_obs["imu.quat.w"],
|
||||||
|
robot_obs["imu.quat.x"],
|
||||||
|
robot_obs["imu.quat.y"],
|
||||||
|
robot_obs["imu.quat.z"],
|
||||||
|
],
|
||||||
|
np.float64,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
q = robot_obs.get("imu.quaternion")
|
||||||
|
if q is not None:
|
||||||
|
q = np.array(q, np.float64)
|
||||||
|
if q is not None:
|
||||||
|
self.heading_init_base_quat = np.array(q, np.float64)
|
||||||
|
with self.motion_lock:
|
||||||
|
rf = min(self.ref_cursor, self.motion_timesteps - 1)
|
||||||
|
if self.encode_mode == 2 and self.smpl_root_quat is not None:
|
||||||
|
# Anchor the heading delta to the SMPL root at init so the
|
||||||
|
# robot turns *relative* to the clip/operator start heading.
|
||||||
|
self.init_ref_quat = np.asarray(self.smpl_root_quat, np.float64)
|
||||||
|
else:
|
||||||
|
self.init_ref_quat = self.motion_body_quats[rf].copy()
|
||||||
|
self.delta_heading = 0.0
|
||||||
|
self.first_motion = False
|
||||||
|
self.reinit_heading = False
|
||||||
|
logger.debug("[Heading] init quat: %s", self.heading_init_base_quat)
|
||||||
|
return super().step(robot_obs, update_encoder=update_encoder, debug=debug)
|
||||||
|
|
||||||
|
def advance_cursor(self):
|
||||||
|
"""Advance the reference cursor one frame per 50 Hz tick (no wall-clock catch-up)."""
|
||||||
|
if not self.playing:
|
||||||
|
return
|
||||||
|
with self.motion_lock:
|
||||||
|
if self.motion_timesteps > 0:
|
||||||
|
self.ref_cursor = min(self.ref_cursor + 1, self.motion_timesteps - 1)
|
||||||
@@ -0,0 +1,415 @@
|
|||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
|
# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
"""SONIC full-body controller for Unitree G1."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from collections import deque
|
||||||
|
import logging
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
from huggingface_hub import hf_hub_download
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
from lerobot.utils.import_utils import _onnxruntime_available, require_package
|
||||||
|
|
||||||
|
from ..g1_utils import (
|
||||||
|
MUJOCO_TO_ISAACLAB,
|
||||||
|
WB_ACTION_DIM,
|
||||||
|
G1_29_JointIndex,
|
||||||
|
lowstate_to_obs,
|
||||||
|
wb_action_key,
|
||||||
|
)
|
||||||
|
from .sonic_pipeline import (
|
||||||
|
CONTROL_DT,
|
||||||
|
DEFAULT_ANGLES,
|
||||||
|
ENCODER_UPDATE_EVERY,
|
||||||
|
TOKEN_DIM,
|
||||||
|
PlannerController,
|
||||||
|
compute_kp_kd,
|
||||||
|
make_ort_session_options,
|
||||||
|
ort_providers,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Action-feature prefix for the latent-token interface (see _extract_token_from_action).
|
||||||
|
TOKEN_ACTION_PREFIX = "motion_token"
|
||||||
|
# Proprio-state prefix for the token interface: the robot echoes the last commanded
|
||||||
|
# token here so ``lerobot-rollout`` aggregates it into a 64-D ``observation.state``.
|
||||||
|
TOKEN_STATE_PREFIX = "motion_token_state"
|
||||||
|
|
||||||
|
|
||||||
|
def token_action_key(i: int) -> str:
|
||||||
|
"""Action-dict key for the i-th component of the 64-D SONIC latent token.
|
||||||
|
|
||||||
|
The ``.pos`` suffix is required so the value flows through ``lerobot-rollout``,
|
||||||
|
which only routes ``.pos`` scalar features onto the policy action vector.
|
||||||
|
"""
|
||||||
|
return f"{TOKEN_ACTION_PREFIX}.{i}.pos"
|
||||||
|
|
||||||
|
|
||||||
|
def token_state_key(i: int) -> str:
|
||||||
|
"""Observation key for the i-th component of the 64-D SONIC latent token state."""
|
||||||
|
return f"{TOKEN_STATE_PREFIX}.{i}.pos"
|
||||||
|
|
||||||
|
if TYPE_CHECKING or _onnxruntime_available:
|
||||||
|
import onnxruntime as ort
|
||||||
|
else:
|
||||||
|
ort = None
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Startup blend duration: over the first control ticks, linearly interpolate every joint
|
||||||
|
# from the robot's initial measured pose into the policy's commanded target, so control
|
||||||
|
# eases in without a snap on the first command.
|
||||||
|
INIT_RAMP_S = 3.0
|
||||||
|
|
||||||
|
# Neutral ("zero pose") SONIC token, held by token_mode until the first real token
|
||||||
|
# arrives. Captured from the encoder's own output while the robot stood idle in sim
|
||||||
|
# (capture_neutral_token.py): the encoder is an FSQ bottleneck (~5 bit/dim, 15.5 half-
|
||||||
|
# width, Div(16)), so its tokens live on the 1/16 grid. We store the integer FSQ codes
|
||||||
|
# and rescale by the same 1/16 step, giving an exact on-grid token -- unlike the literal
|
||||||
|
# all-zero token, which is off the encoder's learned manifold and decodes to a slightly
|
||||||
|
# goofy stance. This one decodes to a stable, natural standing pose.
|
||||||
|
_NEUTRAL_TOKEN_CODES = np.array(
|
||||||
|
[-1, 3, 1, -1, 1, -3, 6, 1, 1, 1, -2, -4, -2, 0, -3, -1,
|
||||||
|
2, -1, -3, -5, 3, 1, 1, -4, -1, -1, 1, -7, 0, 1, 2, -2,
|
||||||
|
5, -2, -2, -4, 0, -1, 3, -1, 0, -5, -1, 0, -4, 0, 0, -1,
|
||||||
|
-1, 2, -2, 1, 3, 3, 1, 0, 0, 6, 0, -7, 3, 0, 2, -2],
|
||||||
|
dtype=np.float32,
|
||||||
|
)
|
||||||
|
NEUTRAL_TOKEN = _NEUTRAL_TOKEN_CODES / 16.0 # FSQ Div(16): integer codes -> on-grid token
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_wb34_from_action(action: dict | None) -> np.ndarray | None:
|
||||||
|
"""Reassemble a dense (34,) whole-body command from ``wb.{i}.pos`` keys, or None.
|
||||||
|
|
||||||
|
This is the OpenHLM / pi0.5 joint-based interface: one 34-D vector per tick
|
||||||
|
(sentinel: presence of ``wb.0.pos``) carrying absolute joint targets in real
|
||||||
|
units. The ``.pos`` suffix lets these flow through ``lerobot-rollout`` as normal
|
||||||
|
joint-position action features.
|
||||||
|
"""
|
||||||
|
if not action:
|
||||||
|
return None
|
||||||
|
keys = [wb_action_key(i) for i in range(WB_ACTION_DIM)]
|
||||||
|
# Require the full dense command: a partial action (e.g. only ``wb.0.pos``)
|
||||||
|
# must not be silently zero-filled, which would drive most joints toward 0.
|
||||||
|
if any(key not in action for key in keys):
|
||||||
|
return None
|
||||||
|
return np.fromiter(
|
||||||
|
(float(action[key]) for key in keys),
|
||||||
|
dtype=np.float32,
|
||||||
|
count=WB_ACTION_DIM,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_token_from_action(action: dict | None) -> np.ndarray | None:
|
||||||
|
"""Reassemble a dense (64,) latent token from ``motion_token.{i}`` keys, or None.
|
||||||
|
|
||||||
|
This is the token-only replay interface: instead of a joint reference driving the
|
||||||
|
encoder, the caller supplies the 64-D encoder latent directly (e.g. a recorded
|
||||||
|
``action.motion_token`` column), which the decoder consumes with the encoder
|
||||||
|
bypassed. Requires the full dense token; a partial one is ignored (returns None).
|
||||||
|
"""
|
||||||
|
if not action:
|
||||||
|
return None
|
||||||
|
keys = [token_action_key(i) for i in range(TOKEN_DIM)]
|
||||||
|
if any(key not in action for key in keys):
|
||||||
|
return None
|
||||||
|
return np.fromiter(
|
||||||
|
(float(action[key]) for key in keys),
|
||||||
|
dtype=np.float32,
|
||||||
|
count=TOKEN_DIM,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _wb34_to_reference(wb: np.ndarray) -> tuple[np.ndarray, np.ndarray]:
|
||||||
|
"""Map a 34-D OpenHLM whole-body command to a SONIC mode-0 reference.
|
||||||
|
|
||||||
|
Returns ``(ref29, anchor_quat)`` where ``ref29`` is the 29 joint targets in
|
||||||
|
IsaacLab order (what SONIC's ``motion_joint_positions`` expects) and
|
||||||
|
``anchor_quat`` (wxyz) encodes the root roll/pitch (yaw=0).
|
||||||
|
|
||||||
|
OpenHLM layout : [L-arm 0:7, L-grip 7, R-arm 8:15, R-grip 15,
|
||||||
|
L-leg 16:22, R-leg 22:28, waist 28:31, root rp+yaw 31:34]
|
||||||
|
The 29 joints are first assembled in MuJoCo / Unitree-SDK order
|
||||||
|
([L-leg 0:6, R-leg 6:12, waist 12:15, L-arm 15:22, R-arm 22:29] — the
|
||||||
|
``G1_29_JointIndex`` grouping OpenHLM uses), then permuted to IsaacLab order via
|
||||||
|
``MUJOCO_TO_ISAACLAB``. Grippers (7, 15) are not part of the 29-DoF SONIC
|
||||||
|
reference, and yaw-rate (33) is integrated into the heading by the caller (it
|
||||||
|
cannot be represented in this static per-tick anchor).
|
||||||
|
"""
|
||||||
|
ref_mj = np.zeros(29, np.float32) # MuJoCo / Unitree-SDK grouped order
|
||||||
|
ref_mj[0:6] = wb[16:22] # left leg
|
||||||
|
ref_mj[6:12] = wb[22:28] # right leg
|
||||||
|
ref_mj[12:15] = wb[28:31] # waist
|
||||||
|
ref_mj[15:22] = wb[0:7] # left arm
|
||||||
|
ref_mj[22:29] = wb[8:15] # right arm
|
||||||
|
ref = ref_mj[MUJOCO_TO_ISAACLAB].astype(np.float32) # -> IsaacLab order for SONIC
|
||||||
|
roll, pitch = float(wb[31]), float(wb[32])
|
||||||
|
cr, sr, cp, sp = np.cos(roll / 2), np.sin(roll / 2), np.cos(pitch / 2), np.sin(pitch / 2)
|
||||||
|
anchor = np.array([cr * cp, sr * cp, cr * sp, sr * sp], np.float32) # Rx(roll)·Ry(pitch)
|
||||||
|
return ref, anchor
|
||||||
|
|
||||||
|
|
||||||
|
class SonicRuntime:
|
||||||
|
"""Loads the SONIC encoder/decoder ONNX models and owns the controller.
|
||||||
|
|
||||||
|
No motion planner: the reference motion buffer is written directly each tick by
|
||||||
|
:class:`SonicWholeBodyController` from the incoming 34-D whole-body command.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, force_cpu: bool = False):
|
||||||
|
require_package("onnxruntime", extra="unitree_g1")
|
||||||
|
encoder_path = hf_hub_download(repo_id="nvidia/GEAR-SONIC", filename="model_encoder.onnx")
|
||||||
|
decoder_path = hf_hub_download(repo_id="nvidia/GEAR-SONIC", filename="model_decoder.onnx")
|
||||||
|
|
||||||
|
providers = ort_providers(force_cpu=force_cpu)
|
||||||
|
so = make_ort_session_options()
|
||||||
|
|
||||||
|
encoder_sess = ort.InferenceSession(encoder_path, sess_options=so, providers=providers)
|
||||||
|
decoder_sess = ort.InferenceSession(decoder_path, sess_options=so, providers=providers)
|
||||||
|
|
||||||
|
# Report the provider actually bound, not the one requested: ORT silently falls
|
||||||
|
# back to CPU if CUDA can't load (e.g. libcudnn not on LD_LIBRARY_PATH), and a
|
||||||
|
# CPU decoder drifts the closed-loop heading. Warn loudly so it can't hide.
|
||||||
|
self.use_gpu = decoder_sess.get_providers()[0] == "CUDAExecutionProvider"
|
||||||
|
if not force_cpu and not self.use_gpu:
|
||||||
|
print(
|
||||||
|
"[SONIC] WARNING: decoder bound to CPUExecutionProvider (CUDA unavailable). "
|
||||||
|
"Closed-loop replay/control will drift. Ensure libcudnn is on LD_LIBRARY_PATH "
|
||||||
|
"(site-packages/nvidia/*/lib).",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.kp, self.kd = compute_kp_kd()
|
||||||
|
self.controller = PlannerController(encoder_sess, decoder_sess)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def pipeline(self):
|
||||||
|
return self.controller
|
||||||
|
|
||||||
|
def reset(self):
|
||||||
|
# Full pipeline reset: clears the encoder token, proprioception history and
|
||||||
|
# heading, and rewinds the motion buffer. reinit_heading is set so the next
|
||||||
|
# step re-latches the reference frame to the current robot orientation.
|
||||||
|
self.controller.reset()
|
||||||
|
|
||||||
|
def shutdown(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
class SonicWholeBodyController:
|
||||||
|
"""Full-body SONIC controller for UnitreeG1's background controller thread."""
|
||||||
|
|
||||||
|
control_dt = CONTROL_DT
|
||||||
|
full_body = True
|
||||||
|
# Advertise a dense 34-D whole-body action space (OpenHLM / pi0.5) so the robot
|
||||||
|
# exposes ``wb.{i}.pos`` action features and ``lerobot-rollout`` can drive it
|
||||||
|
# directly with a 34-D VLA policy.
|
||||||
|
wb_action = True
|
||||||
|
|
||||||
|
def __init__(self, force_cpu: bool = False):
|
||||||
|
logger.info("Loading SONIC whole-body controller...")
|
||||||
|
self._runtime = SonicRuntime(force_cpu=force_cpu)
|
||||||
|
self.kp = self._runtime.kp
|
||||||
|
self.kd = self._runtime.kd
|
||||||
|
self.controller = self._runtime.controller
|
||||||
|
|
||||||
|
# Startup blend: ease from the robot's initial pose into the first commanded
|
||||||
|
# policy targets over INIT_RAMP_S (captured on the first control tick).
|
||||||
|
self._init_ramp_steps = max(1, round(INIT_RAMP_S / CONTROL_DT))
|
||||||
|
self._init_step = 0
|
||||||
|
self._start_pose: dict[str, float] = {}
|
||||||
|
|
||||||
|
# Tick counter for the dense whole-body (OpenHLM, mode-0) path's encoder cadence.
|
||||||
|
self._wb_step = 0
|
||||||
|
# Rolling 50-frame reference trajectory (ref29 + anchor quat) built from the
|
||||||
|
# stream of per-tick whole-body commands, fed to the encoder as a batch.
|
||||||
|
self._wb_traj: deque[np.ndarray] = deque(maxlen=50)
|
||||||
|
self._wb_quat_traj: deque[np.ndarray] = deque(maxlen=50)
|
||||||
|
# Integrated heading (rad) from the whole-body command's yaw-rate (index 33),
|
||||||
|
# forwarded to the pipeline as ``delta_heading`` so turn commands take effect.
|
||||||
|
self._heading = 0.0
|
||||||
|
|
||||||
|
# Token-interface state. ``token_mode`` is set True by the robot when the deploy
|
||||||
|
# is token-driven (``UnitreeG1Config.sonic_token_action``): the controller then
|
||||||
|
# holds a stable *neutral* (all-zero) token until the first real token arrives,
|
||||||
|
# and afterwards holds the *last* token received between ticks (the async
|
||||||
|
# controller runs ~50 Hz while a token VLA streams ~30 Hz). This lives here (not
|
||||||
|
# in the entry-point script) so it applies uniformly to run_g1_onboard,
|
||||||
|
# lerobot-rollout and the sim replays. ``token_mode`` stays False for the dense
|
||||||
|
# 34-D whole-body / OpenHLM path, which keeps its own "hold last target" idle.
|
||||||
|
self.token_mode = False
|
||||||
|
self._last_token: np.ndarray | None = None
|
||||||
|
|
||||||
|
logger.info("SONIC ready (encoder/decoder, 34-D whole-body command path)")
|
||||||
|
|
||||||
|
def _run_wholebody34(self, obs: dict, wb: np.ndarray) -> dict:
|
||||||
|
"""Feed a dense 34-D OpenHLM whole-body command as the mode-0 encoder reference.
|
||||||
|
|
||||||
|
The 29 joint targets are held across the encoder lookahead window (zero
|
||||||
|
velocity) and the root roll/pitch set the anchor orientation, then the
|
||||||
|
encoder/decoder run directly (planner bypassed). One command per tick, so the
|
||||||
|
VLA's commanded pose is what SONIC tracks.
|
||||||
|
"""
|
||||||
|
ref, anchor = _wb34_to_reference(wb)
|
||||||
|
c = self.controller
|
||||||
|
if c.encode_mode != 0:
|
||||||
|
c.encode_mode = 0
|
||||||
|
c.reinit_heading = True
|
||||||
|
# Index 33 is a yaw-rate (rad/s): integrate it into a heading offset and hand
|
||||||
|
# it to the pipeline as ``delta_heading`` so commanded turns are tracked rather
|
||||||
|
# than silently dropped (the anchor from _wb34_to_reference only carries r/p).
|
||||||
|
self._heading += float(wb[33]) * CONTROL_DT
|
||||||
|
c.delta_heading = self._heading
|
||||||
|
# Capture the heading/anchor reference on the first whole-body tick. The
|
||||||
|
# controller only latches ``init_ref_quat`` (and the base heading) inside
|
||||||
|
# ``step()`` when ``first_motion or reinit_heading`` — but it already boots in
|
||||||
|
# mode 0, so the mode-switch guard above misses the very first command and the
|
||||||
|
# anchor would stay identity. This mirrors the GEAR reference, which seeds
|
||||||
|
# ``init_ref_quat`` from the first anchor. Must run before the buffers below so
|
||||||
|
# ``step()`` latches ``motion_body_quats[0]`` = this tick's anchor.
|
||||||
|
if self._wb_step == 0:
|
||||||
|
c.reinit_heading = True
|
||||||
|
|
||||||
|
# Accumulate the per-tick commands into a rolling 50-frame reference
|
||||||
|
# trajectory so the encoder's 10-frame, step-5 lookahead sees an actual
|
||||||
|
# motion sequence (with velocities) instead of one repeated pose. 50 frames
|
||||||
|
# == chunk horizon == 10 lookahead frames × step 5.
|
||||||
|
self._wb_traj.append(ref)
|
||||||
|
self._wb_quat_traj.append(anchor)
|
||||||
|
traj = np.asarray(self._wb_traj, np.float32) # (L, 29), oldest -> newest
|
||||||
|
quats = np.asarray(self._wb_quat_traj, np.float32) # (L, 4)
|
||||||
|
n = len(traj)
|
||||||
|
# Per-frame velocities from finite differences (rad/s at the control rate).
|
||||||
|
vel = np.zeros_like(traj)
|
||||||
|
if n > 1:
|
||||||
|
vel[1:] = (traj[1:] - traj[:-1]) / CONTROL_DT
|
||||||
|
vel[0] = vel[1]
|
||||||
|
with c.motion_lock:
|
||||||
|
c.motion_joint_positions[:n] = traj
|
||||||
|
c.motion_joint_velocities[:n] = vel
|
||||||
|
c.motion_body_quats[:n] = quats
|
||||||
|
c.motion_body_pos[:n] = 0.0
|
||||||
|
c.motion_timesteps = n
|
||||||
|
c.ref_cursor = 0
|
||||||
|
c.playing = True
|
||||||
|
do_enc = self._wb_step % ENCODER_UPDATE_EVERY == 0
|
||||||
|
out = c.step(obs, update_encoder=do_enc, debug=False)
|
||||||
|
if self._wb_step % 25 == 0:
|
||||||
|
tgt = np.array([out[f"{m.name}.q"] for m in G1_29_JointIndex], np.float32)
|
||||||
|
logger.info(
|
||||||
|
"[WB34] step=%d |ref|mean=%.3f |target|mean=%.3f target_std=%.3f init_ref_quat=%s",
|
||||||
|
self._wb_step,
|
||||||
|
float(np.abs(ref).mean()),
|
||||||
|
float(np.abs(tgt).mean()),
|
||||||
|
float(tgt.std()),
|
||||||
|
np.round(c.init_ref_quat, 3).tolist(),
|
||||||
|
)
|
||||||
|
self._wb_step += 1
|
||||||
|
return out
|
||||||
|
|
||||||
|
def _run_token(self, obs: dict, token: np.ndarray) -> dict:
|
||||||
|
"""Decode a supplied 64-D latent token directly (encoder bypassed).
|
||||||
|
|
||||||
|
Token-only replay: set the pipeline's cached token to the supplied one and run
|
||||||
|
a decode-only step (``update_encoder=False``). The decoder still closes the loop
|
||||||
|
on live proprioception (history is refreshed inside ``step`` from ``obs``); only
|
||||||
|
the encoder — which would recompute the token from a motion reference — is
|
||||||
|
skipped. Returns the ``<joint>.q`` target dict.
|
||||||
|
"""
|
||||||
|
c = self.controller
|
||||||
|
c.token = np.asarray(token, np.float32)
|
||||||
|
self._wb_step += 1
|
||||||
|
return c.step(obs, update_encoder=False, debug=False)
|
||||||
|
|
||||||
|
def _startup_blend(self, obs: dict, out: dict) -> dict:
|
||||||
|
"""Ease into policy control at startup: for the first ``INIT_RAMP_S`` seconds,
|
||||||
|
interpolate between the robot's pose captured on the first tick and the policy's
|
||||||
|
live commanded target, so the handoff has no snap.
|
||||||
|
|
||||||
|
``out`` is the policy's ``<joint>.q`` target dict for this tick; the blend ratio
|
||||||
|
climbs 0->1 over the ramp, after which the raw policy target passes through.
|
||||||
|
"""
|
||||||
|
if self._init_step >= self._init_ramp_steps or not out:
|
||||||
|
return out
|
||||||
|
if self._init_step == 0:
|
||||||
|
# Capture the robot's actual pose as the interpolation start point.
|
||||||
|
self._start_pose = {
|
||||||
|
f"{m.name}.q": float(obs.get(f"{m.name}.q", DEFAULT_ANGLES[m.value]))
|
||||||
|
for m in G1_29_JointIndex
|
||||||
|
}
|
||||||
|
self._init_step += 1
|
||||||
|
ratio = min(1.0, self._init_step / self._init_ramp_steps)
|
||||||
|
blended = {
|
||||||
|
k: self._start_pose.get(k, float(tgt)) * (1.0 - ratio) + float(tgt) * ratio
|
||||||
|
for k, tgt in out.items()
|
||||||
|
}
|
||||||
|
if self._init_step >= self._init_ramp_steps:
|
||||||
|
logger.info("SONIC startup blend complete -> full policy control")
|
||||||
|
return blended
|
||||||
|
|
||||||
|
def run_step(self, action: dict, lowstate) -> dict:
|
||||||
|
if lowstate is None:
|
||||||
|
return {}
|
||||||
|
obs = lowstate_to_obs(lowstate)
|
||||||
|
|
||||||
|
# Token-only interface (latent replay / token-output VLA): a dense 64-D
|
||||||
|
# ``motion_token.{i}`` command is decoded directly, bypassing the encoder.
|
||||||
|
# Checked before the joint path so a token action takes precedence.
|
||||||
|
token = _extract_token_from_action(action)
|
||||||
|
if token is not None:
|
||||||
|
self._last_token = token
|
||||||
|
elif self._last_token is None and self.token_mode:
|
||||||
|
# Token-driven deploy, but no token has arrived yet: hold the captured
|
||||||
|
# neutral token (NEUTRAL_TOKEN), which the decoder maps to a stable, natural
|
||||||
|
# standing pose (the encoder's own idle output; see NEUTRAL_TOKEN).
|
||||||
|
self._last_token = NEUTRAL_TOKEN.copy()
|
||||||
|
if self._last_token is not None:
|
||||||
|
# Either a fresh token this tick or the last one received (held between the
|
||||||
|
# ~30 Hz token stream and the ~50 Hz control loop).
|
||||||
|
return self._startup_blend(obs, self._run_token(obs, self._last_token))
|
||||||
|
|
||||||
|
# Dense 34-D whole-body command (OpenHLM / pi0.5 joint interface): a single
|
||||||
|
# vector per tick drives the mode-0 encoder reference directly. Until the
|
||||||
|
# policy produces one, hold (no command) so the robot keeps its last target.
|
||||||
|
wb = _extract_wb34_from_action(action)
|
||||||
|
if wb is None:
|
||||||
|
self._wb_miss = getattr(self, "_wb_miss", 0) + 1
|
||||||
|
if self._wb_miss % 50 == 1:
|
||||||
|
akeys = [k for k in action if isinstance(k, str)]
|
||||||
|
logger.info(
|
||||||
|
"[WB34] no wb.*.pos in action this tick (miss=%d). action keys sample: %s",
|
||||||
|
self._wb_miss,
|
||||||
|
akeys[:8],
|
||||||
|
)
|
||||||
|
return {}
|
||||||
|
return self._startup_blend(obs, self._run_wholebody34(obs, wb))
|
||||||
|
|
||||||
|
def reset(self):
|
||||||
|
self._runtime.reset()
|
||||||
|
self._init_step = 0 # re-run the startup blend after a reset
|
||||||
|
self._start_pose = {}
|
||||||
|
self._wb_step = 0
|
||||||
|
self._wb_traj.clear()
|
||||||
|
self._wb_quat_traj.clear()
|
||||||
|
self._heading = 0.0
|
||||||
|
# Drop the held token so token_mode re-seeds the neutral token after a reset.
|
||||||
|
self._last_token = None
|
||||||
|
|
||||||
|
def shutdown(self):
|
||||||
|
self._runtime.shutdown()
|
||||||
@@ -23,10 +23,102 @@ import numpy as np
|
|||||||
|
|
||||||
NUM_MOTORS = 29
|
NUM_MOTORS = 29
|
||||||
|
|
||||||
|
# Joint-order permutations between the two 29-DoF layouts used across the G1 stack:
|
||||||
|
# IsaacLab (policy/training order) and MuJoCo (deploy order). ``a[ISAACLAB_TO_MUJOCO]``
|
||||||
|
# reorders an IsaacLab-ordered vector into MuJoCo order, and vice-versa.
|
||||||
|
ISAACLAB_TO_MUJOCO = np.array(
|
||||||
|
[
|
||||||
|
0,
|
||||||
|
3,
|
||||||
|
6,
|
||||||
|
9,
|
||||||
|
13,
|
||||||
|
17,
|
||||||
|
1,
|
||||||
|
4,
|
||||||
|
7,
|
||||||
|
10,
|
||||||
|
14,
|
||||||
|
18,
|
||||||
|
2,
|
||||||
|
5,
|
||||||
|
8,
|
||||||
|
11,
|
||||||
|
15,
|
||||||
|
19,
|
||||||
|
21,
|
||||||
|
23,
|
||||||
|
25,
|
||||||
|
27,
|
||||||
|
12,
|
||||||
|
16,
|
||||||
|
20,
|
||||||
|
22,
|
||||||
|
24,
|
||||||
|
26,
|
||||||
|
28,
|
||||||
|
],
|
||||||
|
dtype=np.int32,
|
||||||
|
)
|
||||||
|
MUJOCO_TO_ISAACLAB = np.array(
|
||||||
|
[
|
||||||
|
0,
|
||||||
|
6,
|
||||||
|
12,
|
||||||
|
1,
|
||||||
|
7,
|
||||||
|
13,
|
||||||
|
2,
|
||||||
|
8,
|
||||||
|
14,
|
||||||
|
3,
|
||||||
|
9,
|
||||||
|
15,
|
||||||
|
22,
|
||||||
|
4,
|
||||||
|
10,
|
||||||
|
16,
|
||||||
|
23,
|
||||||
|
5,
|
||||||
|
11,
|
||||||
|
17,
|
||||||
|
24,
|
||||||
|
18,
|
||||||
|
25,
|
||||||
|
19,
|
||||||
|
26,
|
||||||
|
20,
|
||||||
|
27,
|
||||||
|
21,
|
||||||
|
28,
|
||||||
|
],
|
||||||
|
dtype=np.int32,
|
||||||
|
)
|
||||||
|
|
||||||
REMOTE_AXES = ("remote.lx", "remote.ly", "remote.rx", "remote.ry")
|
REMOTE_AXES = ("remote.lx", "remote.ly", "remote.rx", "remote.ry")
|
||||||
REMOTE_BUTTONS = tuple(f"remote.button.{i}" for i in range(16))
|
REMOTE_BUTTONS = tuple(f"remote.button.{i}" for i in range(16))
|
||||||
REMOTE_KEYS = REMOTE_AXES + REMOTE_BUTTONS
|
REMOTE_KEYS = REMOTE_AXES + REMOTE_BUTTONS
|
||||||
|
|
||||||
|
# Reserved action-dict field used to forward the set of currently-pressed keyboard
|
||||||
|
# keys from a KeyboardTeleop through the standard action pipeline to the SONIC
|
||||||
|
# whole-body controller (see SonicWholeBodyController._process_keyboard).
|
||||||
|
KEYBOARD_KEYS_FIELD = "keyboard.keys"
|
||||||
|
|
||||||
|
# ── Dense whole-body joint reference (SONIC encode_mode 0, OpenHLM / pi0.5) ──────
|
||||||
|
# A single 34-D whole-body command per tick, in the OpenHLM action layout:
|
||||||
|
# [L-arm(7), L-grip(1), R-arm(7), R-grip(1), L-leg(6), R-leg(6), waist(3),
|
||||||
|
# root roll/pitch + yaw-rate(3)]
|
||||||
|
# Fed as flat scalars ``wb.0.pos .. wb.33.pos``. The ``.pos`` suffix makes these
|
||||||
|
# behave like ordinary joint-position action features so ``lerobot-rollout`` routes
|
||||||
|
# them straight from a 34-D VLA (OpenHLM / pi0.5) onto the robot.
|
||||||
|
WB_ACTION_PREFIX = "wb."
|
||||||
|
WB_ACTION_DIM = 34
|
||||||
|
|
||||||
|
|
||||||
|
def wb_action_key(i: int) -> str:
|
||||||
|
"""Action-dict key for the ``i``-th whole-body command scalar (``wb.{i}.pos``)."""
|
||||||
|
return f"{WB_ACTION_PREFIX}{i}.pos"
|
||||||
|
|
||||||
|
|
||||||
def default_remote_input() -> dict[str, float]:
|
def default_remote_input() -> dict[str, float]:
|
||||||
"""Return a zeroed-out remote input dict (axes + buttons)."""
|
"""Return a zeroed-out remote input dict (axes + buttons)."""
|
||||||
@@ -63,13 +155,92 @@ class G1_29_JointArmIndex(IntEnum):
|
|||||||
kRightWristYaw = 28
|
kRightWristYaw = 28
|
||||||
|
|
||||||
|
|
||||||
|
def lowstate_to_obs(lowstate) -> dict:
|
||||||
|
"""Build a robot observation dict from a Unitree lowstate.
|
||||||
|
|
||||||
|
Shared by ``UnitreeG1.get_observation`` and the SONIC pipeline so the
|
||||||
|
lowstate -> obs mapping lives in exactly one place. Keys match the
|
||||||
|
``<joint>.q``/``imu.*`` schema consumed across the controllers.
|
||||||
|
"""
|
||||||
|
obs: dict = {}
|
||||||
|
|
||||||
|
for motor in G1_29_JointIndex:
|
||||||
|
idx = motor.value
|
||||||
|
obs[f"{motor.name}.q"] = lowstate.motor_state[idx].q
|
||||||
|
obs[f"{motor.name}.dq"] = lowstate.motor_state[idx].dq
|
||||||
|
obs[f"{motor.name}.tau"] = lowstate.motor_state[idx].tau_est
|
||||||
|
|
||||||
|
imu = lowstate.imu_state
|
||||||
|
if imu.gyroscope:
|
||||||
|
obs["imu.gyro.x"] = imu.gyroscope[0]
|
||||||
|
obs["imu.gyro.y"] = imu.gyroscope[1]
|
||||||
|
obs["imu.gyro.z"] = imu.gyroscope[2]
|
||||||
|
if imu.accelerometer:
|
||||||
|
obs["imu.accel.x"] = imu.accelerometer[0]
|
||||||
|
obs["imu.accel.y"] = imu.accelerometer[1]
|
||||||
|
obs["imu.accel.z"] = imu.accelerometer[2]
|
||||||
|
if imu.quaternion:
|
||||||
|
obs["imu.quat.w"] = imu.quaternion[0]
|
||||||
|
obs["imu.quat.x"] = imu.quaternion[1]
|
||||||
|
obs["imu.quat.y"] = imu.quaternion[2]
|
||||||
|
obs["imu.quat.z"] = imu.quaternion[3]
|
||||||
|
if imu.rpy:
|
||||||
|
obs["imu.rpy.roll"] = imu.rpy[0]
|
||||||
|
obs["imu.rpy.pitch"] = imu.rpy[1]
|
||||||
|
obs["imu.rpy.yaw"] = imu.rpy[2]
|
||||||
|
|
||||||
|
wr = getattr(lowstate, "wireless_remote", None)
|
||||||
|
if wr:
|
||||||
|
obs["wireless_remote"] = bytes(wr) if not isinstance(wr, (bytes, bytearray)) else wr
|
||||||
|
|
||||||
|
return obs
|
||||||
|
|
||||||
|
|
||||||
|
def obs_to_wb34_state(obs: dict) -> np.ndarray:
|
||||||
|
"""Build the 34-D OpenHLM / pi0.5 proprio state from a G1 observation dict.
|
||||||
|
|
||||||
|
Mirrors the whole-body *action* layout so the policy sees state and action in
|
||||||
|
the same coordinates::
|
||||||
|
|
||||||
|
[L-arm(7), L-grip(1), R-arm(7), R-grip(1),
|
||||||
|
L-leg(6), R-leg(6), waist(3), root roll/pitch + yaw-rate(3)]
|
||||||
|
|
||||||
|
Joint positions come from the ``<joint>.q`` obs keys, which are already in
|
||||||
|
MuJoCo / Unitree-SDK order — the same body-part grouping OpenHLM uses
|
||||||
|
([L-leg 0:6, R-leg 6:12, waist 12:15, L-arm 15:22, R-arm 22:29]) — so they are
|
||||||
|
regrouped directly (no IsaacLab permutation). The G1 has no grippers in its
|
||||||
|
29-DoF body, so both gripper slots are 0. Root roll/pitch are the IMU RPY and
|
||||||
|
the last slot is the IMU yaw rate (gyro z).
|
||||||
|
"""
|
||||||
|
q_mj = np.array(
|
||||||
|
[float(obs.get(f"{m.name}.q", 0.0)) for m in G1_29_JointIndex],
|
||||||
|
dtype=np.float32,
|
||||||
|
)
|
||||||
|
lleg, rleg, waist = q_mj[0:6], q_mj[6:12], q_mj[12:15]
|
||||||
|
larm, rarm = q_mj[15:22], q_mj[22:29]
|
||||||
|
|
||||||
|
state = np.zeros(34, dtype=np.float32)
|
||||||
|
state[0:7] = larm
|
||||||
|
# state[7] left gripper — none on 29-DoF G1
|
||||||
|
state[8:15] = rarm
|
||||||
|
# state[15] right gripper — none on 29-DoF G1
|
||||||
|
state[16:22] = lleg
|
||||||
|
state[22:28] = rleg
|
||||||
|
state[28:31] = waist
|
||||||
|
state[31] = float(obs.get("imu.rpy.roll", 0.0))
|
||||||
|
state[32] = float(obs.get("imu.rpy.pitch", 0.0))
|
||||||
|
state[33] = float(obs.get("imu.gyro.z", 0.0))
|
||||||
|
return state
|
||||||
|
|
||||||
|
|
||||||
def make_locomotion_controller(name: str | None):
|
def make_locomotion_controller(name: str | None):
|
||||||
"""Instantiate a locomotion controller by class name. Returns None if name is None."""
|
"""Instantiate a locomotion controller by class name. Returns None if name is None."""
|
||||||
if name is None:
|
if name is None:
|
||||||
return None
|
return None
|
||||||
controllers = {
|
controllers = {
|
||||||
"GrootLocomotionController": "lerobot.robots.unitree_g1.gr00t_locomotion",
|
"GrootLocomotionController": "lerobot.robots.unitree_g1.controllers.gr00t_locomotion",
|
||||||
"HolosomaLocomotionController": "lerobot.robots.unitree_g1.holosoma_locomotion",
|
"HolosomaLocomotionController": "lerobot.robots.unitree_g1.controllers.holosoma_locomotion",
|
||||||
|
"SonicWholeBodyController": "lerobot.robots.unitree_g1.controllers.sonic_whole_body",
|
||||||
}
|
}
|
||||||
module_path = controllers.get(name)
|
module_path = controllers.get(name)
|
||||||
if module_path is None:
|
if module_path is None:
|
||||||
|
|||||||
@@ -0,0 +1,192 @@
|
|||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
|
# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
"""Laptop-side sender for the SONIC whole-body walk policy, onboard deployment.
|
||||||
|
|
||||||
|
This is the counterpart to ``run_g1_onboard.py`` (which runs the SONIC decoder on the
|
||||||
|
robot). The heavy VLA (``nepyope/sonic_walk``, a pi0.5 token policy) runs here on the
|
||||||
|
laptop GPU; only the resulting 64-D latent token is shipped to the robot over ZMQ:
|
||||||
|
|
||||||
|
laptop: camera frame (ZMQ from robot :5555) + previous token
|
||||||
|
-> pi0.5 -> next 64-D token
|
||||||
|
-> PUSH JSON {motion_token.i.pos: ...} to robot :6004
|
||||||
|
robot: run_g1_onboard receives the token, SonicWholeBodyController decodes it
|
||||||
|
into whole-body joint commands against local DDS at full rate.
|
||||||
|
|
||||||
|
The policy's ``observation.state`` is the token currently being executed, so we close
|
||||||
|
the loop by feeding back the *last token we sent* (the decoder holds it until a new one
|
||||||
|
arrives). This mirrors what ``lerobot-rollout`` does via the robot's token echo, but
|
||||||
|
without a controller / DDS on the laptop.
|
||||||
|
|
||||||
|
The policy is pi0.5 with chunk_size=50, so a full diffusion inference runs only about
|
||||||
|
once every 50 ticks; ``select_action`` pops one queued token per tick in between.
|
||||||
|
|
||||||
|
Run ``run_g1_onboard.py --controller SonicWholeBodyController --sonic-token-action
|
||||||
|
--cameras ...`` on the robot first, then this on the laptop:
|
||||||
|
|
||||||
|
python -m lerobot.robots.unitree_g1.infer_sonic_g1_onboard \
|
||||||
|
--policy-path nepyope/sonic_walk --robot-ip 192.168.123.164 \
|
||||||
|
--task "walk back and forth"
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import contextlib
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import signal
|
||||||
|
import time
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import torch
|
||||||
|
|
||||||
|
from lerobot.cameras.zmq import ZMQCamera, ZMQCameraConfig
|
||||||
|
from lerobot.configs.policies import PreTrainedConfig
|
||||||
|
from lerobot.policies.factory import get_policy_class, make_pre_post_processors
|
||||||
|
from lerobot.policies.utils import prepare_observation_for_inference
|
||||||
|
from lerobot.robots.unitree_g1.controllers.sonic_whole_body import (
|
||||||
|
NEUTRAL_TOKEN,
|
||||||
|
TOKEN_DIM,
|
||||||
|
token_action_key,
|
||||||
|
)
|
||||||
|
|
||||||
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s", force=True)
|
||||||
|
logger = logging.getLogger("sonic_sender")
|
||||||
|
|
||||||
|
ACTION_PORT = 6004 # matches run_g1_onboard.py --action-port
|
||||||
|
IMAGE_KEY = "observation.images.ego_view" # pi05 sonic_walk VISUAL input
|
||||||
|
STATE_KEY = "observation.state"
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||||
|
p.add_argument("--policy-path", default="nepyope/sonic_walk", help="Policy repo id or local path")
|
||||||
|
p.add_argument("--robot-ip", default="192.168.123.164", help="Robot IP (camera + action ports)")
|
||||||
|
p.add_argument("--action-port", type=int, default=ACTION_PORT, help="Onboard ZMQ PULL port for actions")
|
||||||
|
p.add_argument("--camera-port", type=int, default=5555, help="Onboard ZMQ camera PUB port")
|
||||||
|
p.add_argument("--camera-name", default="head_camera", help="Camera name served by run_g1_onboard")
|
||||||
|
p.add_argument("--camera-width", type=int, default=640, help="Camera width")
|
||||||
|
p.add_argument("--camera-height", type=int, default=480, help="Camera height")
|
||||||
|
p.add_argument("--task", default="walk back and forth", help="Language prompt for the VLA")
|
||||||
|
p.add_argument("--fps", type=float, default=30.0, help="Token send rate (matches training inference)")
|
||||||
|
p.add_argument("--device", default="cuda", help="Torch device")
|
||||||
|
p.add_argument("--max-ticks", type=int, default=0, help="Stop after N ticks (0 = run forever)")
|
||||||
|
p.add_argument("--dry-run", action="store_true", help="Run inference but do not PUSH tokens to the robot")
|
||||||
|
args = p.parse_args()
|
||||||
|
|
||||||
|
device = torch.device(args.device)
|
||||||
|
|
||||||
|
# --- Policy + processors (normalization stats baked into the checkpoint) ---
|
||||||
|
logger.info("Loading policy from '%s'...", args.policy_path)
|
||||||
|
policy_cfg = PreTrainedConfig.from_pretrained(args.policy_path)
|
||||||
|
policy_cfg.pretrained_path = args.policy_path
|
||||||
|
policy = get_policy_class(policy_cfg.type).from_pretrained(args.policy_path, config=policy_cfg)
|
||||||
|
policy = policy.to(device)
|
||||||
|
policy.eval()
|
||||||
|
policy.reset()
|
||||||
|
|
||||||
|
preprocessor, postprocessor = make_pre_post_processors(
|
||||||
|
policy_cfg=policy_cfg,
|
||||||
|
pretrained_path=args.policy_path,
|
||||||
|
preprocessor_overrides={"device_processor": {"device": str(device)}},
|
||||||
|
)
|
||||||
|
logger.info("Policy loaded (type=%s, device=%s, chunk=%s)", policy_cfg.type, device,
|
||||||
|
getattr(policy_cfg, "chunk_size", "?"))
|
||||||
|
|
||||||
|
# --- Camera (ZMQ from the robot's onboard image server) ---
|
||||||
|
cam = ZMQCamera(
|
||||||
|
ZMQCameraConfig(
|
||||||
|
server_address=args.robot_ip,
|
||||||
|
port=args.camera_port,
|
||||||
|
camera_name=args.camera_name,
|
||||||
|
width=args.camera_width,
|
||||||
|
height=args.camera_height,
|
||||||
|
fps=int(args.fps),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
logger.info("Connecting camera %s@%s:%d ...", args.camera_name, args.robot_ip, args.camera_port)
|
||||||
|
cam.connect()
|
||||||
|
|
||||||
|
# --- Action PUSH socket to the onboard controller ---
|
||||||
|
import zmq
|
||||||
|
|
||||||
|
ctx = zmq.Context.instance()
|
||||||
|
sock = ctx.socket(zmq.PUSH)
|
||||||
|
sock.setsockopt(zmq.SNDHWM, 2)
|
||||||
|
sock.setsockopt(zmq.LINGER, 0)
|
||||||
|
sock.connect(f"tcp://{args.robot_ip}:{args.action_port}")
|
||||||
|
logger.info("Sending tokens to tcp://%s:%d (dry_run=%s)", args.robot_ip, args.action_port, args.dry_run)
|
||||||
|
|
||||||
|
stop = {"flag": False}
|
||||||
|
signal.signal(signal.SIGINT, lambda *_: stop.__setitem__("flag", True))
|
||||||
|
signal.signal(signal.SIGTERM, lambda *_: stop.__setitem__("flag", True))
|
||||||
|
|
||||||
|
# observation.state = the token currently executing on the robot (last one we sent);
|
||||||
|
# start at the neutral token the decoder holds before the first send, so the very
|
||||||
|
# first inference sees the true executing token (not zeros).
|
||||||
|
prev_token = NEUTRAL_TOKEN.copy()
|
||||||
|
period = 1.0 / args.fps
|
||||||
|
n = 0
|
||||||
|
t_infer_total = 0.0
|
||||||
|
logger.info("Streaming tokens at %.0f Hz. Ctrl-C to stop.", args.fps)
|
||||||
|
try:
|
||||||
|
while not stop["flag"]:
|
||||||
|
t0 = time.time()
|
||||||
|
try:
|
||||||
|
frame = cam.read() # HxWxC uint8 RGB
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
logger.warning("Camera read failed: %s", e)
|
||||||
|
time.sleep(period)
|
||||||
|
continue
|
||||||
|
|
||||||
|
raw_obs = {
|
||||||
|
IMAGE_KEY: np.ascontiguousarray(frame),
|
||||||
|
STATE_KEY: prev_token.copy(),
|
||||||
|
}
|
||||||
|
with torch.inference_mode():
|
||||||
|
obs = prepare_observation_for_inference(raw_obs, device, args.task, "unitree_g1")
|
||||||
|
obs = preprocessor(obs)
|
||||||
|
action = policy.select_action(obs)
|
||||||
|
action = postprocessor(action)
|
||||||
|
token = action.squeeze(0).to("cpu").numpy().astype(np.float32)
|
||||||
|
prev_token = token
|
||||||
|
|
||||||
|
if not args.dry_run:
|
||||||
|
msg = {token_action_key(i): float(token[i]) for i in range(TOKEN_DIM)}
|
||||||
|
with contextlib.suppress(zmq.Again):
|
||||||
|
sock.send_string(json.dumps(msg), zmq.NOBLOCK)
|
||||||
|
|
||||||
|
n += 1
|
||||||
|
t_infer_total += time.time() - t0
|
||||||
|
if n % 30 == 0:
|
||||||
|
logger.info(
|
||||||
|
"tick %d | avg %.1f ms/tick | token[:3]=%s",
|
||||||
|
n, 1000.0 * t_infer_total / 30.0, np.round(token[:3], 3).tolist(),
|
||||||
|
)
|
||||||
|
t_infer_total = 0.0
|
||||||
|
|
||||||
|
if args.max_ticks and n >= args.max_ticks:
|
||||||
|
break
|
||||||
|
time.sleep(max(0.0, period - (time.time() - t0)))
|
||||||
|
finally:
|
||||||
|
logger.info("Stopping sender after %d ticks.", n)
|
||||||
|
with contextlib.suppress(Exception):
|
||||||
|
cam.disconnect()
|
||||||
|
with contextlib.suppress(Exception):
|
||||||
|
sock.close(linger=0)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,254 @@
|
|||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
|
# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
"""Run the G1 locomotion / whole-body controller ONBOARD, driven by high-level actions
|
||||||
|
from a laptop.
|
||||||
|
|
||||||
|
The controller (GR00T / Holosoma / SONIC whole-body) runs on the robot itself against
|
||||||
|
local DDS, at full control rate. The laptop ships only the resulting high-level action
|
||||||
|
(arm joint targets + joystick axes + gripper flags, or a 64-D SONIC motion token) as
|
||||||
|
JSON over ZMQ. This process applies each action via ``UnitreeG1.send_action`` while the
|
||||||
|
onboard controller thread keeps the legs balanced / decodes the token.
|
||||||
|
|
||||||
|
This is the real-deploy counterpart to running ``lerobot-rollout`` on the laptop with
|
||||||
|
``--robot.is_simulation=false`` (the ZMQ *socket bridge*): there the 50 Hz lowcmd
|
||||||
|
crosses the network; here only compact high-level actions do, and the control loop stays
|
||||||
|
local to the robot. Pair with a laptop client that produces actions (exo teleop, or a
|
||||||
|
policy such as ``nepyope/sonic_walk`` emitting ``motion_token.{i}.pos``).
|
||||||
|
|
||||||
|
Besides receiving actions, this process publishes ``observation.state`` (29 joint ``.q``)
|
||||||
|
on a ZMQ PUB port so a laptop policy client has proprioception.
|
||||||
|
|
||||||
|
Safety: type ``e`` then Enter in this terminal to stop immediately (zero-torque + exit).
|
||||||
|
Ctrl-C does the normal graceful shutdown (kp ramp).
|
||||||
|
|
||||||
|
Examples (on the robot):
|
||||||
|
|
||||||
|
# GR00T locomotion, arm targets from the laptop:
|
||||||
|
python -m lerobot.robots.unitree_g1.run_g1_onboard --controller GrootLocomotionController
|
||||||
|
|
||||||
|
# SONIC whole-body walk policy: laptop ships 64-D tokens, decoder runs here:
|
||||||
|
python -m lerobot.robots.unitree_g1.run_g1_onboard \
|
||||||
|
--controller SonicWholeBodyController --sonic-token-action \
|
||||||
|
--cameras "head_camera:/dev/v4l/by-path/platform-3610000.usb-usb-0:2.1:1.3-video-index0:640x480"
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import contextlib
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import signal
|
||||||
|
import sys
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import zmq
|
||||||
|
|
||||||
|
from lerobot.cameras.zmq.image_server import ImageServer
|
||||||
|
from lerobot.robots.unitree_g1.config_unitree_g1 import UnitreeG1Config
|
||||||
|
from lerobot.robots.unitree_g1.g1_utils import G1_29_JointIndex
|
||||||
|
from lerobot.robots.unitree_g1.run_g1_server import Gripper, build_gripper, parse_camera_specs
|
||||||
|
from lerobot.robots.unitree_g1.unitree_g1 import UnitreeG1
|
||||||
|
|
||||||
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s", force=True)
|
||||||
|
logger = logging.getLogger("g1_onboard")
|
||||||
|
|
||||||
|
ACTION_PORT = 6004
|
||||||
|
STATE_PORT = 6005
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||||
|
p.add_argument("--controller", default="GrootLocomotionController", help="Controller class name")
|
||||||
|
p.add_argument("--dds-interface", default=None, help="DDS network interface (default: SDK default)")
|
||||||
|
p.add_argument(
|
||||||
|
"--sim",
|
||||||
|
action="store_true",
|
||||||
|
help="Attach to a DDS MuJoCo sim: skip MotionSwitcher + physical remote, default dds-interface 'lo'.",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--sonic-token-action",
|
||||||
|
action="store_true",
|
||||||
|
help="SONIC token interface: actions carry a 64-D motion_token.{i}.pos that the decoder consumes.",
|
||||||
|
)
|
||||||
|
p.add_argument("--action-port", type=int, default=ACTION_PORT, help="ZMQ PULL port for laptop actions")
|
||||||
|
p.add_argument("--state-port", type=int, default=STATE_PORT, help="ZMQ PUB port for observation.state")
|
||||||
|
p.add_argument("--state-fps", type=float, default=30.0, help="observation.state publish rate; <=0 disables")
|
||||||
|
p.add_argument("--gravity-compensation", action="store_true", help="Enable arm gravity compensation")
|
||||||
|
# Gripper control (Damiao over CAN).
|
||||||
|
p.add_argument("--grippers", action="store_true", help="Drive Damiao grippers from action L3/R3 flags")
|
||||||
|
p.add_argument("--gripper-port-left", default="can1", help="CAN interface for LEFT gripper")
|
||||||
|
p.add_argument("--gripper-port-right", default="can0", help="CAN interface for RIGHT gripper")
|
||||||
|
p.add_argument("--gripper-send-id", type=lambda x: int(x, 0), default=0x08, help="Motor send CAN id")
|
||||||
|
p.add_argument("--gripper-recv-id", type=lambda x: int(x, 0), default=0x18, help="Motor recv CAN id")
|
||||||
|
p.add_argument("--gripper-motor-type", default="dm4310", help="Damiao motor type")
|
||||||
|
p.add_argument("--gripper-open-deg", type=float, default=-65.0, help="Gripper OPEN position (deg)")
|
||||||
|
p.add_argument("--gripper-close-deg", type=float, default=0.0, help="Gripper CLOSE position (deg)")
|
||||||
|
p.add_argument("--gripper-kp", type=float, default=15.0, help="MIT position gain (stiffness)")
|
||||||
|
p.add_argument("--gripper-kd", type=float, default=0.5, help="MIT damping gain")
|
||||||
|
p.add_argument("--gripper-no-fd", dest="gripper_fd", action="store_false", help="Classic CAN (non-FD)")
|
||||||
|
p.set_defaults(gripper_fd=True)
|
||||||
|
# Optional camera streaming (ZMQ) so the laptop policy client / viewer can connect.
|
||||||
|
p.add_argument("--cameras", default=None, help="Camera spec 'name:device[:WxH[:FOURCC]]', comma-sep")
|
||||||
|
p.add_argument("--camera-fps", type=int, default=30, help="Camera FPS")
|
||||||
|
p.add_argument("--camera-port", type=int, default=5555, help="Camera ZMQ port")
|
||||||
|
p.add_argument("--camera-width", type=int, default=640, help="Default camera width")
|
||||||
|
p.add_argument("--camera-height", type=int, default=480, help="Default camera height")
|
||||||
|
args = p.parse_args()
|
||||||
|
|
||||||
|
dds_interface = args.dds_interface
|
||||||
|
if args.sim and dds_interface is None:
|
||||||
|
dds_interface = "lo"
|
||||||
|
|
||||||
|
cfg = UnitreeG1Config(
|
||||||
|
is_simulation=False,
|
||||||
|
onboard=True,
|
||||||
|
controller=args.controller,
|
||||||
|
dds_interface=dds_interface,
|
||||||
|
gravity_compensation=args.gravity_compensation,
|
||||||
|
release_motion_control=not args.sim,
|
||||||
|
physical_remote=not args.sim,
|
||||||
|
sonic_token_action=args.sonic_token_action,
|
||||||
|
cameras={},
|
||||||
|
)
|
||||||
|
|
||||||
|
# Optional camera server (background thread; independent of DDS/CAN).
|
||||||
|
camera_server = None
|
||||||
|
if args.cameras:
|
||||||
|
cameras = parse_camera_specs(args.cameras, args.camera_width, args.camera_height)
|
||||||
|
camera_server = ImageServer({"fps": args.camera_fps, "cameras": cameras}, port=args.camera_port)
|
||||||
|
threading.Thread(target=camera_server.run, daemon=True).start()
|
||||||
|
cam_summary = ", ".join(f"{name}(dev {c['device_id']})" for name, c in cameras.items())
|
||||||
|
logger.info("Camera server started on :%d: %s", args.camera_port, cam_summary)
|
||||||
|
|
||||||
|
robot = UnitreeG1(cfg)
|
||||||
|
logger.info("Connecting onboard robot (controller=%s, token=%s)...", args.controller, args.sonic_token_action)
|
||||||
|
robot.connect()
|
||||||
|
# Note: with --sonic-token-action the SonicWholeBodyController holds a neutral
|
||||||
|
# (all-zero) token until the first laptop token arrives, then holds the last token
|
||||||
|
# between ticks -- see SonicWholeBodyController.token_mode (set from config).
|
||||||
|
|
||||||
|
grippers: dict[str, Gripper] = {}
|
||||||
|
if args.grippers:
|
||||||
|
for side, port in (("L", args.gripper_port_left), ("R", args.gripper_port_right)):
|
||||||
|
grippers[side] = build_gripper(
|
||||||
|
side, port, args.gripper_send_id, args.gripper_recv_id, args.gripper_motor_type,
|
||||||
|
args.gripper_fd, args.gripper_open_deg, args.gripper_close_deg, args.gripper_kp, args.gripper_kd,
|
||||||
|
)
|
||||||
|
logger.info("Grippers enabled: L3 -> left, R3 -> right")
|
||||||
|
|
||||||
|
ctx = zmq.Context.instance()
|
||||||
|
sock = ctx.socket(zmq.PULL)
|
||||||
|
sock.setsockopt(zmq.CONFLATE, 1) # only ever act on the freshest command
|
||||||
|
sock.setsockopt(zmq.RCVTIMEO, 200) # keeps the loop responsive to the stop event
|
||||||
|
sock.bind(f"tcp://0.0.0.0:{args.action_port}")
|
||||||
|
logger.info("Onboard controller live. Waiting for laptop actions on :%d ...", args.action_port)
|
||||||
|
logger.info("Type 'e' then Enter to STOP immediately (or Ctrl-C for graceful shutdown).")
|
||||||
|
|
||||||
|
stop = threading.Event()
|
||||||
|
signal.signal(signal.SIGINT, lambda *_: stop.set())
|
||||||
|
signal.signal(signal.SIGTERM, lambda *_: stop.set())
|
||||||
|
|
||||||
|
def estop_listener() -> None:
|
||||||
|
for line in sys.stdin:
|
||||||
|
if line.strip().lower() == "e":
|
||||||
|
logger.warning("E-STOP ('e'): going passive NOW.")
|
||||||
|
try:
|
||||||
|
robot._shutdown_event.set() # stop the controller loop publishing
|
||||||
|
time.sleep(0.05)
|
||||||
|
robot._send_zero_torque() # motors limp; nothing overwrites it now
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
logger.warning("E-stop zero-torque failed: %s", e)
|
||||||
|
os._exit(0) # immediate hard exit, no slow cleanup
|
||||||
|
|
||||||
|
threading.Thread(target=estop_listener, daemon=True).start()
|
||||||
|
|
||||||
|
# Proprioception feedback: publish observation.state (29 joint .q) so a laptop
|
||||||
|
# inference client can feed it to a policy. DDS stays local; only compact JSON
|
||||||
|
# state crosses the network. (For a token policy the laptop closes the loop on the
|
||||||
|
# token instead, but publishing joint state is harmless and useful for logging.)
|
||||||
|
state_sock = None
|
||||||
|
if args.state_fps > 0:
|
||||||
|
state_sock = ctx.socket(zmq.PUB)
|
||||||
|
state_sock.setsockopt(zmq.SNDHWM, 2)
|
||||||
|
state_sock.setsockopt(zmq.LINGER, 0)
|
||||||
|
state_sock.bind(f"tcp://0.0.0.0:{args.state_port}")
|
||||||
|
logger.info("Publishing observation.state on :%d at %.0f Hz", args.state_port, args.state_fps)
|
||||||
|
|
||||||
|
def publish_state() -> None:
|
||||||
|
period = 1.0 / args.state_fps
|
||||||
|
joint_names = [j.name for j in G1_29_JointIndex]
|
||||||
|
while not stop.is_set():
|
||||||
|
t0 = time.time()
|
||||||
|
obs = robot.get_observation()
|
||||||
|
if obs:
|
||||||
|
state = {f"{name}.q": float(obs.get(f"{name}.q", 0.0)) for name in joint_names}
|
||||||
|
with contextlib.suppress(zmq.Again):
|
||||||
|
state_sock.send_json(state, zmq.NOBLOCK)
|
||||||
|
time.sleep(max(0.0, period - (time.time() - t0)))
|
||||||
|
|
||||||
|
threading.Thread(target=publish_state, daemon=True).start()
|
||||||
|
else:
|
||||||
|
logger.info("observation.state PUB disabled (--state-fps<=0)")
|
||||||
|
|
||||||
|
n = 0
|
||||||
|
try:
|
||||||
|
while not stop.is_set():
|
||||||
|
try:
|
||||||
|
payload = sock.recv()
|
||||||
|
except zmq.Again:
|
||||||
|
continue
|
||||||
|
except zmq.ContextTerminated:
|
||||||
|
break
|
||||||
|
|
||||||
|
try:
|
||||||
|
action = json.loads(payload.decode("utf-8"))
|
||||||
|
except (ValueError, UnicodeDecodeError) as e:
|
||||||
|
logger.warning("Dropping malformed action: %s", e)
|
||||||
|
continue
|
||||||
|
|
||||||
|
robot.send_action(action)
|
||||||
|
|
||||||
|
if grippers:
|
||||||
|
# L3 = remote.button.4 -> left, R3 = remote.button.0 -> right.
|
||||||
|
if "L" in grippers and "remote.button.4" in action:
|
||||||
|
grippers["L"].apply(bool(action["remote.button.4"]))
|
||||||
|
if "R" in grippers and "remote.button.0" in action:
|
||||||
|
grippers["R"].apply(bool(action["remote.button.0"]))
|
||||||
|
|
||||||
|
n += 1
|
||||||
|
if n % 60 == 0:
|
||||||
|
axes = {k: round(float(action.get(k, 0.0)), 3) for k in ("remote.lx", "remote.ly", "remote.rx", "remote.ry")}
|
||||||
|
logger.info("Applied %d actions | axes=%s", n, axes)
|
||||||
|
finally:
|
||||||
|
logger.info("Shutting down onboard controller...")
|
||||||
|
stop.set()
|
||||||
|
if state_sock is not None:
|
||||||
|
with contextlib.suppress(Exception):
|
||||||
|
state_sock.close(linger=0)
|
||||||
|
if camera_server is not None:
|
||||||
|
with contextlib.suppress(Exception):
|
||||||
|
camera_server.stop()
|
||||||
|
for g in grippers.values():
|
||||||
|
with contextlib.suppress(Exception):
|
||||||
|
g.bus.disconnect()
|
||||||
|
robot.disconnect()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -28,9 +28,11 @@ import argparse
|
|||||||
import base64
|
import base64
|
||||||
import contextlib
|
import contextlib
|
||||||
import json
|
import json
|
||||||
|
import re
|
||||||
import threading
|
import threading
|
||||||
import time
|
import time
|
||||||
from typing import Any
|
from dataclasses import dataclass
|
||||||
|
from typing import TYPE_CHECKING, Any
|
||||||
|
|
||||||
import zmq
|
import zmq
|
||||||
from unitree_sdk2py.comm.motion_switcher.motion_switcher_client import MotionSwitcherClient
|
from unitree_sdk2py.comm.motion_switcher.motion_switcher_client import MotionSwitcherClient
|
||||||
@@ -41,6 +43,9 @@ from unitree_sdk2py.utils.crc import CRC
|
|||||||
|
|
||||||
from lerobot.cameras.zmq.image_server import ImageServer
|
from lerobot.cameras.zmq.image_server import ImageServer
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from lerobot.motors.damiao.damiao import DamiaoMotorsBus
|
||||||
|
|
||||||
# DDS topic names follow Unitree SDK naming conventions
|
# DDS topic names follow Unitree SDK naming conventions
|
||||||
# ruff: noqa: N816
|
# ruff: noqa: N816
|
||||||
kTopicLowCommand_Debug = "rt/lowcmd" # action to robot
|
kTopicLowCommand_Debug = "rt/lowcmd" # action to robot
|
||||||
@@ -51,6 +56,105 @@ LOWSTATE_PORT = 6001
|
|||||||
NUM_MOTORS = 35
|
NUM_MOTORS = 35
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Gripper:
|
||||||
|
"""A single Damiao gripper that only writes to CAN when the open/close state changes."""
|
||||||
|
|
||||||
|
name: str
|
||||||
|
bus: "DamiaoMotorsBus"
|
||||||
|
open_deg: float
|
||||||
|
close_deg: float
|
||||||
|
_last_cmd: str | None = None # "open" | "close"
|
||||||
|
|
||||||
|
def apply(self, want_close: bool) -> None:
|
||||||
|
want = "close" if want_close else "open"
|
||||||
|
if want == self._last_cmd:
|
||||||
|
return
|
||||||
|
target = self.close_deg if want_close else self.open_deg
|
||||||
|
self.bus.write("Goal_Position", "gripper", target)
|
||||||
|
self._last_cmd = want
|
||||||
|
print(f"[gripper] {self.name} -> {want.upper()} ({target:.1f} deg)")
|
||||||
|
|
||||||
|
|
||||||
|
def build_gripper(
|
||||||
|
name: str,
|
||||||
|
port: str,
|
||||||
|
send_id: int,
|
||||||
|
recv_id: int,
|
||||||
|
motor_type: str,
|
||||||
|
use_can_fd: bool,
|
||||||
|
open_deg: float,
|
||||||
|
close_deg: float,
|
||||||
|
kp: float,
|
||||||
|
kd: float,
|
||||||
|
) -> Gripper:
|
||||||
|
from lerobot.motors.damiao.damiao import DamiaoMotorsBus
|
||||||
|
from lerobot.motors.motors_bus import Motor, MotorNormMode
|
||||||
|
|
||||||
|
motors = {
|
||||||
|
"gripper": Motor(
|
||||||
|
id=send_id,
|
||||||
|
model=motor_type,
|
||||||
|
norm_mode=MotorNormMode.DEGREES,
|
||||||
|
motor_type_str=motor_type,
|
||||||
|
recv_id=recv_id,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
bus = DamiaoMotorsBus(port=port, motors=motors, use_can_fd=use_can_fd)
|
||||||
|
print(f"Connecting {name} gripper on {port} (fd={use_can_fd})...")
|
||||||
|
bus.connect(handshake=True)
|
||||||
|
bus.write("Kp", "gripper", kp)
|
||||||
|
bus.write("Kd", "gripper", kd)
|
||||||
|
bus.write("Goal_Position", "gripper", open_deg) # start open
|
||||||
|
print(f" {name}: connected, torque enabled, opened.")
|
||||||
|
return Gripper(name, bus, open_deg, close_deg, _last_cmd="open")
|
||||||
|
|
||||||
|
|
||||||
|
def parse_camera_specs(spec: str, default_width: int, default_height: int) -> dict[str, dict]:
|
||||||
|
"""Parse a multi-camera spec string into an ImageServer ``cameras`` dict.
|
||||||
|
|
||||||
|
Format: comma-separated ``name:device[:WxH[:FOURCC]]`` entries, e.g.
|
||||||
|
``head_camera:6,left_wrist:0``. ``device`` may be an integer index or an explicit
|
||||||
|
device path (e.g. ``/dev/video6``), including stable ``by-path`` names like
|
||||||
|
``/dev/v4l/by-path/platform-...:2.1:1.3-video-index0`` which survive USB
|
||||||
|
re-enumeration (unlike bare ``/dev/videoN`` indices). Because a by-path name
|
||||||
|
itself contains colons, the optional ``WxH`` and ``FOURCC`` are parsed from the
|
||||||
|
*right* so the device-path colons are preserved.
|
||||||
|
"""
|
||||||
|
wh_re = re.compile(r"\d+x\d+", re.IGNORECASE)
|
||||||
|
fourcc_re = re.compile(r"[A-Za-z0-9]{4}")
|
||||||
|
|
||||||
|
cameras: dict[str, dict] = {}
|
||||||
|
for entry in spec.split(","):
|
||||||
|
entry = entry.strip()
|
||||||
|
if not entry:
|
||||||
|
continue
|
||||||
|
if ":" not in entry:
|
||||||
|
raise ValueError(f"Invalid camera spec '{entry}', expected 'name:device[:WxH[:FOURCC]]'")
|
||||||
|
name, rest = entry.split(":", 1)
|
||||||
|
name = name.strip()
|
||||||
|
tokens = [t.strip() for t in rest.split(":")]
|
||||||
|
|
||||||
|
fourcc = None
|
||||||
|
if len(tokens) >= 3 and wh_re.fullmatch(tokens[-2]) and fourcc_re.fullmatch(tokens[-1]):
|
||||||
|
fourcc = tokens.pop().upper()
|
||||||
|
width, height = default_width, default_height
|
||||||
|
if len(tokens) >= 2 and wh_re.fullmatch(tokens[-1]):
|
||||||
|
w, h = tokens.pop().lower().split("x")
|
||||||
|
width, height = int(w), int(h)
|
||||||
|
|
||||||
|
raw_id = ":".join(tokens).strip()
|
||||||
|
if not raw_id:
|
||||||
|
raise ValueError(f"Invalid camera spec '{entry}', missing device")
|
||||||
|
device_id: int | str = int(raw_id) if raw_id.lstrip("-").isdigit() else raw_id
|
||||||
|
if name in cameras:
|
||||||
|
raise ValueError(f"Duplicate camera name '{name}' in --cameras")
|
||||||
|
cameras[name] = {"device_id": device_id, "shape": [height, width], "fourcc": fourcc}
|
||||||
|
if not cameras:
|
||||||
|
raise ValueError("No cameras parsed from --cameras spec")
|
||||||
|
return cameras
|
||||||
|
|
||||||
|
|
||||||
def lowstate_to_dict(msg: hg_LowState) -> dict[str, Any]:
|
def lowstate_to_dict(msg: hg_LowState) -> dict[str, Any]:
|
||||||
"""Convert LowState SDK message to a JSON-serializable dictionary."""
|
"""Convert LowState SDK message to a JSON-serializable dictionary."""
|
||||||
motor_states = []
|
motor_states = []
|
||||||
@@ -155,7 +259,11 @@ def main() -> None:
|
|||||||
"""Main entry point for the robot server bridge."""
|
"""Main entry point for the robot server bridge."""
|
||||||
parser = argparse.ArgumentParser(description="DDS-to-ZMQ bridge server for Unitree G1")
|
parser = argparse.ArgumentParser(description="DDS-to-ZMQ bridge server for Unitree G1")
|
||||||
parser.add_argument("--camera", action="store_true", help="Also launch camera server")
|
parser.add_argument("--camera", action="store_true", help="Also launch camera server")
|
||||||
parser.add_argument("--camera-device", type=int, default=4, help="Camera device ID (default: 4)")
|
parser.add_argument("--camera-device", default="4",
|
||||||
|
help="Camera device: index or /dev/video path or by-path name (default: 4)")
|
||||||
|
parser.add_argument("--cameras", default=None,
|
||||||
|
help="Multi-camera spec 'name:device[:WxH[:FOURCC]]', comma-separated. Overrides "
|
||||||
|
"--camera-device; device may be a by-path name to survive USB re-enumeration.")
|
||||||
parser.add_argument("--camera-fps", type=int, default=30, help="Camera FPS (default: 30)")
|
parser.add_argument("--camera-fps", type=int, default=30, help="Camera FPS (default: 30)")
|
||||||
parser.add_argument("--camera-width", type=int, default=640, help="Camera width (default: 640)")
|
parser.add_argument("--camera-width", type=int, default=640, help="Camera width (default: 640)")
|
||||||
parser.add_argument("--camera-height", type=int, default=480, help="Camera height (default: 480)")
|
parser.add_argument("--camera-height", type=int, default=480, help="Camera height (default: 480)")
|
||||||
@@ -164,20 +272,20 @@ def main() -> None:
|
|||||||
|
|
||||||
# Optionally start camera server in background thread
|
# Optionally start camera server in background thread
|
||||||
camera_thread = None
|
camera_thread = None
|
||||||
if args.camera:
|
if args.camera or args.cameras:
|
||||||
camera_config = {
|
if args.cameras:
|
||||||
"fps": args.camera_fps,
|
cameras = parse_camera_specs(args.cameras, args.camera_width, args.camera_height)
|
||||||
"cameras": {
|
else:
|
||||||
"head_camera": {
|
# Single camera; accept an int index or a device/by-path string.
|
||||||
"device_id": args.camera_device,
|
dev = args.camera_device
|
||||||
"shape": [args.camera_height, args.camera_width],
|
dev = int(dev) if str(dev).lstrip("-").isdigit() else dev
|
||||||
}
|
cameras = {"head_camera": {"device_id": dev, "shape": [args.camera_height, args.camera_width]}}
|
||||||
},
|
camera_config = {"fps": args.camera_fps, "cameras": cameras}
|
||||||
}
|
|
||||||
camera_server = ImageServer(camera_config, port=args.camera_port)
|
camera_server = ImageServer(camera_config, port=args.camera_port)
|
||||||
camera_thread = threading.Thread(target=camera_server.run, daemon=True)
|
camera_thread = threading.Thread(target=camera_server.run, daemon=True)
|
||||||
camera_thread.start()
|
camera_thread.start()
|
||||||
print(f"Camera server started on port {args.camera_port} (device {args.camera_device})")
|
cam_summary = ", ".join(f"{n}(dev {c['device_id']})" for n, c in cameras.items())
|
||||||
|
print(f"Camera server started on port {args.camera_port}: {cam_summary}")
|
||||||
|
|
||||||
# initialize DDS
|
# initialize DDS
|
||||||
ChannelFactoryInitialize(0)
|
ChannelFactoryInitialize(0)
|
||||||
|
|||||||
@@ -33,12 +33,14 @@ from ..robot import Robot
|
|||||||
from .config_unitree_g1 import UnitreeG1Config
|
from .config_unitree_g1 import UnitreeG1Config
|
||||||
from .g1_kinematics import G1_29_ArmIK
|
from .g1_kinematics import G1_29_ArmIK
|
||||||
from .g1_utils import (
|
from .g1_utils import (
|
||||||
|
KEYBOARD_KEYS_FIELD,
|
||||||
REMOTE_AXES,
|
REMOTE_AXES,
|
||||||
REMOTE_KEYS,
|
|
||||||
G1_29_JointArmIndex,
|
G1_29_JointArmIndex,
|
||||||
G1_29_JointIndex,
|
G1_29_JointIndex,
|
||||||
default_remote_input,
|
default_remote_input,
|
||||||
|
lowstate_to_obs,
|
||||||
make_locomotion_controller,
|
make_locomotion_controller,
|
||||||
|
obs_to_wb34_state,
|
||||||
)
|
)
|
||||||
|
|
||||||
if TYPE_CHECKING or _unitree_sdk_available:
|
if TYPE_CHECKING or _unitree_sdk_available:
|
||||||
@@ -47,8 +49,12 @@ if TYPE_CHECKING or _unitree_sdk_available:
|
|||||||
ChannelPublisher as _SDKChannelPublisher,
|
ChannelPublisher as _SDKChannelPublisher,
|
||||||
ChannelSubscriber as _SDKChannelSubscriber,
|
ChannelSubscriber as _SDKChannelSubscriber,
|
||||||
)
|
)
|
||||||
from unitree_sdk2py.idl.default import unitree_hg_msg_dds__LowCmd_
|
from unitree_sdk2py.idl.default import (
|
||||||
|
unitree_hg_msg_dds__HandCmd_ as hg_HandCmd_default,
|
||||||
|
unitree_hg_msg_dds__LowCmd_,
|
||||||
|
)
|
||||||
from unitree_sdk2py.idl.unitree_hg.msg.dds_ import (
|
from unitree_sdk2py.idl.unitree_hg.msg.dds_ import (
|
||||||
|
HandCmd_ as hg_HandCmd,
|
||||||
LowCmd_ as hg_LowCmd,
|
LowCmd_ as hg_LowCmd,
|
||||||
LowState_ as hg_LowState,
|
LowState_ as hg_LowState,
|
||||||
)
|
)
|
||||||
@@ -58,6 +64,8 @@ else:
|
|||||||
_SDKChannelPublisher = None
|
_SDKChannelPublisher = None
|
||||||
_SDKChannelSubscriber = None
|
_SDKChannelSubscriber = None
|
||||||
unitree_hg_msg_dds__LowCmd_ = None
|
unitree_hg_msg_dds__LowCmd_ = None
|
||||||
|
hg_HandCmd_default = None
|
||||||
|
hg_HandCmd = None
|
||||||
hg_LowCmd = None
|
hg_LowCmd = None
|
||||||
hg_LowState = None
|
hg_LowState = None
|
||||||
CRC = None
|
CRC = None
|
||||||
@@ -79,6 +87,14 @@ class LocomotionController(Protocol):
|
|||||||
kTopicLowCommand_Debug = "rt/lowcmd"
|
kTopicLowCommand_Debug = "rt/lowcmd"
|
||||||
kTopicLowState = "rt/lowstate"
|
kTopicLowState = "rt/lowstate"
|
||||||
|
|
||||||
|
# Wireless-remote button byte layout, mapped to the positional button indices the
|
||||||
|
# locomotion controllers expect. Used in onboard mode to read the physical Unitree
|
||||||
|
# remote from lowstate (mirrors the exo teleoperator's RemoteController).
|
||||||
|
_REMOTE_BUTTON_MAP: list[str] = [
|
||||||
|
"RB", "LB", "start", "back", "RT", "LT", "", "",
|
||||||
|
"A", "B", "X", "Y", "up", "right", "down", "left",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class MotorState:
|
class MotorState:
|
||||||
@@ -122,8 +138,10 @@ class UnitreeG1(Robot):
|
|||||||
# Initialize cameras config (ZMQ-based) - actual connection in connect()
|
# Initialize cameras config (ZMQ-based) - actual connection in connect()
|
||||||
self._cameras = make_cameras_from_configs(config.cameras)
|
self._cameras = make_cameras_from_configs(config.cameras)
|
||||||
|
|
||||||
# Import channel classes based on mode
|
# Import channel classes based on mode. Simulation and onboard both talk to a
|
||||||
if config.is_simulation:
|
# real (local) DDS via the Unitree SDK; only the laptop-side bridge client uses
|
||||||
|
# the ZMQ socket shim.
|
||||||
|
if config.is_simulation or config.onboard:
|
||||||
self._ChannelFactoryInitialize = _SDKChannelFactoryInitialize
|
self._ChannelFactoryInitialize = _SDKChannelFactoryInitialize
|
||||||
self._ChannelPublisher = _SDKChannelPublisher
|
self._ChannelPublisher = _SDKChannelPublisher
|
||||||
self._ChannelSubscriber = _SDKChannelSubscriber
|
self._ChannelSubscriber = _SDKChannelSubscriber
|
||||||
@@ -151,19 +169,100 @@ class UnitreeG1(Robot):
|
|||||||
# Lower-body controller loaded dynamically
|
# Lower-body controller loaded dynamically
|
||||||
self.controller: LocomotionController | None = make_locomotion_controller(config.controller)
|
self.controller: LocomotionController | None = make_locomotion_controller(config.controller)
|
||||||
|
|
||||||
|
# Token-driven deploy: let a SONIC controller hold a neutral token until the
|
||||||
|
# first real one arrives, then hold the last token between control ticks.
|
||||||
|
if config.sonic_token_action and hasattr(self.controller, "token_mode"):
|
||||||
|
self.controller.token_mode = True
|
||||||
|
|
||||||
# Controller thread state
|
# Controller thread state
|
||||||
self._controller_thread = None
|
self._controller_thread = None
|
||||||
|
# When set, the controller loop stops publishing low commands so reset() can
|
||||||
|
# drive the joints directly without two publishers fighting (single-publisher).
|
||||||
|
self._controller_paused = threading.Event()
|
||||||
self._controller_action_lock = threading.Lock()
|
self._controller_action_lock = threading.Lock()
|
||||||
self.controller_input = default_remote_input()
|
self.controller_input = default_remote_input()
|
||||||
self.controller_output = {}
|
self.controller_output = {}
|
||||||
|
|
||||||
|
# Onboard-only: parser for the physical Unitree wireless remote (read straight
|
||||||
|
# from local lowstate so joystick locomotion works without a laptop round-trip).
|
||||||
|
self._joystick = None
|
||||||
|
|
||||||
|
# Replay-camera state: keep the encoded (raw) cells per camera and decode
|
||||||
|
# frames lazily as the play cursor advances, with a small frame cache, so we
|
||||||
|
# don't materialize gigabytes of decoded RGB at construction time.
|
||||||
|
self._replay_raw: dict[str, list] = {}
|
||||||
|
self._replay_cache: dict[tuple[str, int], np.ndarray] = {}
|
||||||
|
self._replay_cache_cap = 8
|
||||||
|
self._replay_len = 0
|
||||||
|
self._replay_idx = 0
|
||||||
|
if config.replay_camera_parquet and config.replay_camera_map:
|
||||||
|
self._load_replay_frames()
|
||||||
|
|
||||||
|
# Token-mode state: last 64-D SONIC latent token commanded by the policy,
|
||||||
|
# echoed back as ``observation.state`` so a token-output VLA closes the loop
|
||||||
|
# on its own previous token (see ``sonic_token_action``). Seeded to zeros;
|
||||||
|
# the controller's startup blend eases joints in regardless.
|
||||||
|
self._last_token: np.ndarray | None = None
|
||||||
|
if config.sonic_token_action:
|
||||||
|
from .controllers.sonic_whole_body import TOKEN_DIM
|
||||||
|
|
||||||
|
self._last_token = np.zeros(TOKEN_DIM, dtype=np.float32)
|
||||||
|
|
||||||
|
def _load_replay_frames(self) -> None:
|
||||||
|
"""Load only the mapped parquet columns (encoded frames); decode on demand."""
|
||||||
|
import pyarrow.parquet as pq
|
||||||
|
|
||||||
|
cols_needed = list(dict.fromkeys(self.config.replay_camera_map.values()))
|
||||||
|
table = pq.read_table(self.config.replay_camera_parquet, columns=cols_needed)
|
||||||
|
self._replay_len = table.num_rows
|
||||||
|
self._replay_raw = {
|
||||||
|
cam_name: table.column(column).to_pylist()
|
||||||
|
for cam_name, column in self.config.replay_camera_map.items()
|
||||||
|
}
|
||||||
|
logger.info(
|
||||||
|
"Loaded %d replay frames (lazy-decode) for cameras %s from %s",
|
||||||
|
self._replay_len,
|
||||||
|
list(self.config.replay_camera_map),
|
||||||
|
self.config.replay_camera_parquet,
|
||||||
|
)
|
||||||
|
|
||||||
|
def _decode_replay_cell(self, cell) -> np.ndarray:
|
||||||
|
import io
|
||||||
|
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
data = cell["bytes"] if isinstance(cell, dict) else cell
|
||||||
|
return np.asarray(Image.open(io.BytesIO(data)).convert("RGB"), dtype=np.uint8)
|
||||||
|
|
||||||
|
def _replay_frame(self, cam_name: str, idx: int) -> np.ndarray:
|
||||||
|
"""Decode (and briefly cache) a single replay frame for a camera."""
|
||||||
|
key = (cam_name, idx)
|
||||||
|
cached = self._replay_cache.get(key)
|
||||||
|
if cached is not None:
|
||||||
|
return cached
|
||||||
|
frame = self._decode_replay_cell(self._replay_raw[cam_name][idx])
|
||||||
|
if len(self._replay_cache) >= self._replay_cache_cap:
|
||||||
|
self._replay_cache.pop(next(iter(self._replay_cache)))
|
||||||
|
self._replay_cache[key] = frame
|
||||||
|
return frame
|
||||||
|
|
||||||
def _subscribe_lowstate(self): # polls robot state @ 250Hz
|
def _subscribe_lowstate(self): # polls robot state @ 250Hz
|
||||||
while not self._shutdown_event.is_set():
|
while not self._shutdown_event.is_set():
|
||||||
start_time = time.time()
|
start_time = time.time()
|
||||||
|
|
||||||
# Step simulation if in simulation mode
|
# Step simulation if in simulation mode
|
||||||
if self.config.is_simulation and self.sim_env is not None:
|
if self.config.is_simulation and self.sim_env is not None:
|
||||||
self.sim_env.step()
|
try:
|
||||||
|
self.sim_env.step()
|
||||||
|
except ValueError as e:
|
||||||
|
# Startup race: the sim thread can step once before reset() has
|
||||||
|
# written a valid base pose, giving a zero-norm pelvis quaternion
|
||||||
|
# (scipy>=1.11 raises instead of normalizing). Skip and retry so
|
||||||
|
# the thread survives instead of dying and freezing the sim.
|
||||||
|
if "zero norm" not in str(e).lower():
|
||||||
|
raise
|
||||||
|
time.sleep(self.control_dt)
|
||||||
|
continue
|
||||||
|
|
||||||
msg = self.lowstate_subscriber.Read()
|
msg = self.lowstate_subscriber.Read()
|
||||||
if msg is not None:
|
if msg is not None:
|
||||||
@@ -231,15 +330,80 @@ class UnitreeG1(Robot):
|
|||||||
features[f"{cam}_depth"] = (cfg.height, cfg.width, 1)
|
features[f"{cam}_depth"] = (cfg.height, cfg.width, 1)
|
||||||
return features
|
return features
|
||||||
|
|
||||||
|
@property
|
||||||
|
def _wb_state_ft(self) -> dict[str, type]:
|
||||||
|
"""34-D whole-body proprio state (``wb_state.{i}.pos``) for dense controllers.
|
||||||
|
|
||||||
|
Exposed only when the controller consumes a dense whole-body command
|
||||||
|
(OpenHLM / pi0.5). These ``.pos`` scalars are aggregated by the rollout
|
||||||
|
pipeline into a single 34-D ``observation.state`` for the policy.
|
||||||
|
"""
|
||||||
|
if self.config.sonic_token_action:
|
||||||
|
return {}
|
||||||
|
if not getattr(self.controller, "wb_action", False):
|
||||||
|
return {}
|
||||||
|
from .g1_utils import WB_ACTION_DIM
|
||||||
|
|
||||||
|
return {f"wb_state.{i}.pos": float for i in range(WB_ACTION_DIM)}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def _token_state_ft(self) -> dict[str, type]:
|
||||||
|
"""64-D SONIC latent-token proprio state (``motion_token_state.{i}.pos``).
|
||||||
|
|
||||||
|
Exposed only in ``sonic_token_action`` mode; aggregated by the rollout into a
|
||||||
|
64-D ``observation.state`` (the last token the policy commanded).
|
||||||
|
"""
|
||||||
|
if not self.config.sonic_token_action:
|
||||||
|
return {}
|
||||||
|
from .controllers.sonic_whole_body import TOKEN_DIM, token_state_key
|
||||||
|
|
||||||
|
return {token_state_key(i): float for i in range(TOKEN_DIM)}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def _empty_cameras_ft(self) -> dict[str, tuple]:
|
||||||
|
"""Synthetic zero-image cameras (see ``UnitreeG1Config.empty_cameras``)."""
|
||||||
|
h, w = self.config.empty_camera_hw
|
||||||
|
return dict.fromkeys(self.config.empty_cameras, (h, w, 3))
|
||||||
|
|
||||||
|
@property
|
||||||
|
def _replay_cameras_ft(self) -> dict[str, tuple]:
|
||||||
|
"""Replay cameras, shaped from their first (lazily decoded) frame."""
|
||||||
|
if not self._replay_len:
|
||||||
|
return {}
|
||||||
|
return {name: self._replay_frame(name, 0).shape for name in self._replay_raw}
|
||||||
|
|
||||||
@cached_property
|
@cached_property
|
||||||
def observation_features(self) -> dict[str, type | tuple]:
|
def observation_features(self) -> dict[str, type | tuple]:
|
||||||
return {**self._motors_ft, **self._cameras_ft}
|
return {
|
||||||
|
**self._motors_ft,
|
||||||
|
**self._wb_state_ft,
|
||||||
|
**self._token_state_ft,
|
||||||
|
**self._empty_cameras_ft,
|
||||||
|
**self._replay_cameras_ft,
|
||||||
|
**self._cameras_ft,
|
||||||
|
}
|
||||||
|
|
||||||
@cached_property
|
@cached_property
|
||||||
def action_features(self) -> dict[str, type]:
|
def action_features(self) -> dict[str, type]:
|
||||||
if self.controller is None:
|
if self.controller is None:
|
||||||
return {f"{G1_29_JointIndex(motor).name}.q": float for motor in G1_29_JointIndex}
|
return {f"{G1_29_JointIndex(motor).name}.q": float for motor in G1_29_JointIndex}
|
||||||
|
|
||||||
|
# Token-output VLA (SONIC decoder): advertise a 64-D latent-token action space
|
||||||
|
# (``motion_token.{i}.pos``) so ``lerobot-rollout`` maps a 64-D policy output
|
||||||
|
# straight onto the decoder, bypassing the encoder.
|
||||||
|
if self.config.sonic_token_action:
|
||||||
|
from .controllers.sonic_whole_body import TOKEN_DIM, token_action_key
|
||||||
|
|
||||||
|
return {token_action_key(i): float for i in range(TOKEN_DIM)}
|
||||||
|
|
||||||
|
# Dense whole-body controllers (SONIC / OpenHLM, pi0.5) consume a single
|
||||||
|
# 34-D command per tick. Expose it as ``wb.{i}.pos`` joint-position features
|
||||||
|
# so ``lerobot-rollout`` maps a 34-D policy output straight onto the robot.
|
||||||
|
if getattr(self.controller, "wb_action", False):
|
||||||
|
from .g1_utils import WB_ACTION_DIM, wb_action_key
|
||||||
|
|
||||||
|
return {wb_action_key(i): float for i in range(WB_ACTION_DIM)}
|
||||||
|
|
||||||
arm_features = {f"{G1_29_JointArmIndex(motor).name}.q": float for motor in G1_29_JointArmIndex}
|
arm_features = {f"{G1_29_JointArmIndex(motor).name}.q": float for motor in G1_29_JointArmIndex}
|
||||||
remote_features = dict.fromkeys(REMOTE_AXES, float)
|
remote_features = dict.fromkeys(REMOTE_AXES, float)
|
||||||
return {**arm_features, **remote_features}
|
return {**arm_features, **remote_features}
|
||||||
@@ -255,6 +419,11 @@ class UnitreeG1(Robot):
|
|||||||
while not self._shutdown_event.is_set():
|
while not self._shutdown_event.is_set():
|
||||||
start_time = time.time()
|
start_time = time.time()
|
||||||
|
|
||||||
|
# Paused during reset() so the reset routine is the sole low-cmd publisher.
|
||||||
|
if self._controller_paused.is_set():
|
||||||
|
time.sleep(control_dt)
|
||||||
|
continue
|
||||||
|
|
||||||
with self._lowstate_lock:
|
with self._lowstate_lock:
|
||||||
lowstate = self._lowstate
|
lowstate = self._lowstate
|
||||||
|
|
||||||
@@ -271,6 +440,13 @@ class UnitreeG1(Robot):
|
|||||||
with self._controller_action_lock:
|
with self._controller_action_lock:
|
||||||
controller_input = dict(self.controller_input)
|
controller_input = dict(self.controller_input)
|
||||||
|
|
||||||
|
# Onboard: the physical Unitree remote (in local lowstate) takes
|
||||||
|
# priority for locomotion when active; otherwise laptop/ZMQ axes stand.
|
||||||
|
if self.config.onboard:
|
||||||
|
wl = self._wireless_remote_input(lowstate)
|
||||||
|
if wl is not None:
|
||||||
|
controller_input.update(wl)
|
||||||
|
|
||||||
# Run controller step
|
# Run controller step
|
||||||
controller_action = self.controller.run_step(controller_input, lowstate)
|
controller_action = self.controller.run_step(controller_input, lowstate)
|
||||||
|
|
||||||
@@ -293,15 +469,105 @@ class UnitreeG1(Robot):
|
|||||||
def configure(self) -> None:
|
def configure(self) -> None:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
def _wireless_remote_input(self, lowstate) -> dict | None:
|
||||||
|
"""Parse the physical Unitree remote from lowstate into controller inputs.
|
||||||
|
|
||||||
|
Onboard only. Returns None when the remote is idle so the laptop-provided
|
||||||
|
(ZMQ) axes keep control; otherwise the physical remote takes priority.
|
||||||
|
"""
|
||||||
|
js = self._joystick
|
||||||
|
if js is None:
|
||||||
|
return None
|
||||||
|
wr = getattr(lowstate, "wireless_remote", None)
|
||||||
|
if not wr or len(wr) < 24:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
js.extract(wr)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
return None
|
||||||
|
|
||||||
|
axes = {
|
||||||
|
"remote.lx": float(js.lx.data),
|
||||||
|
"remote.ly": float(js.ly.data),
|
||||||
|
"remote.rx": float(js.rx.data),
|
||||||
|
"remote.ry": float(js.ry.data),
|
||||||
|
}
|
||||||
|
active = any(abs(v) > 1e-2 for v in axes.values())
|
||||||
|
out = dict(axes)
|
||||||
|
for i, name in enumerate(_REMOTE_BUTTON_MAP):
|
||||||
|
if name:
|
||||||
|
val = float(getattr(js, name).data)
|
||||||
|
out[f"remote.button.{i}"] = val
|
||||||
|
if val:
|
||||||
|
active = True
|
||||||
|
return out if active else None
|
||||||
|
|
||||||
|
def _release_motion_control(self) -> None:
|
||||||
|
"""Release the robot's built-in motion services so we can send raw lowcmd.
|
||||||
|
|
||||||
|
Onboard-only. Mirrors run_g1_server.py: on the real robot the factory
|
||||||
|
locomotion/hand services must relinquish control before our controller can
|
||||||
|
write to ``rt/lowcmd``, otherwise commands are ignored or fought.
|
||||||
|
"""
|
||||||
|
from unitree_sdk2py.comm.motion_switcher.motion_switcher_client import MotionSwitcherClient
|
||||||
|
|
||||||
|
msc = MotionSwitcherClient()
|
||||||
|
msc.SetTimeout(5.0)
|
||||||
|
msc.Init()
|
||||||
|
_, result = msc.CheckMode()
|
||||||
|
while result is not None and "name" in result and result["name"]:
|
||||||
|
logger.info("[UnitreeG1] Releasing built-in mode '%s'...", result["name"])
|
||||||
|
msc.ReleaseMode()
|
||||||
|
_, result = msc.CheckMode()
|
||||||
|
time.sleep(1.0)
|
||||||
|
|
||||||
def connect(self, calibrate: bool = True) -> None: # connect to DDS
|
def connect(self, calibrate: bool = True) -> None: # connect to DDS
|
||||||
# Initialize DDS channel and simulation environment
|
# Initialize DDS channel and simulation environment
|
||||||
if self.config.is_simulation:
|
if self.config.is_simulation:
|
||||||
from lerobot.envs import make_env
|
from lerobot.envs.utils import (
|
||||||
|
_download_hub_file,
|
||||||
|
_import_hub_module,
|
||||||
|
_normalize_hub_result,
|
||||||
|
)
|
||||||
|
|
||||||
self._ChannelFactoryInitialize(0, "lo")
|
self._ChannelFactoryInitialize(0, "lo")
|
||||||
self._env_wrapper = make_env("lerobot/unitree-g1-mujoco", trust_remote_code=True)
|
# Call the hub env's make_env directly so we can disable the offscreen
|
||||||
|
# head_camera renderer. We drive image-conditioned policies from recorded
|
||||||
|
# frames (see replay_camera_parquet / external obs), never the sim's own
|
||||||
|
# camera, so building a MuJoCo offscreen GL context is pure liability: it
|
||||||
|
# crashes with "Failed to make the EGL context current" when GLFW/SDL
|
||||||
|
# already own a context, killing the sim thread and hanging on
|
||||||
|
# "Waiting for robot state...". publish_images=False -> no renderer.
|
||||||
|
repo_id, _, local_file, _ = _download_hub_file(
|
||||||
|
"lerobot/unitree-g1-mujoco", True, None
|
||||||
|
)
|
||||||
|
hub_mod = _import_hub_module(local_file, repo_id)
|
||||||
|
raw = hub_mod.make_env(n_envs=1, use_async_envs=False, publish_images=False, cameras=[])
|
||||||
|
self._env_wrapper = _normalize_hub_result(raw)
|
||||||
# Extract the actual gym env from the dict structure
|
# Extract the actual gym env from the dict structure
|
||||||
self.sim_env = self._env_wrapper["hub_env"][0].envs[0]
|
self.sim_env = self._env_wrapper["hub_env"][0].envs[0]
|
||||||
|
elif self.config.onboard:
|
||||||
|
# Real robot, controller running onboard against local DDS. Initialize the
|
||||||
|
# real SDK channel factory on the robot's DDS interface and take low-level
|
||||||
|
# control from the built-in services before we start writing lowcmd.
|
||||||
|
if self.config.dds_interface:
|
||||||
|
self._ChannelFactoryInitialize(0, self.config.dds_interface)
|
||||||
|
else:
|
||||||
|
self._ChannelFactoryInitialize(0)
|
||||||
|
# Real robot: hand low-level control over from the built-in services.
|
||||||
|
# A DDS sim has no MotionSwitcher, so this is skipped there.
|
||||||
|
if self.config.release_motion_control:
|
||||||
|
self._release_motion_control()
|
||||||
|
# Real robot: read the physical wireless remote from lowstate for
|
||||||
|
# locomotion. A sim has no physical remote, so leave _joystick=None and
|
||||||
|
# let send_action (ZMQ) drive the locomotion axes instead.
|
||||||
|
if self.config.physical_remote:
|
||||||
|
from unitree_sdk2py.utils.joystick import Joystick
|
||||||
|
|
||||||
|
self._joystick = Joystick()
|
||||||
|
for axis in (self._joystick.lx, self._joystick.ly, self._joystick.rx, self._joystick.ry):
|
||||||
|
axis.smooth = 1.0
|
||||||
|
axis.deadzone = 0.0
|
||||||
else:
|
else:
|
||||||
self._ChannelFactoryInitialize(0, config=self.config)
|
self._ChannelFactoryInitialize(0, config=self.config)
|
||||||
|
|
||||||
@@ -311,6 +577,17 @@ class UnitreeG1(Robot):
|
|||||||
self.lowstate_subscriber = self._ChannelSubscriber(kTopicLowState, hg_LowState)
|
self.lowstate_subscriber = self._ChannelSubscriber(kTopicLowState, hg_LowState)
|
||||||
self.lowstate_subscriber.Init()
|
self.lowstate_subscriber.Init()
|
||||||
|
|
||||||
|
# Dex3 hand command publishers (grasping). Driven by the OpenHLM grip scalars.
|
||||||
|
self._hand_publishers = {}
|
||||||
|
if self.config.publish_hands:
|
||||||
|
self._left_hand_cmd = hg_HandCmd_default()
|
||||||
|
self._right_hand_cmd = hg_HandCmd_default()
|
||||||
|
self._hand_publishers["left"] = self._ChannelPublisher("rt/dex3/left/cmd", hg_HandCmd)
|
||||||
|
self._hand_publishers["right"] = self._ChannelPublisher("rt/dex3/right/cmd", hg_HandCmd)
|
||||||
|
for pub in self._hand_publishers.values():
|
||||||
|
pub.Init()
|
||||||
|
logger.info("Dex3 hand command publishers initialized (rt/dex3/{left,right}/cmd)")
|
||||||
|
|
||||||
# Start subscribe thread to read robot state
|
# Start subscribe thread to read robot state
|
||||||
self.subscribe_thread = threading.Thread(target=self._subscribe_lowstate)
|
self.subscribe_thread = threading.Thread(target=self._subscribe_lowstate)
|
||||||
self.subscribe_thread.start()
|
self.subscribe_thread.start()
|
||||||
@@ -343,6 +620,9 @@ class UnitreeG1(Robot):
|
|||||||
|
|
||||||
self.kp = np.array(self.config.kp, dtype=np.float32)
|
self.kp = np.array(self.config.kp, dtype=np.float32)
|
||||||
self.kd = np.array(self.config.kd, dtype=np.float32)
|
self.kd = np.array(self.config.kd, dtype=np.float32)
|
||||||
|
if self.controller is not None and hasattr(self.controller, "kp"):
|
||||||
|
self.kp = np.array(self.controller.kp, dtype=np.float32)
|
||||||
|
self.kd = np.array(self.controller.kd, dtype=np.float32)
|
||||||
|
|
||||||
for joint in G1_29_JointIndex:
|
for joint in G1_29_JointIndex:
|
||||||
self.msg.motor_cmd[joint].mode = 1
|
self.msg.motor_cmd[joint].mode = 1
|
||||||
@@ -350,12 +630,16 @@ class UnitreeG1(Robot):
|
|||||||
self.msg.motor_cmd[joint].kd = self.kd[joint.value]
|
self.msg.motor_cmd[joint].kd = self.kd[joint.value]
|
||||||
self.msg.motor_cmd[joint].q = lowstate.motor_state[joint.value].q
|
self.msg.motor_cmd[joint].q = lowstate.motor_state[joint.value].q
|
||||||
|
|
||||||
# Start controller thread if enabled
|
# Start controller thread if enabled. Skipped when run_controller_thread is
|
||||||
if self.controller is not None:
|
# False so a caller can step the controller synchronously (faithful replay).
|
||||||
|
if self.controller is not None and self.config.run_controller_thread:
|
||||||
self._controller_thread = threading.Thread(target=self._controller_loop, daemon=True)
|
self._controller_thread = threading.Thread(target=self._controller_loop, daemon=True)
|
||||||
self._controller_thread.start()
|
self._controller_thread.start()
|
||||||
fps = int(1.0 / self.controller.control_dt)
|
fps = int(1.0 / self.controller.control_dt)
|
||||||
logger.info(f"Controller thread started ({fps}Hz)")
|
logger.info(f"Controller thread started ({fps}Hz)")
|
||||||
|
elif self.controller is not None:
|
||||||
|
logger.info("Controller thread disabled (run_controller_thread=False); "
|
||||||
|
"caller must drive controller.run_step synchronously.")
|
||||||
|
|
||||||
def _send_zero_torque(self) -> None:
|
def _send_zero_torque(self) -> None:
|
||||||
"""Send a zero-gain command to make joints passive before shutting down."""
|
"""Send a zero-gain command to make joints passive before shutting down."""
|
||||||
@@ -371,13 +655,59 @@ class UnitreeG1(Robot):
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning(f"Failed to send zero-torque on disconnect: {e}")
|
logger.warning(f"Failed to send zero-torque on disconnect: {e}")
|
||||||
|
|
||||||
def disconnect(self):
|
def _graceful_stop(self) -> None:
|
||||||
# Put robot in passive mode before stopping threads
|
"""Soft shutdown: hold the current pose and ramp joint stiffness (kp) to zero
|
||||||
if not self.config.is_simulation:
|
over ``graceful_stop_s`` while keeping damping (kd), then go passive.
|
||||||
self._send_zero_torque()
|
|
||||||
|
|
||||||
# Signal thread to stop and unblock any waits
|
Prevents the robot from collapsing the instant control ends (a bare
|
||||||
|
zero-torque command is kp=kd=0 ≈ free-fall). Must run after the controller
|
||||||
|
loop has stopped so the two aren't publishing at once.
|
||||||
|
"""
|
||||||
|
if self.config.graceful_stop_s <= 0:
|
||||||
|
self._send_zero_torque()
|
||||||
|
return
|
||||||
|
with self._lowstate_lock:
|
||||||
|
lowstate = self._lowstate
|
||||||
|
if lowstate is None:
|
||||||
|
self._send_zero_torque()
|
||||||
|
return
|
||||||
|
q_hold = {f"{motor.name}.q": lowstate.motor_state[motor.value].q for motor in G1_29_JointIndex}
|
||||||
|
kp = np.array(self.kp, dtype=np.float32)
|
||||||
|
kd = np.array(self.kd, dtype=np.float32)
|
||||||
|
zeros = np.zeros(29, dtype=np.float32)
|
||||||
|
dt = self.controller.control_dt if self.controller is not None else self.config.control_dt
|
||||||
|
steps = max(1, int(self.config.graceful_stop_s / dt))
|
||||||
|
logger.info("Graceful stop: damping down over %.1fs", self.config.graceful_stop_s)
|
||||||
|
for i in range(steps):
|
||||||
|
ratio = (i + 1) / steps
|
||||||
|
self.publish_lowcmd(q_hold, kp=kp * (1.0 - ratio), kd=kd, tau=zeros)
|
||||||
|
time.sleep(dt)
|
||||||
|
self._send_zero_torque()
|
||||||
|
|
||||||
|
def disconnect(self):
|
||||||
|
# Stop the controller loop first so it isn't fighting the shutdown ramp.
|
||||||
self._shutdown_event.set()
|
self._shutdown_event.set()
|
||||||
|
controller_stopped = True
|
||||||
|
if self._controller_thread is not None:
|
||||||
|
# Wait long enough for any in-flight inference tick to finish and the loop
|
||||||
|
# to observe the shutdown flag, so no stray low command is published while
|
||||||
|
# the ramp runs (the shutdown routine must be the single publisher).
|
||||||
|
self._controller_thread.join(timeout=5.0)
|
||||||
|
if self._controller_thread.is_alive():
|
||||||
|
controller_stopped = False
|
||||||
|
logger.error(
|
||||||
|
"Controller thread did not stop; skipping graceful ramp to avoid "
|
||||||
|
"concurrent low commands (fail-safe: joints keep last command until exit)"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Soft, damped settle instead of an instant limp (real robot only; the
|
||||||
|
# subscribe thread is still alive here to supply the current pose). Only ramp
|
||||||
|
# once the controller thread has definitely exited.
|
||||||
|
if not self.config.is_simulation and controller_stopped:
|
||||||
|
self._graceful_stop()
|
||||||
|
|
||||||
|
if self.controller is not None and hasattr(self.controller, "shutdown"):
|
||||||
|
self.controller.shutdown()
|
||||||
|
|
||||||
# Wait for subscribe thread to finish
|
# Wait for subscribe thread to finish
|
||||||
if self.subscribe_thread is not None:
|
if self.subscribe_thread is not None:
|
||||||
@@ -385,12 +715,6 @@ class UnitreeG1(Robot):
|
|||||||
if self.subscribe_thread.is_alive():
|
if self.subscribe_thread.is_alive():
|
||||||
logger.warning("Subscribe thread did not stop cleanly")
|
logger.warning("Subscribe thread did not stop cleanly")
|
||||||
|
|
||||||
# Wait for controller thread to finish
|
|
||||||
if self._controller_thread is not None:
|
|
||||||
self._controller_thread.join(timeout=2.0)
|
|
||||||
if self._controller_thread.is_alive():
|
|
||||||
logger.warning("Controller thread did not stop cleanly")
|
|
||||||
|
|
||||||
# Close simulation environment
|
# Close simulation environment
|
||||||
if self.config.is_simulation and self.sim_env is not None:
|
if self.config.is_simulation and self.sim_env is not None:
|
||||||
try:
|
try:
|
||||||
@@ -422,44 +746,41 @@ class UnitreeG1(Robot):
|
|||||||
if lowstate is None:
|
if lowstate is None:
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
obs = {}
|
# Motors + IMU + wireless remote (shared lowstate -> obs mapping)
|
||||||
|
obs = lowstate_to_obs(lowstate)
|
||||||
|
|
||||||
# Motors - q, dq, tau for all joints
|
# Dense whole-body controllers (OpenHLM / pi0.5): expose the 34-D proprio
|
||||||
for motor in G1_29_JointIndex:
|
# state as ``wb_state.{i}.pos`` so the rollout aggregates it into
|
||||||
name = motor.name
|
# ``observation.state`` for the policy.
|
||||||
idx = motor.value
|
if self.config.sonic_token_action:
|
||||||
obs[f"{name}.q"] = lowstate.motor_state[idx].q
|
# Token mode: echo the last commanded latent token as observation.state
|
||||||
obs[f"{name}.dq"] = lowstate.motor_state[idx].dq
|
# so a token-output VLA closes the loop on its own previous token.
|
||||||
obs[f"{name}.tau"] = lowstate.motor_state[idx].tau_est
|
from .controllers.sonic_whole_body import token_state_key
|
||||||
|
|
||||||
# IMU - gyroscope
|
token = self._last_token if self._last_token is not None else []
|
||||||
if lowstate.imu_state.gyroscope:
|
for i, v in enumerate(token):
|
||||||
obs["imu.gyro.x"] = lowstate.imu_state.gyroscope[0]
|
obs[token_state_key(i)] = float(v)
|
||||||
obs["imu.gyro.y"] = lowstate.imu_state.gyroscope[1]
|
elif getattr(self.controller, "wb_action", False):
|
||||||
obs["imu.gyro.z"] = lowstate.imu_state.gyroscope[2]
|
wb_state = obs_to_wb34_state(obs)
|
||||||
|
for i, v in enumerate(wb_state):
|
||||||
|
obs[f"wb_state.{i}.pos"] = float(v)
|
||||||
|
|
||||||
# IMU - accelerometer
|
# Synthetic empty cameras: black frames so image-conditioned policies run
|
||||||
if lowstate.imu_state.accelerometer:
|
# before real cameras are wired.
|
||||||
obs["imu.accel.x"] = lowstate.imu_state.accelerometer[0]
|
if self.config.empty_cameras:
|
||||||
obs["imu.accel.y"] = lowstate.imu_state.accelerometer[1]
|
h, w = self.config.empty_camera_hw
|
||||||
obs["imu.accel.z"] = lowstate.imu_state.accelerometer[2]
|
black = np.zeros((h, w, 3), dtype=np.uint8)
|
||||||
|
for name in self.config.empty_cameras:
|
||||||
|
obs[name] = black
|
||||||
|
|
||||||
# IMU - quaternion
|
# Replay cameras: serve the current recorded frame per camera, then advance.
|
||||||
if lowstate.imu_state.quaternion:
|
if self._replay_len:
|
||||||
obs["imu.quat.w"] = lowstate.imu_state.quaternion[0]
|
idx = self._replay_idx
|
||||||
obs["imu.quat.x"] = lowstate.imu_state.quaternion[1]
|
if idx >= self._replay_len:
|
||||||
obs["imu.quat.y"] = lowstate.imu_state.quaternion[2]
|
idx = self._replay_len - 1 if not self.config.replay_camera_loop else idx % self._replay_len
|
||||||
obs["imu.quat.z"] = lowstate.imu_state.quaternion[3]
|
for name in self._replay_raw:
|
||||||
|
obs[name] = self._replay_frame(name, idx)
|
||||||
# IMU - rpy
|
self._replay_idx += 1
|
||||||
if lowstate.imu_state.rpy:
|
|
||||||
obs["imu.rpy.roll"] = lowstate.imu_state.rpy[0]
|
|
||||||
obs["imu.rpy.pitch"] = lowstate.imu_state.rpy[1]
|
|
||||||
obs["imu.rpy.yaw"] = lowstate.imu_state.rpy[2]
|
|
||||||
|
|
||||||
# Wireless remote (raw bytes for teleoperator)
|
|
||||||
if lowstate.wireless_remote:
|
|
||||||
obs["wireless_remote"] = lowstate.wireless_remote
|
|
||||||
|
|
||||||
# Cameras - read images from ZMQ cameras
|
# Cameras - read images from ZMQ cameras
|
||||||
for cam_name, cam in self._cameras.items():
|
for cam_name, cam in self._cameras.items():
|
||||||
@@ -473,9 +794,19 @@ class UnitreeG1(Robot):
|
|||||||
def send_action(self, action: RobotAction) -> RobotAction:
|
def send_action(self, action: RobotAction) -> RobotAction:
|
||||||
action_to_publish = action
|
action_to_publish = action
|
||||||
if self.controller is not None:
|
if self.controller is not None:
|
||||||
|
if self.config.sonic_token_action:
|
||||||
|
from .controllers.sonic_whole_body import _extract_token_from_action
|
||||||
|
|
||||||
|
token = _extract_token_from_action(action)
|
||||||
|
if token is not None:
|
||||||
|
self._last_token = token
|
||||||
|
self._update_controller_action(action)
|
||||||
|
if self.config.publish_hands and getattr(self.controller, "wb_action", False):
|
||||||
|
self._publish_hand_cmds(action)
|
||||||
|
if getattr(self.controller, "full_body", False):
|
||||||
|
return action
|
||||||
# Controller thread owns legs/waist. Here we only update joystick inputs
|
# Controller thread owns legs/waist. Here we only update joystick inputs
|
||||||
# and publish arm targets from the teleoperator.
|
# and publish arm targets from the teleoperator.
|
||||||
self._update_controller_action(action)
|
|
||||||
arm_prefixes = tuple(j.name for j in G1_29_JointArmIndex)
|
arm_prefixes = tuple(j.name for j in G1_29_JointArmIndex)
|
||||||
action_to_publish = {
|
action_to_publish = {
|
||||||
key: value
|
key: value
|
||||||
@@ -503,11 +834,67 @@ class UnitreeG1(Robot):
|
|||||||
return action
|
return action
|
||||||
|
|
||||||
def _update_controller_action(self, action: RobotAction) -> None:
|
def _update_controller_action(self, action: RobotAction) -> None:
|
||||||
"""Update controller input state from incoming teleop action."""
|
"""Update controller input state from an incoming teleop action.
|
||||||
|
|
||||||
|
Controller-agnostic: every value-carrying key is forwarded verbatim into
|
||||||
|
``controller_input`` (whole-body ``wb.{i}.pos`` from a 34-D VLA, or whatever a
|
||||||
|
future controller expects), and each controller extracts only the keys it
|
||||||
|
understands. The robot deliberately does not enumerate any controller's key
|
||||||
|
schema here.
|
||||||
|
|
||||||
|
KeyboardTeleop is the one special case: it emits the currently-pressed keys as
|
||||||
|
bare action keys with a ``None`` value (``dict.fromkeys(pressed, None)``), so
|
||||||
|
those are collected into a single held-key set under ``KEYBOARD_KEYS_FIELD``,
|
||||||
|
rebuilt each tick so releases clear. Special keys arrive as pynput objects and
|
||||||
|
are normalised to their name ("space", ...).
|
||||||
|
"""
|
||||||
with self._controller_action_lock:
|
with self._controller_action_lock:
|
||||||
for key in REMOTE_KEYS:
|
self.controller_input[KEYBOARD_KEYS_FIELD] = {
|
||||||
if key in action:
|
(k if isinstance(k, str) else getattr(k, "name", str(k)))
|
||||||
self.controller_input[key] = action[key]
|
for k, value in action.items()
|
||||||
|
if value is None
|
||||||
|
}
|
||||||
|
for key, value in action.items():
|
||||||
|
if isinstance(key, str) and value is not None:
|
||||||
|
self.controller_input[key] = value
|
||||||
|
|
||||||
|
def _publish_hand_cmds(self, action: RobotAction) -> None:
|
||||||
|
"""Drive the Dex3 hands from the OpenHLM grip scalars in a 34-D wb action.
|
||||||
|
|
||||||
|
``wb.7.pos`` is the left grip and ``wb.15.pos`` the right grip. Each scalar in
|
||||||
|
[0, 1] (``hand_open_grip_value`` == fully open) is turned into a curl amount and
|
||||||
|
scaled onto ``hand_closed_pose`` (7 joints), then published as a PD target on
|
||||||
|
``rt/dex3/{left,right}/cmd`` so the fingers close when the policy grips.
|
||||||
|
"""
|
||||||
|
if not self._hand_publishers:
|
||||||
|
return
|
||||||
|
from .g1_utils import wb_action_key
|
||||||
|
|
||||||
|
open_val = float(self.config.hand_open_grip_value)
|
||||||
|
closed_val = float(self.config.hand_closed_grip_value)
|
||||||
|
closed_pose = self.config.hand_closed_pose
|
||||||
|
kp, kd = float(self.config.hand_kp), float(self.config.hand_kd)
|
||||||
|
span = (closed_val - open_val) or 1.0
|
||||||
|
|
||||||
|
def curl_amount(grip: float) -> float:
|
||||||
|
# Fraction of the way from the open scalar to the closed scalar, in [0, 1].
|
||||||
|
return float(min(max((grip - open_val) / span, 0.0), 1.0))
|
||||||
|
|
||||||
|
for side, grip_idx, cmd in (
|
||||||
|
("left", 7, self._left_hand_cmd),
|
||||||
|
("right", 15, self._right_hand_cmd),
|
||||||
|
):
|
||||||
|
grip = action.get(wb_action_key(grip_idx))
|
||||||
|
if grip is None:
|
||||||
|
continue
|
||||||
|
amount = curl_amount(float(grip))
|
||||||
|
for i, closed_q in enumerate(closed_pose):
|
||||||
|
cmd.motor_cmd[i].q = float(closed_q) * amount
|
||||||
|
cmd.motor_cmd[i].dq = 0.0
|
||||||
|
cmd.motor_cmd[i].kp = kp
|
||||||
|
cmd.motor_cmd[i].kd = kd
|
||||||
|
cmd.motor_cmd[i].tau = 0.0
|
||||||
|
self._hand_publishers[side].Write(cmd)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def is_calibrated(self) -> bool:
|
def is_calibrated(self) -> bool:
|
||||||
@@ -537,43 +924,64 @@ class UnitreeG1(Robot):
|
|||||||
if default_positions is None:
|
if default_positions is None:
|
||||||
default_positions = np.array(self.config.default_positions, dtype=np.float32)
|
default_positions = np.array(self.config.default_positions, dtype=np.float32)
|
||||||
|
|
||||||
if self.config.is_simulation and self.sim_env is not None:
|
# Full-body controllers (SONIC / OpenHLM) own the whole 29-DoF command and
|
||||||
self.sim_env.reset()
|
# ignore ``<joint>.q`` in send_action(), so reset() must publish the default
|
||||||
self.publish_lowcmd(
|
# pose directly. Pause the background controller first so the two aren't both
|
||||||
{f"{motor.name}.q": float(default_positions[motor.value]) for motor in G1_29_JointIndex}
|
# writing low commands while the robot moves to the default pose.
|
||||||
)
|
full_body = getattr(self.controller, "full_body", False)
|
||||||
else:
|
paused = False
|
||||||
total_time = 3.0
|
if full_body and self._controller_thread is not None:
|
||||||
num_steps = int(total_time / control_dt)
|
self._controller_paused.set()
|
||||||
|
paused = True
|
||||||
|
time.sleep(control_dt) # let any in-flight controller tick settle
|
||||||
|
|
||||||
# get current state
|
try:
|
||||||
obs = self.get_observation()
|
if self.config.is_simulation and self.sim_env is not None:
|
||||||
|
self.sim_env.reset()
|
||||||
|
self.publish_lowcmd(
|
||||||
|
{f"{motor.name}.q": float(default_positions[motor.value]) for motor in G1_29_JointIndex}
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
total_time = 3.0
|
||||||
|
num_steps = int(total_time / control_dt)
|
||||||
|
|
||||||
# record current positions
|
# get current state
|
||||||
init_dof_pos = np.zeros(29, dtype=np.float32)
|
obs = self.get_observation()
|
||||||
for motor in G1_29_JointIndex:
|
|
||||||
init_dof_pos[motor.value] = obs[f"{motor.name}.q"]
|
|
||||||
|
|
||||||
# Interpolate to default position
|
# record current positions
|
||||||
for step in range(num_steps):
|
init_dof_pos = np.zeros(29, dtype=np.float32)
|
||||||
start_time = time.time()
|
|
||||||
|
|
||||||
alpha = step / num_steps
|
|
||||||
action_dict = {}
|
|
||||||
for motor in G1_29_JointIndex:
|
for motor in G1_29_JointIndex:
|
||||||
target_pos = default_positions[motor.value]
|
init_dof_pos[motor.value] = obs[f"{motor.name}.q"]
|
||||||
interp_pos = init_dof_pos[motor.value] * (1 - alpha) + target_pos * alpha
|
|
||||||
action_dict[f"{motor.name}.q"] = float(interp_pos)
|
|
||||||
|
|
||||||
self.send_action(action_dict)
|
# Interpolate to default position
|
||||||
|
for step in range(num_steps):
|
||||||
|
start_time = time.time()
|
||||||
|
|
||||||
# Maintain constant control rate
|
alpha = step / num_steps
|
||||||
elapsed = time.time() - start_time
|
action_dict = {}
|
||||||
sleep_time = max(0, control_dt - elapsed)
|
for motor in G1_29_JointIndex:
|
||||||
time.sleep(sleep_time)
|
target_pos = default_positions[motor.value]
|
||||||
|
interp_pos = init_dof_pos[motor.value] * (1 - alpha) + target_pos * alpha
|
||||||
|
action_dict[f"{motor.name}.q"] = float(interp_pos)
|
||||||
|
|
||||||
# Reset controller internal state (gait phase, obs history, etc.)
|
# Full-body controllers no-op in send_action(); publish the pose
|
||||||
if self.controller is not None and hasattr(self.controller, "reset"):
|
# directly (arm-only controllers keep the send_action() path).
|
||||||
self.controller.reset()
|
if full_body:
|
||||||
|
self.publish_lowcmd(action_dict)
|
||||||
|
else:
|
||||||
|
self.send_action(action_dict)
|
||||||
|
|
||||||
|
# Maintain constant control rate
|
||||||
|
elapsed = time.time() - start_time
|
||||||
|
sleep_time = max(0, control_dt - elapsed)
|
||||||
|
time.sleep(sleep_time)
|
||||||
|
|
||||||
|
# Reset controller internal state (gait phase, obs history, etc.) before
|
||||||
|
# resuming so its buffers reflect the post-reset pose.
|
||||||
|
if self.controller is not None and hasattr(self.controller, "reset"):
|
||||||
|
self.controller.reset()
|
||||||
|
finally:
|
||||||
|
if paused:
|
||||||
|
self._controller_paused.clear()
|
||||||
|
|
||||||
logger.info("Reached default position")
|
logger.info("Reached default position")
|
||||||
|
|||||||
@@ -60,8 +60,18 @@ def is_package_available(
|
|||||||
# If the package can't be imported, it's not available
|
# If the package can't be imported, it's not available
|
||||||
package_exists = False
|
package_exists = False
|
||||||
else:
|
else:
|
||||||
# For packages other than "torch", don't attempt the fallback and set as not available
|
# The distribution may be published under a name that differs from the
|
||||||
package_exists = False
|
# import name (e.g. ``onnxruntime`` imports from ``onnxruntime-gpu`` /
|
||||||
|
# ``onnxruntime-silicon``). Resolve the import name to its actual
|
||||||
|
# distribution(s) and read the version from there before giving up.
|
||||||
|
try:
|
||||||
|
dists = importlib.metadata.packages_distributions().get(import_name, [])
|
||||||
|
if dists:
|
||||||
|
package_version = importlib.metadata.version(dists[0])
|
||||||
|
else:
|
||||||
|
package_exists = False
|
||||||
|
except importlib.metadata.PackageNotFoundError:
|
||||||
|
package_exists = False
|
||||||
logging.debug(f"Detected {pkg_name} version: {package_version}")
|
logging.debug(f"Detected {pkg_name} version: {package_version}")
|
||||||
if return_version:
|
if return_version:
|
||||||
return package_exists, package_version
|
return package_exists, package_version
|
||||||
@@ -123,6 +133,8 @@ _pyrealsense2_available = is_package_available("pyrealsense2") or is_package_ava
|
|||||||
"pyrealsense2-macosx", import_name="pyrealsense2"
|
"pyrealsense2-macosx", import_name="pyrealsense2"
|
||||||
)
|
)
|
||||||
_zmq_available = is_package_available("pyzmq", import_name="zmq")
|
_zmq_available = is_package_available("pyzmq", import_name="zmq")
|
||||||
|
_onnxruntime_available = is_package_available("onnxruntime")
|
||||||
|
_onnx_available = is_package_available("onnx")
|
||||||
_hebi_available = is_package_available("hebi-py", import_name="hebi")
|
_hebi_available = is_package_available("hebi-py", import_name="hebi")
|
||||||
_teleop_available = is_package_available("teleop")
|
_teleop_available = is_package_available("teleop")
|
||||||
_placo_available = is_package_available("placo")
|
_placo_available = is_package_available("placo")
|
||||||
|
|||||||
Reference in New Issue
Block a user