mirror of
https://github.com/huggingface/lerobot.git
synced 2026-07-24 18:26:11 +00:00
feat(policy): adding return_extra to policy contracts
This commit is contained in:
@@ -40,6 +40,7 @@ T = TypeVar("T", bound="PreTrainedPolicy")
|
|||||||
|
|
||||||
class ActionSelectKwargs(TypedDict, total=False):
|
class ActionSelectKwargs(TypedDict, total=False):
|
||||||
noise: Tensor | None
|
noise: Tensor | None
|
||||||
|
return_extra: bool
|
||||||
|
|
||||||
|
|
||||||
class PreTrainedPolicy(nn.Module, HubMixin, abc.ABC):
|
class PreTrainedPolicy(nn.Module, HubMixin, abc.ABC):
|
||||||
@@ -187,20 +188,32 @@ class PreTrainedPolicy(nn.Module, HubMixin, abc.ABC):
|
|||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def predict_action_chunk(self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]) -> Tensor:
|
def predict_action_chunk(
|
||||||
|
self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]
|
||||||
|
) -> Tensor | tuple[Tensor, dict[str, Tensor]]:
|
||||||
"""Returns the action chunk (for action chunking policies) for a given observation, potentially in batch mode.
|
"""Returns the action chunk (for action chunking policies) for a given observation, potentially in batch mode.
|
||||||
|
|
||||||
Child classes using action chunking should use this method within `select_action` to form the action chunk
|
Child classes using action chunking should use this method within `select_action` to form the action chunk
|
||||||
cached for selection.
|
cached for selection.
|
||||||
|
|
||||||
|
By default returns just the action `Tensor`. If `return_extra=True`, returns `(action, extra)`
|
||||||
|
where `extra` is a (possibly empty) `dict[str, Tensor]` of auxiliary outputs a policy may
|
||||||
|
expose (e.g. world-model predictions). Policies that produce nothing extra may ignore the kwarg.
|
||||||
"""
|
"""
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def select_action(self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]) -> Tensor:
|
def select_action(
|
||||||
|
self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]
|
||||||
|
) -> Tensor | tuple[Tensor, dict[str, Tensor]]:
|
||||||
"""Return one action to run in the environment (potentially in batch mode).
|
"""Return one action to run in the environment (potentially in batch mode).
|
||||||
|
|
||||||
When the model uses a history of observations, or outputs a sequence of actions, this method deals
|
When the model uses a history of observations, or outputs a sequence of actions, this method deals
|
||||||
with caching.
|
with caching.
|
||||||
|
|
||||||
|
By default returns just the action `Tensor`. If `return_extra=True`, returns `(action, extra)`
|
||||||
|
where `extra` is a (possibly empty) `dict[str, Tensor]` of auxiliary outputs a policy may
|
||||||
|
expose (e.g. world-model predictions). Policies that produce nothing extra may ignore the kwarg.
|
||||||
"""
|
"""
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user