Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion docs/docs/docs/getting_started/policy_support.md
Original file line number Diff line number Diff line change
Expand Up @@ -58,7 +58,7 @@ python scripts/evaluation/policy_inference.py \
- `--eval_rounts`: Number of evaluation rounds. 0 means don't add time out termination, policy will run until success or manual reset (default: `0`)

- `--policy_type`: Type of policy to use (default: `gr00tn1.5`).
- now we support `gr00tn1.5`, `lerobot-<model_type>`
- now we support `gr00tn1.5`, `gr00tn1.6`, `lerobot-<model_type>`

- `--policy_host`: Host address of the policy server (default: `localhost`).

Expand Down
52 changes: 47 additions & 5 deletions docs/docs/resources/available_policy.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,17 +4,18 @@ This page lists the policy inference methods currently supported by LeIsaac.

Depending on your use case, you may need to install additional dependencies to enable inference:

```shell
pip install -e "source/leisaac[gr00t]"
pip install -e "source/leisaac[lerobot-async]"
pip install -e "source/leisaac[openpi]"
```
:::tip
For each supported policy, we have specified the verified commit. If the corresponding repository is updated, it may cause compatibility issues. If you encounter such cases, feel free to open an issue.
:::

## Finetuned gr00t n1.5

Install additional dependencies:

```shell
pip install -e "source/leisaac[gr00t]"
```

Inference Scripts:

```shell
Expand All @@ -35,10 +36,45 @@ python scripts/evaluation/policy_inference.py \
target commit: https://github.com/NVIDIA/Isaac-GR00T/commit/4af2b622892f7dcb5aae5a3fb70bcb02dc217b96
:::

## Finetuned gr00t n1.6

Install additional dependencies:

```shell
pip install -e "source/leisaac[gr00t]"
```

Inference Scripts:

```shell
python scripts/evaluation/policy_inference.py \
--task=LeIsaac-SO101-PickOrange-v0 \
--eval_rounds=10 \
--policy_type=gr00tn1.6 \
--policy_host=localhost \
--policy_port=5555 \
--policy_timeout_ms=5000 \
--policy_action_horizon=16 \
--policy_language_instruction="Pick up the orange and place it on the plate" \
--device=cuda \
--enable_cameras
```

:::tip
target commit: https://github.com/NVIDIA/Isaac-GR00T/commit/e8e625f4f21898c506a1d8f7d20a289c97a52acf
:::

## Lerobot official policy

Install additional dependencies:

```shell
pip install -e "source/leisaac[lerobot-async]"
```

We utilize lerobot's async inference capabilities for policy execution. For detailed information, please refer to the [official documentation](https://huggingface.co/docs/lerobot/async). Prior to execution, ensure that the policy server is running.


```shell
python scripts/evaluation/policy_inference.py \
--task=LeIsaac-SO101-PickOrange-v0 \
Expand All @@ -59,6 +95,12 @@ target commit: https://github.com/huggingface/lerobot/tree/v0.3.3

## Finetuned openpi

Install additional dependencies:

```shell
pip install -e "source/leisaac[openpi]"
```

We utilize openpi's remote inference capabilities for policy execution. For detailed information, please refer to the [official documentation](https://github.com/Physical-Intelligence/openpi/blob/main/docs/remote_inference.md). Prior to execution, ensure that the policy server is running.

```shell
Expand Down
21 changes: 19 additions & 2 deletions scripts/evaluation/policy_inference.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,7 @@
"--policy_type",
type=str,
default="gr00tn1.5",
help="Type of policy to use. support gr00tn1.5, lerobot-<model_type>, openpi",
help="Type of policy to use. support gr00tn1.5, gr00tn1.6, lerobot-<model_type>, openpi",
)
parser.add_argument("--policy_host", type=str, default="localhost", help="Host of the policy server.")
parser.add_argument("--policy_port", type=int, default=5555, help="Port of the policy server.")
Expand Down Expand Up @@ -123,7 +123,7 @@ def _on_keyboard_event(self, event, *args, **kwargs):

def preprocess_obs_dict(obs_dict: dict, model_type: str, language_instruction: str):
"""Preprocess the observation dictionary to the format expected by the policy."""
if model_type in ["gr00tn1.5", "lerobot", "openpi"]:
if model_type in ["gr00tn1.5", "gr00tn1.6", "lerobot", "openpi"]:
obs_dict["task_description"] = language_instruction
return obs_dict
else:
Expand Down Expand Up @@ -167,6 +167,23 @@ def main():
camera_keys=[key for key, sensor in env.scene.sensors.items() if isinstance(sensor, Camera)],
modality_keys=modality_keys,
)
elif args_cli.policy_type == "gr00tn1.6":
from isaaclab.sensors import Camera
from leisaac.policy import Gr00t16ServicePolicyClient

if task_type == "so101leader":
modality_keys = ["single_arm", "gripper"]
else:
raise ValueError(f"Task type {task_type} not supported when using GR00T N1.5 policy yet.")

policy = Gr00t16ServicePolicyClient(
host=args_cli.policy_host,
port=args_cli.policy_port,
timeout_ms=args_cli.policy_timeout_ms,
camera_keys=[key for key, sensor in env.scene.sensors.items() if isinstance(sensor, Camera)],
modality_keys=modality_keys,
)

elif "lerobot" in args_cli.policy_type:
from isaaclab.sensors import Camera
from leisaac.policy import LeRobotServicePolicyClient
Expand Down
96 changes: 96 additions & 0 deletions source/leisaac/leisaac/policy/service_policy_clients.py
Original file line number Diff line number Diff line change
Expand Up @@ -84,6 +84,102 @@ def get_action(self, observation_dict: dict) -> torch.Tensor:
return torch.from_numpy(concat_action[:, None, :])


class Gr00t16ServicePolicyClient(ZMQServicePolicy):
"""
Service policy client for GR00T N1.6: https://github.com/NVIDIA/Isaac-GR00T
Target commit: https://github.com/NVIDIA/Isaac-GR00T/commit/e8e625f4f21898c506a1d8f7d20a289c97a52acf
"""

def __init__(
self,
host: str = "localhost",
port: int = 5555,
timeout_ms: int = 5000,
camera_keys: list[str] = ["front", "wrist"],
modality_keys: list[str] = ["single_arm", "gripper"],
):
"""
Args:
host: Host of the policy server.
port: Port of the policy server.
camera_keys: Keys of the cameras.
timeout_ms: Timeout of the policy server.
modality_keys: Keys of the modality.
"""
super().__init__(host=host, port=port, timeout_ms=timeout_ms, ping_endpoint="ping")
self.camera_keys = camera_keys
self.modality_keys = modality_keys

def get_action(self, observation_dict: dict) -> torch.Tensor:
# Build the 'video' dictionary: {camera_name: (B, T, H, W, 3), dtype=uint8}
video = {
camera_key: np.expand_dims(observation_dict[camera_key].cpu().numpy().astype(np.uint8), axis=0)
for camera_key in self.camera_keys
}

# Build the 'state' dictionary (single_arm, gripper)
state = {}
if "single_arm" in self.modality_keys:
joint_pos = convert_leisaac_action_to_lerobot(observation_dict["joint_pos"])
# Add a new axis at the front (batch dim)
joint_pos = np.expand_dims(joint_pos, axis=0)
# Ensure joint_pos shape is (B, T, 6), we need (B, T, D) for each stream
# e.g., single_arm: first 5 dims, gripper: last dim
state["single_arm"] = joint_pos[..., 0:5].astype(np.float32)
state["gripper"] = joint_pos[..., 5:6].astype(np.float32)
# TODO: add bi-arm support

# Build the 'language' dictionary
language = {
"annotation.human.task_description": [[observation_dict["task_description"]]],
}

# Compose the final observation dictionary as required
obs_dict = {
"video": video,
"state": state,
"language": language,
}

"""
Example of obs_dict for single arm task:
obs_dict = {
"video": {
"front": np.zeros((1, 1, 480, 640, 3), dtype=np.uint8),
"wrist": np.zeros((1, 1, 480, 640, 3), dtype=np.uint8),
},
"state": {
"single_arm": np.zeros((1, 1, 5)),
"gripper": np.zeros((1, 1, 1)),
},
"language": {
"task": [["pick and place"]],
}
}
"""
obs_dict = {"observation": obs_dict}
# get the action chunk via the policy server
action_chunk = self.call_endpoint("get_action", obs_dict)

"""
Example of action_chunk for single arm task:
action_chunk = [{
"single_arm": np.zeros((1, 16, 5)),
"gripper": np.zeros((1, 16, 1)),
}]
"""
action_chunk = action_chunk[0]
concat_action = np.concatenate(
[action_chunk["single_arm"], action_chunk["gripper"]],
axis=-1,
)
# squeeze the first dimension
concat_action = concat_action.squeeze(0)
concat_action = convert_lerobot_action_to_leisaac(concat_action)

return torch.from_numpy(concat_action[:, None, :])


class LeRobotServicePolicyClient(Policy):
"""
Service policy client for Lerobot: https://github.com/huggingface/lerobot
Expand Down