diff --git a/.agents/skills/review-pr/SKILL.md b/.agents/skills/review-pr/SKILL.md index 9e24e60ca..049d9fce4 100644 --- a/.agents/skills/review-pr/SKILL.md +++ b/.agents/skills/review-pr/SKILL.md @@ -74,6 +74,56 @@ impact are established. file or subsystem coverage ledger and review dependency foundations before their consumers. +## Explicitly requested PR publication + +Keep review read-only unless the user explicitly asks to publish the findings +to a remote PR. A request to review, audit, or approve a change alone does not +authorize remote comments or a review event. + +When publication is explicitly requested: + +1. Verify the repository, PR number, base branch, head commit, and changed + files before posting. If the target is not unambiguous, ask for the PR + identifier rather than guessing. +2. Write every comment and review summary posted to the remote PR in English. + The conversational review may use the user's language, but translate the + finding faithfully before publication. Preserve code identifiers, paths, + and short log excerpts when they are needed as evidence. +3. Publish each actionable finding as an inline comment on the smallest + relevant changed line when the hosting integration supports it. Put + findings that cannot be anchored to a changed line in a top-level review + summary. Preserve the finding priority and evidence in the posted text. +4. Use a normal comment review by default. Submit `request changes` or + `approve` only when the user explicitly requests that exact review event; + never infer approval authority from a review request. +5. Use an available GitHub connector or authenticated `gh` workflow. Do not + claim that comments were posted if the integration is unavailable or a + request fails. If publication is unavailable, return the complete, + copy-ready Markdown review instead. +6. Report the published comment or review URLs, any findings that were not + posted, and any partial failures. Avoid duplicate comments when the same + finding has already been posted for the same head commit. + +Do not publish speculative open questions or residual risks as defects. They +may be included in the top-level summary only when clearly labeled as such. + +### Trigger phrases + +The skill may be selected for review requests containing terms such as +`review`, `audit`, `inspect`, `assess`, or `approve`. These terms authorize +analysis only; they do not authorize remote PR changes. + +Publication requires an explicit remote-posting request, for example: + +- “publish/post/submit the review findings to the PR” +- “add inline comments to the PR” +- “leave the review comments on GitHub” +- “把评审意见提交到 PR” or “在 PR 上添加评审评论” + +Phrases such as “review this PR”, “approve this PR”, or “给我评审建议” do not +by themselves trigger publication. In particular, “approve this PR” selects +the review task but does not authorize an `approve` review event. + ## 1. Resolve the review target Read the applicable `AGENTS.md` instructions first. Determine the exact delta diff --git a/docs/source/api_reference/public_api.rst b/docs/source/api_reference/public_api.rst index 9c68364ee..55f4740ef 100644 --- a/docs/source/api_reference/public_api.rst +++ b/docs/source/api_reference/public_api.rst @@ -341,6 +341,7 @@ embodichain.lab.gym.envs.managers.actions QposTerm QposDenormalizedTerm QposNormalizedTerm + EefPoseGripperTerm EefPoseTerm QvelTerm QfTerm @@ -2190,6 +2191,15 @@ embodichain_tasks.configs get_config_path +embodichain_tasks.manipulation.repeated_pick_place +------------------------------------------------ + +.. currentmodule:: embodichain_tasks.manipulation.repeated_pick_place + +.. autosummary:: + + RepeatedPickPlaceRlinfEnv + embodichain_tasks.manipulation.push_cube ---------------------------------------- diff --git a/embodichain/lab/gym/envs/managers/actions.py b/embodichain/lab/gym/envs/managers/actions.py index a60133c41..09ac0ebab 100644 --- a/embodichain/lab/gym/envs/managers/actions.py +++ b/embodichain/lab/gym/envs/managers/actions.py @@ -42,6 +42,7 @@ from embodichain.lab.sim.types import EnvAction from embodichain.utils.math import matrix_from_euler, matrix_from_quat + from .action_manager import ActionTerm from .cfg import ActionTermCfg @@ -54,12 +55,13 @@ __all__ = [ "DefaultJointPositionTerm", "DeltaQposTerm", - "QposTerm", + "EefPoseGripperTerm", + "EefPoseTerm", + "QfTerm", "QposDenormalizedTerm", "QposNormalizedTerm", - "EefPoseTerm", + "QposTerm", "QvelTerm", - "QfTerm", ] @@ -259,6 +261,7 @@ def __init__(self, cfg: ActionTermCfg, env: EmbodiedEnv): super().__init__(cfg, env) self._scale = cfg.params.get("scale", 1.0) self._pose_dim = cfg.params.get("pose_dim", 7) # 6 for euler, 7 for quat + self._part_name = cfg.params.get("part_name") @property def input_key(self) -> str: @@ -286,14 +289,30 @@ def process_action(self, action: torch.Tensor) -> EnvAction: f"EEF pose action must be 6D or 7D, got {scaled.shape[-1]}D" ) # Batch IK: robot.compute_ik supports (num_envs, 4, 4) pose and (num_envs, dof) seed + joint_seed = current_qpos + part_joint_ids = None + if self._part_name is not None: + part_joint_ids = self._env.robot.get_joint_ids( + self._part_name, remove_mimic=True + ) + joint_seed = current_qpos[:, part_joint_ids] ret, qpos_ik = self._env.robot.compute_ik( pose=target_pose, - joint_seed=current_qpos, - ) - # Fallback to current_qpos where IK failed - result_qpos = torch.where( - ret.unsqueeze(-1).expand_as(qpos_ik), qpos_ik, current_qpos + joint_seed=joint_seed, + name=self._part_name, ) + if part_joint_ids is None or qpos_ik.shape[-1] == current_qpos.shape[-1]: + result_qpos = torch.where( + ret.unsqueeze(-1).expand_as(qpos_ik), qpos_ik, current_qpos + ) + else: + result_qpos = current_qpos.clone() + selected_qpos = torch.where( + ret.unsqueeze(-1).expand_as(qpos_ik), + qpos_ik, + current_qpos[:, part_joint_ids], + ) + result_qpos[:, part_joint_ids] = selected_qpos return TensorDict( {"qpos": result_qpos, "ik_success": ret}, batch_size=[batch_size], @@ -301,6 +320,37 @@ def process_action(self, action: torch.Tensor) -> EnvAction: ) +class EefPoseGripperTerm(EefPoseTerm): + """Convert a 7D Cartesian action into arm IK targets and gripper joints. + + The action is ``[x, y, z, roll, pitch, yaw, gripper]``. The first six + values use the same absolute-pose convention as :class:`EefPoseTerm`; the + final value is normalized from ``[-1, 1]`` to the active hand joint limits. + """ + + @property + def action_dim(self) -> int: + return 7 + + def process_action(self, action: torch.Tensor) -> EnvAction: + if action.shape[-1] != 7: + raise ValueError( + f"EefPoseGripperTerm expects 7D actions, got {action.shape[-1]}D." + ) + + arm_action = action[..., :6] + result = super().process_action(arm_action) + hand_joint_ids = self._env.robot.get_joint_ids("hand", remove_mimic=True) + if not hand_joint_ids: + raise ValueError("EefPoseGripperTerm requires a robot hand control part.") + + limits = self._env.robot.body_data.qpos_limits[0, hand_joint_ids] + gripper = action[..., 6:7].clamp(-1.0, 1.0) + hand_qpos = limits[:, 0] + (gripper + 1.0) * 0.5 * (limits[:, 1] - limits[:, 0]) + result["qpos"][:, hand_joint_ids] = hand_qpos.expand(-1, len(hand_joint_ids)) + return result + + class QvelTerm(ActionTerm): """Joint velocity action: scale * action -> qvel. diff --git a/embodichain_tasks/configs/components/embodiments/franka_panda_vla.yaml b/embodichain_tasks/configs/components/embodiments/franka_panda_vla.yaml new file mode 100644 index 000000000..113e65769 --- /dev/null +++ b/embodichain_tasks/configs/components/embodiments/franka_panda_vla.yaml @@ -0,0 +1,45 @@ +embodiment_id: franka_panda_vla + +simulation: + class_type: FrankaPanda + robot_type: panda + uid: FrankaPanda + solver_cfg: + arm: + class_type: PytorchSolver + end_link_name: fr3_hand_tcp + root_link_name: base + tcp: + - [1.0, 0.0, 0.0, 0.0] + - [0.0, 1.0, 0.0, 0.0] + - [0.0, 0.0, 1.0, 0.0] + - [0.0, 0.0, 0.0, 1.0] + num_samples: 30 + link_attrs: + newton_gripper_contacts: + link_names_expr: ["fr3_leftfinger|fr3_rightfinger"] + attrs: + collision_props: + backend: newton + condim: 4 + material_props: + backend: newton + ke: 40000.0 + kd: 400.0 + torsional_friction: 0.1 + rolling_friction: 0.01 + +sensor: + - sensor_type: Camera + uid: cam_high + width: 640 + height: 480 + intrinsics: [488.1665, 488.1665, 320.0, 240.0] + extrinsics: + eye: [-1.2, -1.2, 1.25] + target: [-0.1, 0.15, 0.20] + up: [0.0, 0.0, 1.0] + +skill_profile: + contract_id: single_arm_parallel_gripper + profile_id: franka_panda diff --git a/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/env.rlinf.yaml b/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/env.rlinf.yaml new file mode 100644 index 000000000..427377892 --- /dev/null +++ b/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/env.rlinf.yaml @@ -0,0 +1,54 @@ +environment_id: repeated_pick_place_rlinf +physics: default +max_episodes: 1 +max_episode_steps: 240 +num_envs: 1 +arena_space: 2.5 + +simulation: + light: + direct: + - uid: main_light + light_type: sun + color: [0.6, 0.6, 0.6] + intensity: 5.0 + direction: [0.0, 0.0, -1.0] + background: + - uid: target + shape: + shape_type: Cube + size: [0.11, 0.11, 0.01] + body_type: kinematic + init_pos: [-0.40, 0.48, 0.005] + attrs: + material_props: {static_friction: 0.0, dynamic_friction: 0.0} + rigid_object: + - uid: cube + shape: + shape_type: Cube + size: [0.05, 0.05, 0.05] + body_type: dynamic + init_pos: [-0.42, -0.08, 0.025] + attrs: + mass_props: {mass: 0.05} + rigid_props: {linear_damping: 0.2, angular_damping: 0.2} + material_props: {dynamic_friction: 0.97, static_friction: 0.99} + +env: + sim_steps_per_control: 4 + observations: + eef_pose: + func: get_robot_eef_pose + mode: add + name: robot/eef_pose + params: {part_name: arm, position_only: false} + actions: + eef_pose_gripper: + func: EefPoseGripperTerm + params: {scale: 1.0, pose_dim: 6, part_name: arm} + rewards: + success_bonus: + func: success_reward + mode: add + weight: 10.0 + params: {} diff --git a/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/task.franka.rlinf.yaml b/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/task.franka.rlinf.yaml new file mode 100644 index 000000000..e88aac2ac --- /dev/null +++ b/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/task.franka.rlinf.yaml @@ -0,0 +1,7 @@ +id: RepeatedPickPlaceRlinf-Franka-v1 + +environment: + component: env.rlinf.yaml + +embodiment: + component: ../../../components/embodiments/franka_panda_vla.yaml diff --git a/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/task.franka.rlinf_expert.yaml b/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/task.franka.rlinf_expert.yaml new file mode 100644 index 000000000..545f07c64 --- /dev/null +++ b/embodichain_tasks/configs/tasks/manipulation/repeated_pick_place/task.franka.rlinf_expert.yaml @@ -0,0 +1,14 @@ +id: TaskProgramRepeatedPickPlaceRlinfExpert-Franka-v1 + +environment: + component: env.yaml + +task_program: + program: task_program/program.yaml + integration: task_program/integration.yaml + execution_policy: ../../../components/execution_policies/trajectory_open_loop.yaml + +embodiment: + component: ../../../components/embodiments/franka_panda_vla.yaml + overrides: + init_rot: [0.0, 0.0, 180.0] diff --git a/embodichain_tasks/embodichain_tasks/manipulation/repeated_pick_place.py b/embodichain_tasks/embodichain_tasks/manipulation/repeated_pick_place.py new file mode 100644 index 000000000..3c2f5d570 --- /dev/null +++ b/embodichain_tasks/embodichain_tasks/manipulation/repeated_pick_place.py @@ -0,0 +1,90 @@ +# ---------------------------------------------------------------------------- +# Copyright (c) 2021-2026 DexForce Technology Co., Ltd. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ---------------------------------------------------------------------------- + +"""RLinf-facing repeated pick-and-place environment.""" + +from __future__ import annotations + +from typing import Any + +import torch + +from embodichain.lab.gym.envs import EmbodiedEnv, EmbodiedEnvCfg +from embodichain.lab.gym.utils.registration import register_env +from embodichain.lab.sim.types import EnvObs + +__all__ = ["RepeatedPickPlaceRlinfEnv"] + + +@register_env( + "RepeatedPickPlaceRlinf-Franka-v1", + max_episode_steps=240, + override=True, + supports_rl=True, +) +class RepeatedPickPlaceRlinfEnv(EmbodiedEnv): + """Single-cube pick-and-place task with a flat VLA action interface.""" + + def __init__(self, cfg: EmbodiedEnvCfg | None = None, **kwargs: Any) -> None: + super().__init__(cfg or EmbodiedEnvCfg(), **kwargs) + self._success_hold_steps = torch.zeros( + self.num_envs, device=self.device, dtype=torch.int32 + ) + + def reset(self, *args: Any, **kwargs: Any) -> tuple[EnvObs, dict[str, Any]]: + observation, info = super().reset(*args, **kwargs) + options = kwargs.get("options") + reset_ids = None if options is None else options.get("reset_ids") + if reset_ids is None: + self._success_hold_steps.zero_() + else: + reset_ids = torch.as_tensor(reset_ids, dtype=torch.long, device=self.device) + self._success_hold_steps[reset_ids] = 0 + return observation, info + + def compute_task_state( + self, **kwargs: Any + ) -> tuple[torch.Tensor, torch.Tensor, dict[str, Any]]: + cube = self.sim.get_rigid_object("cube") + target = self.sim.get_rigid_object("target") + cube_pos = cube.get_local_pose(to_matrix=True)[:, :3, 3] + target_pos = target.get_local_pose(to_matrix=True)[:, :3, 3] + distance = torch.linalg.vector_norm(cube_pos[:, :2] - target_pos[:, :2], dim=1) + height_error = torch.abs(cube_pos[:, 2] - target_pos[:, 2]) + hand_ids = self.robot.get_joint_ids("hand", remove_mimic=True) + hand_open = self.robot.get_qpos()[:, hand_ids].mean(dim=-1) > 0.02 + candidate = (distance < 0.08) & (height_error < 0.08) & hand_open + counters = getattr( + self, + "_success_hold_steps", + torch.zeros(self.num_envs, device=self.device, dtype=torch.int32), + ) + counters = torch.where(candidate, counters + 1, torch.zeros_like(counters)) + self._success_hold_steps = counters + success = counters >= 3 + fail = torch.zeros_like(success) + return ( + success, + fail, + { + "cube_target_distance": distance, + "success_hold_steps": counters, + }, + ) + + def check_truncated(self, obs: EnvObs, info: dict[str, Any]) -> torch.Tensor: + del obs, info + return torch.zeros(self.num_envs, device=self.device, dtype=torch.bool) diff --git a/tests/gym/envs/managers/test_action_manager.py b/tests/gym/envs/managers/test_action_manager.py index 3a0cb16b3..352d2b36c 100644 --- a/tests/gym/envs/managers/test_action_manager.py +++ b/tests/gym/envs/managers/test_action_manager.py @@ -80,8 +80,9 @@ class MockEnvForEef(MockEnv): def __init__(self, num_envs: int = 2, action_dim: int = 6): super().__init__(num_envs, action_dim) - def compute_ik(self, pose, joint_seed): + def compute_ik(self, pose, joint_seed, name=None): """Return (all success, joint_seed) to simulate IK success.""" + del pose, name batch_size = joint_seed.shape[0] ret = torch.ones(batch_size, dtype=torch.bool, device=self.device) return ret, joint_seed.clone() diff --git a/tests/gym/envs/task_program/test_configured_integration.py b/tests/gym/envs/task_program/test_configured_integration.py index d1d6d416a..4826aa8bf 100644 --- a/tests/gym/envs/task_program/test_configured_integration.py +++ b/tests/gym/envs/task_program/test_configured_integration.py @@ -94,6 +94,7 @@ frozenset({"pick", "place", "hand_over"}), ), } +_PYTHON_TASK_ENV_MODULES = frozenset({"repeated_pick_place"}) _TEST_ENV_ID = "ConfiguredTaskProgramIntegrationTest-v1" _TABLEWARE_TASKS = { "pour_water": ( @@ -953,8 +954,10 @@ def test_integration_registration_rejects_reusing_an_id_for_changed_config( def test_examples_have_no_importable_task_environment_modules() -> None: - """All three environment implementations are now serialized configuration.""" + """Config-defined examples stay module-free unless RL needs Python state logic.""" for task_name in _TASKS: + if task_name in _PYTHON_TASK_ENV_MODULES: + continue module_name = f"embodichain_tasks.manipulation.{task_name}" assert importlib.util.find_spec(module_name) is None diff --git a/tests/gym/envs/test_official_task_layout.py b/tests/gym/envs/test_official_task_layout.py index 525d8146b..f7634631b 100644 --- a/tests/gym/envs/test_official_task_layout.py +++ b/tests/gym/envs/test_official_task_layout.py @@ -38,6 +38,7 @@ "MatchObjectContainer-v1": "embodichain_tasks.manipulation.tableware.match_object_container", "PlaceObjectDrawer-v1": "embodichain_tasks.manipulation.tableware.place_object_drawer", "PushCubeRL": "embodichain_tasks.manipulation.push_cube", + "RepeatedPickPlaceRlinf-Franka-v1": "embodichain_tasks.manipulation.repeated_pick_place", "ScoopIce-v1": "embodichain_tasks.manipulation.tableware.scoop_ice", "SimpleTask-v1": "embodichain_tasks.special.simple_task", "StackBlocksTwo-v1": "embodichain_tasks.manipulation.tableware.stack_blocks_two", @@ -46,7 +47,7 @@ } REMOVED_AGENT_ENV_IDS = {"PourWaterAgent-v3", "RearrangementAgent-v3"} CONFIG_DEFINED_TASK_PROGRAM_TASKS = {"pour_water"} -RL_SIMULATOR_ENV_IDS = {"CartPoleRL", "PushCubeRL"} +RL_SIMULATOR_ENV_IDS = {"CartPoleRL", "PushCubeRL", "RepeatedPickPlaceRlinf-Franka-v1"} TABLEWARE_CONFIG_TASKS = { "blocks_ranking_rgb", "blocks_ranking_size", diff --git a/tests/test_task_program_package_data.py b/tests/test_task_program_package_data.py index 03c6732d1..3ae9a4c50 100644 --- a/tests/test_task_program_package_data.py +++ b/tests/test_task_program_package_data.py @@ -104,6 +104,8 @@ *_DEPLOYMENTS, Path("tasks/manipulation/repeated_pick_place/catalog.yaml"), Path("tasks/manipulation/repeated_pick_place/README.md"), + Path("tasks/manipulation/repeated_pick_place/task.franka.rlinf.yaml"), + Path("tasks/manipulation/repeated_pick_place/task.franka.rlinf_expert.yaml"), Path("tasks/manipulation/push_cube/catalog.yaml"), Path("tasks/manipulation/push_cube/README.md"), Path("tasks/manipulation/tableware/stack_cups/catalog.yaml"), @@ -114,6 +116,7 @@ Path("components/embodiments/cobotmagic.yaml"), Path("components/embodiments/dual_ur5_dh_pgi_140_80.yaml"), Path("components/embodiments/franka_panda.yaml"), + Path("components/embodiments/franka_panda_vla.yaml"), Path("components/embodiments/ur5_dh_pgi_140_80.yaml"), Path("tasks/manipulation/hand_over/env.yaml"), Path("tasks/manipulation/hand_over/task_program/integration.yaml"), @@ -121,6 +124,7 @@ Path("tasks/manipulation/open_drawer/env.newton.yaml"), Path("tasks/manipulation/open_drawer/task_program/integration.yaml"), Path("tasks/manipulation/repeated_pick_place/env.yaml"), + Path("tasks/manipulation/repeated_pick_place/env.rlinf.yaml"), Path("tasks/manipulation/repeated_pick_place/env.newton.yaml"), Path("tasks/manipulation/repeated_pick_place/task_program/integration.yaml"), Path("tasks/manipulation/tableware/pour_water/env.yaml"),