release
This commit is contained in:
Vendored
+14
@@ -0,0 +1,14 @@
|
||||
from .multi_step import MultiStep
|
||||
from .robomimic_lowdim import RobomimicLowdimWrapper
|
||||
from .robomimic_image import RobomimicImageWrapper
|
||||
from .d3il_lowdim import D3ilLowdimWrapper
|
||||
from .mujoco_locomotion_lowdim import MujocoLocomotionLowdimWrapper
|
||||
|
||||
|
||||
wrapper_dict = {
|
||||
"multi_step": MultiStep,
|
||||
"robomimic_lowdim": RobomimicLowdimWrapper,
|
||||
"robomimic_image": RobomimicImageWrapper,
|
||||
"d3il_lowdim": D3ilLowdimWrapper,
|
||||
"mujoco_locomotion_lowdim": MujocoLocomotionLowdimWrapper,
|
||||
}
|
||||
Vendored
+87
@@ -0,0 +1,87 @@
|
||||
"""
|
||||
Environment wrapper for D3IL environments with state observations.
|
||||
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import gym
|
||||
|
||||
|
||||
class D3ilLowdimWrapper(gym.Env):
|
||||
def __init__(
|
||||
self,
|
||||
env,
|
||||
normalization_path,
|
||||
# init_state=None,
|
||||
# render_hw=(256, 256),
|
||||
# render_camera_name="agentview",
|
||||
):
|
||||
self.env = env
|
||||
# self.init_state = init_state
|
||||
# self.render_hw = render_hw
|
||||
# self.render_camera_name = render_camera_name
|
||||
|
||||
# setup spaces
|
||||
self.action_space = env.action_space
|
||||
self.observation_space = env.observation_space
|
||||
normalization = np.load(normalization_path)
|
||||
self.obs_min = normalization["obs_min"]
|
||||
self.obs_max = normalization["obs_max"]
|
||||
self.action_min = normalization["action_min"]
|
||||
self.action_max = normalization["action_max"]
|
||||
|
||||
# def get_observation(self):
|
||||
# raw_obs = self.env.get_observation()
|
||||
# obs = np.concatenate([raw_obs[key] for key in self.obs_keys], axis=0)
|
||||
# return obs
|
||||
|
||||
def seed(self, seed=None):
|
||||
if seed is not None:
|
||||
np.random.seed(seed=seed)
|
||||
else:
|
||||
np.random.seed()
|
||||
|
||||
def reset(self, **kwargs):
|
||||
"""Ignore passed-in arguments like seed"""
|
||||
options = kwargs.get("options", {})
|
||||
|
||||
new_seed = options.get(
|
||||
"seed", None
|
||||
) # used to set all environments to specified seeds
|
||||
# if self.init_state is not None:
|
||||
# # always reset to the same state to be compatible with gym
|
||||
# self.env.reset_to({"states": self.init_state})
|
||||
if new_seed is not None:
|
||||
self.seed(seed=new_seed)
|
||||
obs = self.env.reset()
|
||||
else:
|
||||
# random reset
|
||||
obs = self.env.reset()
|
||||
|
||||
# normalize
|
||||
obs = self.normalize_obs(obs)
|
||||
return obs
|
||||
|
||||
def normalize_obs(self, obs):
|
||||
return 2 * ((obs - self.obs_min) / (self.obs_max - self.obs_min + 1e-6) - 0.5)
|
||||
|
||||
def unnormaliza_action(self, action):
|
||||
action = (action + 1) / 2 # [-1, 1] -> [0, 1]
|
||||
return action * (self.action_max - self.action_min) + self.action_min
|
||||
|
||||
def step(self, action):
|
||||
action = self.unnormaliza_action(action)
|
||||
obs, reward, done, info = self.env.step(action)
|
||||
|
||||
# normalize
|
||||
obs = self.normalize_obs(obs)
|
||||
return obs, reward, done, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
return self.env.render(
|
||||
mode=mode,
|
||||
height=h,
|
||||
width=w,
|
||||
camera_name=self.render_camera_name,
|
||||
)
|
||||
Vendored
+152
@@ -0,0 +1,152 @@
|
||||
"""
|
||||
Environment wrapper for Furniture-Bench environments.
|
||||
|
||||
"""
|
||||
|
||||
import gym
|
||||
import numpy as np
|
||||
from furniture_bench.envs.furniture_rl_sim_env import FurnitureRLSimEnv
|
||||
import torch
|
||||
from furniture_bench.controllers.control_utils import proprioceptive_quat_to_6d_rotation
|
||||
from ..furniture_normalizer import LinearNormalizer
|
||||
from .multi_step import repeated_space
|
||||
|
||||
import logging
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
|
||||
env: FurnitureRLSimEnv
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
env: FurnitureRLSimEnv,
|
||||
n_obs_steps=1,
|
||||
n_action_steps=1,
|
||||
max_episode_steps=None,
|
||||
sparse_reward=False,
|
||||
reward_agg_method="sum", # never use other types
|
||||
reset_within_step=False,
|
||||
pass_full_observations=False,
|
||||
normalization_path=None,
|
||||
prev_action=False,
|
||||
):
|
||||
assert (
|
||||
not reset_within_step
|
||||
), "reset_within_step must be False for furniture envs"
|
||||
assert n_obs_steps == 1, "n_obs_steps must be 1"
|
||||
assert reward_agg_method == "sum", "reward_agg_method must be sum"
|
||||
assert (
|
||||
not pass_full_observations
|
||||
), "pass_full_observations is not implemented yet"
|
||||
assert not prev_action, "prev_action is not implemented yet"
|
||||
|
||||
super().__init__(env)
|
||||
self._single_action_space = env.action_space
|
||||
self._action_space = repeated_space(env.action_space, n_action_steps)
|
||||
self._observation_space = repeated_space(env.observation_space, n_obs_steps)
|
||||
self.max_episode_steps = max_episode_steps
|
||||
self.n_obs_steps = n_obs_steps
|
||||
self.n_action_steps = n_action_steps
|
||||
self.pass_full_observations = pass_full_observations
|
||||
|
||||
# Use the original reward function where the robot does not receive new reward after completing one part
|
||||
self.sparse_reward = sparse_reward
|
||||
|
||||
# set up normalization
|
||||
self.normalize = normalization_path is not None
|
||||
self.normalizer = LinearNormalizer()
|
||||
self.normalizer.load_state_dict(
|
||||
torch.load(normalization_path, map_location=self.device, weights_only=True)
|
||||
)
|
||||
log.info(f"Loaded normalization from {normalization_path}")
|
||||
|
||||
def reset(
|
||||
self,
|
||||
**kwargs,
|
||||
):
|
||||
"""Resets the environment."""
|
||||
obs = self.env.reset()
|
||||
nobs = self.process_obs(obs)
|
||||
self.best_reward = torch.zeros(self.env.num_envs).to(self.device)
|
||||
self.done = list()
|
||||
|
||||
return nobs
|
||||
|
||||
def reset_arg(self, options_list=None):
|
||||
return self.reset()
|
||||
|
||||
def reset_one_arg(self, env_ind=None, options=None):
|
||||
if env_ind is not None:
|
||||
env_ind = torch.tensor([env_ind], device=self.device)
|
||||
|
||||
return self.reset()
|
||||
|
||||
def step(self, action: np.ndarray):
|
||||
"""
|
||||
Takes in a chunk of actions of length n_action_steps
|
||||
and steps the environment n_action_steps times
|
||||
and returns an aggregated observation, reward, and done signal
|
||||
"""
|
||||
# action: (n_envs, n_action_steps, action_dim)
|
||||
action = torch.tensor(action, device=self.device)
|
||||
|
||||
# Denormalize the action
|
||||
action = self.normalizer(action, "actions", forward=False)
|
||||
|
||||
# Step the environment n_action_steps times
|
||||
obs, sparse_reward, dense_reward, done, info = self._inner_step(action)
|
||||
if self.sparse_reward:
|
||||
reward = sparse_reward.clone().cpu().numpy()
|
||||
else:
|
||||
reward = dense_reward.clone().cpu().numpy()
|
||||
|
||||
# Only mark the environment as done if it times out, ignore done from inner steps
|
||||
truncated = self.env.env_steps >= self.max_env_steps
|
||||
done = truncated
|
||||
|
||||
nobs: np.ndarray = self.process_obs(obs)
|
||||
done: np.ndarray = done.squeeze().cpu().numpy()
|
||||
|
||||
return (nobs, reward, done, info)
|
||||
|
||||
def _inner_step(self, action_chunk: torch.Tensor):
|
||||
dones = torch.zeros(
|
||||
action_chunk.shape[0], dtype=torch.bool, device=action_chunk.device
|
||||
)
|
||||
dense_reward = torch.zeros(action_chunk.shape[0], device=action_chunk.device)
|
||||
sparse_reward = torch.zeros(action_chunk.shape[0], device=action_chunk.device)
|
||||
for i in range(self.n_action_steps):
|
||||
# The dimensions of the action_chunk are (num_envs, chunk_size, action_dim)
|
||||
obs, reward, done, info = self.env.step(action_chunk[:, i, :])
|
||||
|
||||
# track raw reward
|
||||
sparse_reward += reward.squeeze()
|
||||
|
||||
# track best reward --- reward nonzero only one part is assembled
|
||||
self.best_reward += reward.squeeze()
|
||||
|
||||
# assign "permanent" rewards
|
||||
dense_reward += self.best_reward
|
||||
|
||||
dones = dones | done.squeeze()
|
||||
|
||||
return obs, sparse_reward, dense_reward, dones, info
|
||||
|
||||
def process_obs(self, obs: torch.Tensor) -> np.ndarray:
|
||||
robot_state = obs["robot_state"]
|
||||
|
||||
# Convert the robot state to have 6D pose
|
||||
robot_state = proprioceptive_quat_to_6d_rotation(robot_state)
|
||||
|
||||
parts_poses = obs["parts_poses"]
|
||||
|
||||
obs = torch.cat([robot_state, parts_poses], dim=-1)
|
||||
nobs = self.normalizer(obs, "observations", forward=True)
|
||||
nobs = torch.clamp(nobs, -5, 5)
|
||||
|
||||
# Insert a dummy dimension for the n_obs_steps (n_envs, obs_dim) -> (n_envs, n_obs_steps, obs_dim)
|
||||
nobs = nobs.unsqueeze(1).cpu().numpy()
|
||||
|
||||
return nobs
|
||||
@@ -0,0 +1,61 @@
|
||||
"""
|
||||
Environment wrapper for Gym environments (MuJoCo locomotion tasks) with state observations.
|
||||
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import gym
|
||||
|
||||
|
||||
class MujocoLocomotionLowdimWrapper(gym.Env):
|
||||
def __init__(
|
||||
self,
|
||||
env,
|
||||
normalization_path,
|
||||
):
|
||||
self.env = env
|
||||
|
||||
# setup spaces
|
||||
self.action_space = env.action_space
|
||||
self.observation_space = env.observation_space
|
||||
normalization = np.load(normalization_path)
|
||||
self.obs_min = normalization["obs_min"]
|
||||
self.obs_max = normalization["obs_max"]
|
||||
self.action_min = normalization["action_min"]
|
||||
self.action_max = normalization["action_max"]
|
||||
|
||||
def seed(self, seed=None):
|
||||
if seed is not None:
|
||||
np.random.seed(seed=seed)
|
||||
else:
|
||||
np.random.seed()
|
||||
|
||||
def reset(self, **kwargs):
|
||||
"""Ignore passed-in arguments like seed"""
|
||||
options = kwargs.get("options", {})
|
||||
new_seed = options.get("seed", None)
|
||||
if new_seed is not None:
|
||||
self.seed(seed=new_seed)
|
||||
raw_obs = self.env.reset()
|
||||
|
||||
# normalize
|
||||
obs = self.normalize_obs(raw_obs)
|
||||
return obs
|
||||
|
||||
def normalize_obs(self, obs):
|
||||
return 2 * ((obs - self.obs_min) / (self.obs_max - self.obs_min + 1e-6) - 0.5)
|
||||
|
||||
def unnormaliza_action(self, action):
|
||||
action = (action + 1) / 2 # [-1, 1] -> [0, 1]
|
||||
return action * (self.action_max - self.action_min) + self.action_min
|
||||
|
||||
def step(self, action):
|
||||
raw_action = self.unnormaliza_action(action)
|
||||
raw_obs, reward, done, info = self.env.step(raw_action)
|
||||
|
||||
# normalize
|
||||
obs = self.normalize_obs(raw_obs)
|
||||
return obs, reward, done, info
|
||||
|
||||
def render(self, **kwargs):
|
||||
return self.env.render()
|
||||
Vendored
+283
@@ -0,0 +1,283 @@
|
||||
"""
|
||||
Multi-step wrapper. Allow executing multiple environmnt steps. Returns stacked observation and optionally stacked previous action.
|
||||
|
||||
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/gym_util/multistep_wrapper.py
|
||||
|
||||
"""
|
||||
|
||||
import gym
|
||||
from typing import Optional
|
||||
from gym import spaces
|
||||
import numpy as np
|
||||
from collections import defaultdict, deque
|
||||
|
||||
# import dill
|
||||
|
||||
|
||||
def stack_repeated(x, n):
|
||||
return np.repeat(np.expand_dims(x, axis=0), n, axis=0)
|
||||
|
||||
|
||||
def repeated_box(box_space, n):
|
||||
return spaces.Box(
|
||||
low=stack_repeated(box_space.low, n),
|
||||
high=stack_repeated(box_space.high, n),
|
||||
shape=(n,) + box_space.shape,
|
||||
dtype=box_space.dtype,
|
||||
)
|
||||
|
||||
|
||||
def repeated_space(space, n):
|
||||
if isinstance(space, spaces.Box):
|
||||
return repeated_box(space, n)
|
||||
elif isinstance(space, spaces.Dict):
|
||||
result_space = spaces.Dict()
|
||||
for key, value in space.items():
|
||||
result_space[key] = repeated_space(value, n)
|
||||
return result_space
|
||||
else:
|
||||
raise RuntimeError(f"Unsupported space type {type(space)}")
|
||||
|
||||
|
||||
def take_last_n(x, n):
|
||||
x = list(x)
|
||||
n = min(len(x), n)
|
||||
return np.array(x[-n:])
|
||||
|
||||
|
||||
def dict_take_last_n(x, n):
|
||||
result = dict()
|
||||
for key, value in x.items():
|
||||
result[key] = take_last_n(value, n)
|
||||
return result
|
||||
|
||||
|
||||
def aggregate(data, method="max"):
|
||||
if method == "max":
|
||||
# equivalent to any
|
||||
return np.max(data)
|
||||
elif method == "min":
|
||||
# equivalent to all
|
||||
return np.min(data)
|
||||
elif method == "mean":
|
||||
return np.mean(data)
|
||||
elif method == "sum":
|
||||
return np.sum(data)
|
||||
else:
|
||||
raise NotImplementedError()
|
||||
|
||||
|
||||
def stack_last_n_obs(all_obs, n_steps):
|
||||
"""Apply padding"""
|
||||
assert len(all_obs) > 0
|
||||
all_obs = list(all_obs)
|
||||
result = np.zeros((n_steps,) + all_obs[-1].shape, dtype=all_obs[-1].dtype)
|
||||
start_idx = -min(n_steps, len(all_obs))
|
||||
result[start_idx:] = np.array(all_obs[start_idx:])
|
||||
if n_steps > len(all_obs):
|
||||
# pad
|
||||
result[:start_idx] = result[start_idx]
|
||||
return result
|
||||
|
||||
|
||||
class MultiStep(gym.Wrapper):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
env,
|
||||
n_obs_steps=1,
|
||||
n_action_steps=1,
|
||||
max_episode_steps=None,
|
||||
reward_agg_method="sum", # never use other types
|
||||
prev_action=True,
|
||||
reset_within_step=False,
|
||||
pass_full_observations=False,
|
||||
verbose=False,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(env)
|
||||
self._single_action_space = env.action_space
|
||||
self._action_space = repeated_space(env.action_space, n_action_steps)
|
||||
self._observation_space = repeated_space(env.observation_space, n_obs_steps)
|
||||
self.max_episode_steps = max_episode_steps
|
||||
self.n_obs_steps = n_obs_steps
|
||||
self.n_action_steps = n_action_steps
|
||||
self.reward_agg_method = reward_agg_method
|
||||
self.prev_action = prev_action
|
||||
self.reset_within_step = reset_within_step
|
||||
self.pass_full_observations = pass_full_observations
|
||||
self.verbose = verbose
|
||||
|
||||
def reset(
|
||||
self,
|
||||
seed: Optional[int] = None,
|
||||
return_info: bool = False,
|
||||
options: dict = {},
|
||||
):
|
||||
"""Resets the environment."""
|
||||
obs = self.env.reset(
|
||||
seed=seed,
|
||||
options=options,
|
||||
return_info=return_info,
|
||||
)
|
||||
self.obs = deque([obs], maxlen=max(self.n_obs_steps + 1, self.n_action_steps))
|
||||
if self.prev_action:
|
||||
self.action = deque(
|
||||
[self._single_action_space.sample()], maxlen=self.n_obs_steps
|
||||
)
|
||||
self.reward = list()
|
||||
self.done = list()
|
||||
self.info = defaultdict(lambda: deque(maxlen=self.n_obs_steps + 1))
|
||||
obs = self._get_obs(self.n_obs_steps)
|
||||
|
||||
self.cnt = 0
|
||||
return obs
|
||||
|
||||
def step(self, action):
|
||||
"""
|
||||
actions: (n_action_steps,) + action_shape
|
||||
"""
|
||||
if action.ndim == 1: # in case action_steps = 1
|
||||
action = action[None]
|
||||
for act_step, act in enumerate(action):
|
||||
self.cnt += 1
|
||||
|
||||
if len(self.done) > 0 and self.done[-1]:
|
||||
# termination
|
||||
break
|
||||
observation, reward, done, info = self.env.step(act)
|
||||
|
||||
self.obs.append(observation)
|
||||
self.action.append(act)
|
||||
self.reward.append(reward)
|
||||
if (
|
||||
self.max_episode_steps is not None
|
||||
) and self.cnt >= self.max_episode_steps:
|
||||
# truncation
|
||||
done = True
|
||||
self.done.append(done)
|
||||
self._add_info(info)
|
||||
|
||||
observation = self._get_obs(self.n_obs_steps)
|
||||
reward = aggregate(self.reward, self.reward_agg_method)
|
||||
done = aggregate(self.done, "max")
|
||||
info = dict_take_last_n(self.info, self.n_obs_steps)
|
||||
if self.pass_full_observations: # right now this assume n_obs_steps = 1
|
||||
info["full_obs"] = self._get_obs(act_step + 1)
|
||||
|
||||
# In mujoco case, done can happen within the loop above
|
||||
if self.reset_within_step and self.done[-1]:
|
||||
observation = (
|
||||
self.reset()
|
||||
) # TODO: arguments? this cannot handle video recording right now since needs to pass in options
|
||||
self.verbose and print("Reset env within wrapper.")
|
||||
|
||||
# reset reward and done for next step
|
||||
self.reward = list()
|
||||
self.done = list()
|
||||
return observation, reward, done, info
|
||||
|
||||
def _get_obs(self, n_steps=1):
|
||||
"""
|
||||
Output (n_steps,) + obs_shape
|
||||
"""
|
||||
assert len(self.obs) > 0
|
||||
if isinstance(self.observation_space, spaces.Box):
|
||||
return stack_last_n_obs(self.obs, n_steps)
|
||||
elif isinstance(self.observation_space, spaces.Dict):
|
||||
result = dict()
|
||||
for key in self.observation_space.keys():
|
||||
result[key] = stack_last_n_obs([obs[key] for obs in self.obs], n_steps)
|
||||
return result
|
||||
else:
|
||||
raise RuntimeError("Unsupported space type")
|
||||
|
||||
def get_prev_action(self, n_steps=None):
|
||||
if n_steps is None:
|
||||
n_steps = self.n_obs_steps - 1 # exclude current step
|
||||
assert len(self.action) > 0
|
||||
return stack_last_n_obs(self.action, n_steps)
|
||||
|
||||
def _add_info(self, info):
|
||||
for key, value in info.items():
|
||||
self.info[key].append(value)
|
||||
|
||||
def render(self, **kwargs):
|
||||
"""Not the best design"""
|
||||
return self.env.render(**kwargs)
|
||||
|
||||
# def get_rewards(self):
|
||||
# return self.reward
|
||||
|
||||
# def get_attr(self, name):
|
||||
# return getattr(self, name)
|
||||
|
||||
# def run_dill_function(self, dill_fn):
|
||||
# fn = dill.loads(dill_fn)
|
||||
# return fn(self)
|
||||
|
||||
# def get_infos(self):
|
||||
# result = dict()
|
||||
# for k, v in self.info.items():
|
||||
# result[k] = list(v)
|
||||
# return result
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import os
|
||||
from omegaconf import OmegaConf
|
||||
import json
|
||||
|
||||
os.environ["MUJOCO_GL"] = "egl"
|
||||
|
||||
cfg = OmegaConf.load("cfg/robomimic/finetune/can/ft_ppo_diffusion_mlp_img.yaml")
|
||||
shape_meta = cfg["shape_meta"]
|
||||
|
||||
import robomimic.utils.env_utils as EnvUtils
|
||||
import robomimic.utils.obs_utils as ObsUtils
|
||||
import matplotlib.pyplot as plt
|
||||
from env.gym_utils.wrapper.robomimic_image import RobomimicImageWrapper
|
||||
|
||||
wrappers = cfg.env.wrappers
|
||||
obs_modality_dict = {
|
||||
"low_dim": (
|
||||
wrappers.robomimic_image.low_dim_keys
|
||||
if "robomimic_image" in wrappers
|
||||
else wrappers.robomimic_lowdim.low_dim_keys
|
||||
),
|
||||
"rgb": (
|
||||
wrappers.robomimic_image.image_keys
|
||||
if "robomimic_image" in wrappers
|
||||
else None
|
||||
),
|
||||
}
|
||||
if obs_modality_dict["rgb"] is None:
|
||||
obs_modality_dict.pop("rgb")
|
||||
ObsUtils.initialize_obs_modality_mapping_from_dict(obs_modality_dict)
|
||||
|
||||
with open(cfg.robomimic_env_cfg_path, "r") as f:
|
||||
env_meta = json.load(f)
|
||||
env = EnvUtils.create_env_from_metadata(
|
||||
env_meta=env_meta,
|
||||
render=False,
|
||||
render_offscreen=False,
|
||||
use_image_obs=True,
|
||||
)
|
||||
env.env.hard_reset = False
|
||||
|
||||
wrapper = MultiStep(
|
||||
env=RobomimicImageWrapper(
|
||||
env=env,
|
||||
shape_meta=shape_meta,
|
||||
image_keys=["robot0_eye_in_hand_image"],
|
||||
),
|
||||
n_obs_steps=1,
|
||||
n_action_steps=1,
|
||||
)
|
||||
wrapper.seed(0)
|
||||
obs = wrapper.reset()
|
||||
print(obs.keys())
|
||||
img = wrapper.render()
|
||||
wrapper.close()
|
||||
plt.imshow(img)
|
||||
plt.savefig("test.png")
|
||||
+227
@@ -0,0 +1,227 @@
|
||||
"""
|
||||
Environment wrapper for Robomimic environments with image observations.
|
||||
|
||||
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/env/robomimic/robomimic_image_wrapper.py
|
||||
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import gym
|
||||
from gym import spaces
|
||||
import imageio
|
||||
|
||||
|
||||
class RobomimicImageWrapper(gym.Env):
|
||||
def __init__(
|
||||
self,
|
||||
env,
|
||||
shape_meta: dict,
|
||||
normalization_path=None,
|
||||
low_dim_keys=[
|
||||
"robot0_eef_pos",
|
||||
"robot0_eef_quat",
|
||||
"robot0_gripper_qpos",
|
||||
],
|
||||
image_keys=[
|
||||
"agentview_image",
|
||||
"robot0_eye_in_hand_image",
|
||||
],
|
||||
clamp_obs=False,
|
||||
init_state=None,
|
||||
render_hw=(256, 256),
|
||||
render_camera_name="agentview",
|
||||
):
|
||||
self.env = env
|
||||
self.init_state = init_state
|
||||
self.has_reset_before = False
|
||||
self.render_hw = render_hw
|
||||
self.render_camera_name = render_camera_name
|
||||
self.video_writer = None
|
||||
self.clamp_obs = clamp_obs
|
||||
|
||||
# set up normalization
|
||||
self.normalize = normalization_path is not None
|
||||
if self.normalize:
|
||||
normalization = np.load(normalization_path)
|
||||
self.obs_min = normalization["obs_min"]
|
||||
self.obs_max = normalization["obs_max"]
|
||||
self.action_min = normalization["action_min"]
|
||||
self.action_max = normalization["action_max"]
|
||||
|
||||
# setup spaces
|
||||
low = np.full(env.action_dimension, fill_value=-1)
|
||||
high = np.full(env.action_dimension, fill_value=1)
|
||||
self.action_space = gym.spaces.Box(
|
||||
low=low,
|
||||
high=high,
|
||||
shape=low.shape,
|
||||
dtype=low.dtype,
|
||||
)
|
||||
self.low_dim_keys = low_dim_keys
|
||||
self.image_keys = image_keys
|
||||
self.obs_keys = low_dim_keys + image_keys
|
||||
observation_space = spaces.Dict()
|
||||
for key, value in shape_meta["obs"].items():
|
||||
shape = value["shape"]
|
||||
if key.endswith("rgb"):
|
||||
min_value, max_value = 0, 1
|
||||
elif key.endswith("state"):
|
||||
min_value, max_value = -1, 1
|
||||
else:
|
||||
raise RuntimeError(f"Unsupported type {key}")
|
||||
this_space = spaces.Box(
|
||||
low=min_value,
|
||||
high=max_value,
|
||||
shape=shape,
|
||||
dtype=np.float32,
|
||||
)
|
||||
observation_space[key] = this_space
|
||||
self.observation_space = observation_space
|
||||
|
||||
def normalize_obs(self, obs):
|
||||
obs = 2 * (
|
||||
(obs - self.obs_min) / (self.obs_max - self.obs_min + 1e-6) - 0.5
|
||||
) # -> [-1, 1]
|
||||
if self.clamp_obs:
|
||||
obs = np.clip(obs, -1, 1)
|
||||
return obs
|
||||
|
||||
def unnormalize_action(self, action):
|
||||
action = (action + 1) / 2 # [-1, 1] -> [0, 1]
|
||||
return action * (self.action_max - self.action_min) + self.action_min
|
||||
|
||||
def get_observation(self, raw_obs=None):
|
||||
if raw_obs is None:
|
||||
raw_obs = self.env.get_observation()
|
||||
obs = {"rgb": None, "state": None} # stack rgb if multiple cameras
|
||||
for key in self.obs_keys:
|
||||
if key in self.image_keys:
|
||||
if obs["rgb"] is None:
|
||||
obs["rgb"] = raw_obs[key]
|
||||
else:
|
||||
obs["rgb"] = np.concatenate(
|
||||
[obs["rgb"], raw_obs[key]], axis=0
|
||||
) # C H W
|
||||
else:
|
||||
if obs["state"] is None:
|
||||
obs["state"] = raw_obs[key]
|
||||
else:
|
||||
obs["state"] = np.concatenate([obs["state"], raw_obs[key]], axis=-1)
|
||||
if self.normalize:
|
||||
obs["state"] = self.normalize_obs(obs["state"])
|
||||
obs["rgb"] *= 255 # [0, 1] -> [0, 255], in float64
|
||||
return obs
|
||||
|
||||
def seed(self, seed=None):
|
||||
if seed is not None:
|
||||
np.random.seed(seed=seed)
|
||||
else:
|
||||
np.random.seed()
|
||||
|
||||
def reset(self, options={}, **kwargs):
|
||||
"""Ignore passed-in arguments like seed"""
|
||||
# Close video if exists
|
||||
if self.video_writer is not None:
|
||||
self.video_writer.close()
|
||||
self.video_writer = None
|
||||
|
||||
# Start video if specified
|
||||
if "video_path" in options:
|
||||
self.video_writer = imageio.get_writer(options["video_path"], fps=30)
|
||||
|
||||
# Call reset
|
||||
new_seed = options.get(
|
||||
"seed", None
|
||||
) # used to set all environments to specified seeds
|
||||
if self.init_state is not None:
|
||||
if not self.has_reset_before:
|
||||
# the env must be fully reset at least once to ensure correct rendering
|
||||
self.env.reset()
|
||||
self.has_reset_before = True
|
||||
|
||||
# always reset to the same state to be compatible with gym
|
||||
raw_obs = self.env.reset_to({"states": self.init_state})
|
||||
elif new_seed is not None:
|
||||
self.seed(seed=new_seed)
|
||||
raw_obs = self.env.reset()
|
||||
else:
|
||||
# random reset
|
||||
raw_obs = self.env.reset()
|
||||
return self.get_observation(raw_obs)
|
||||
|
||||
def step(self, action):
|
||||
if self.normalize:
|
||||
action = self.unnormalize_action(action)
|
||||
raw_obs, reward, done, info = self.env.step(action)
|
||||
obs = self.get_observation(raw_obs)
|
||||
|
||||
# render if specified
|
||||
if self.video_writer is not None:
|
||||
video_img = self.render(mode="rgb_array")
|
||||
self.video_writer.append_data(video_img)
|
||||
|
||||
return obs, reward, done, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
return self.env.render(
|
||||
mode=mode,
|
||||
height=h,
|
||||
width=w,
|
||||
camera_name=self.render_camera_name,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import os
|
||||
from omegaconf import OmegaConf
|
||||
import json
|
||||
|
||||
os.environ["MUJOCO_GL"] = "egl"
|
||||
|
||||
cfg = OmegaConf.load("cfg/robomimic/finetune/can/ft_ppo_diffusion_mlp_img.yaml")
|
||||
shape_meta = cfg["shape_meta"]
|
||||
|
||||
import robomimic.utils.env_utils as EnvUtils
|
||||
import robomimic.utils.obs_utils as ObsUtils
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
wrappers = cfg.env.wrappers
|
||||
obs_modality_dict = {
|
||||
"low_dim": (
|
||||
wrappers.robomimic_image.low_dim_keys
|
||||
if "robomimic_image" in wrappers
|
||||
else wrappers.robomimic_lowdim.low_dim_keys
|
||||
),
|
||||
"rgb": (
|
||||
wrappers.robomimic_image.image_keys
|
||||
if "robomimic_image" in wrappers
|
||||
else None
|
||||
),
|
||||
}
|
||||
if obs_modality_dict["rgb"] is None:
|
||||
obs_modality_dict.pop("rgb")
|
||||
ObsUtils.initialize_obs_modality_mapping_from_dict(obs_modality_dict)
|
||||
|
||||
with open(cfg.robomimic_env_cfg_path, "r") as f:
|
||||
env_meta = json.load(f)
|
||||
env = EnvUtils.create_env_from_metadata(
|
||||
env_meta=env_meta,
|
||||
render=False,
|
||||
render_offscreen=False,
|
||||
use_image_obs=True,
|
||||
)
|
||||
env.env.hard_reset = False
|
||||
|
||||
wrapper = RobomimicImageWrapper(
|
||||
env=env,
|
||||
shape_meta=shape_meta,
|
||||
image_keys=["robot0_eye_in_hand_image"],
|
||||
)
|
||||
wrapper.seed(0)
|
||||
obs = wrapper.reset()
|
||||
print(obs.keys())
|
||||
img = wrapper.render()
|
||||
wrapper.close()
|
||||
plt.imshow(img)
|
||||
plt.savefig("test.png")
|
||||
+142
@@ -0,0 +1,142 @@
|
||||
"""
|
||||
Environment wrapper for Robomimic environments with state observations.
|
||||
|
||||
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/env/robomimic/robomimic_lowdim_wrapper.py
|
||||
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import gym
|
||||
from gym.spaces import Box
|
||||
import imageio
|
||||
|
||||
|
||||
class RobomimicLowdimWrapper(gym.Env):
|
||||
def __init__(
|
||||
self,
|
||||
env,
|
||||
normalization_path=None,
|
||||
low_dim_keys=[
|
||||
"robot0_eef_pos",
|
||||
"robot0_eef_quat",
|
||||
"robot0_gripper_qpos",
|
||||
"object",
|
||||
],
|
||||
clamp_obs=False,
|
||||
init_state=None,
|
||||
render_hw=(256, 256),
|
||||
render_camera_name="agentview",
|
||||
):
|
||||
self.env = env
|
||||
self.obs_keys = low_dim_keys
|
||||
self.init_state = init_state
|
||||
self.render_hw = render_hw
|
||||
self.render_camera_name = render_camera_name
|
||||
self.video_writer = None
|
||||
self.clamp_obs = clamp_obs
|
||||
|
||||
# set up normalization
|
||||
self.normalize = normalization_path is not None
|
||||
if self.normalize:
|
||||
normalization = np.load(normalization_path)
|
||||
self.obs_min = normalization["obs_min"]
|
||||
self.obs_max = normalization["obs_max"]
|
||||
self.action_min = normalization["action_min"]
|
||||
self.action_max = normalization["action_max"]
|
||||
|
||||
# setup spaces - use [-1, 1]
|
||||
low = np.full(env.action_dimension, fill_value=-1)
|
||||
high = np.full(env.action_dimension, fill_value=1)
|
||||
self.action_space = Box(
|
||||
low=low,
|
||||
high=high,
|
||||
shape=low.shape,
|
||||
dtype=low.dtype,
|
||||
)
|
||||
obs_example = self.get_observation()
|
||||
low = np.full_like(obs_example, fill_value=-1)
|
||||
high = np.full_like(obs_example, fill_value=1)
|
||||
self.observation_space = Box(
|
||||
low=low,
|
||||
high=high,
|
||||
shape=low.shape,
|
||||
dtype=low.dtype,
|
||||
)
|
||||
|
||||
def normalize_obs(self, obs):
|
||||
obs = 2 * (
|
||||
(obs - self.obs_min) / (self.obs_max - self.obs_min + 1e-6) - 0.5
|
||||
) # -> [-1, 1]
|
||||
if self.clamp_obs:
|
||||
obs = np.clip(obs, -1, 1)
|
||||
return obs
|
||||
|
||||
def unnormalize_action(self, action):
|
||||
action = (action + 1) / 2 # [-1, 1] -> [0, 1]
|
||||
return action * (self.action_max - self.action_min) + self.action_min
|
||||
|
||||
def get_observation(self):
|
||||
raw_obs = self.env.get_observation()
|
||||
raw_obs = np.concatenate([raw_obs[key] for key in self.obs_keys], axis=0)
|
||||
if self.normalize:
|
||||
return self.normalize_obs(raw_obs)
|
||||
return raw_obs
|
||||
|
||||
def seed(self, seed=None):
|
||||
if seed is not None:
|
||||
np.random.seed(seed=seed)
|
||||
else:
|
||||
np.random.seed()
|
||||
|
||||
def reset(self, options={}, **kwargs):
|
||||
"""Ignore passed-in arguments like seed"""
|
||||
|
||||
# Close video if exists
|
||||
if self.video_writer is not None:
|
||||
self.video_writer.close()
|
||||
self.video_writer = None
|
||||
|
||||
# Start video if specified
|
||||
if "video_path" in options:
|
||||
self.video_writer = imageio.get_writer(options["video_path"], fps=30)
|
||||
|
||||
# Call reset
|
||||
new_seed = options.get(
|
||||
"seed", None
|
||||
) # used to set all environments to specified seeds
|
||||
if self.init_state is not None:
|
||||
# always reset to the same state to be compatible with gym
|
||||
self.env.reset_to({"states": self.init_state})
|
||||
elif new_seed is not None:
|
||||
self.seed(seed=new_seed)
|
||||
self.env.reset()
|
||||
else:
|
||||
# random reset
|
||||
self.env.reset()
|
||||
return self.get_observation()
|
||||
|
||||
def step(self, action):
|
||||
if self.normalize:
|
||||
action = self.unnormalize_action(action)
|
||||
raw_obs, reward, done, info = self.env.step(action)
|
||||
raw_obs = np.concatenate([raw_obs[key] for key in self.obs_keys], axis=0)
|
||||
if self.normalize:
|
||||
obs = self.normalize_obs(raw_obs)
|
||||
else:
|
||||
obs = raw_obs
|
||||
|
||||
# render if specified
|
||||
if self.video_writer is not None:
|
||||
video_img = self.render(mode="rgb_array")
|
||||
self.video_writer.append_data(video_img)
|
||||
|
||||
return obs, reward, done, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
return self.env.render(
|
||||
mode=mode,
|
||||
height=h,
|
||||
width=w,
|
||||
camera_name=self.render_camera_name,
|
||||
)
|
||||
Reference in New Issue
Block a user