v0.5 to main (#10)
* v0.5 (#9) * update idql configs * update awr configs * update dipo configs * update qsm configs * update dqm configs * update project version to 0.5.0
This commit is contained in:
Vendored
+3
-1
@@ -1,6 +1,8 @@
|
||||
"""
|
||||
Environment wrapper for D3IL environments with state observations.
|
||||
|
||||
Also return done=False since we do not terminate episode early.
|
||||
|
||||
For consistency, we will use Dict{} for the observation space, with the key "state" for the state observation.
|
||||
"""
|
||||
|
||||
@@ -73,7 +75,7 @@ class D3ilLowdimWrapper(gym.Env):
|
||||
|
||||
# normalize
|
||||
obs = self.normalize_obs(obs)
|
||||
return {"state": obs}, reward, done, info
|
||||
return {"state": obs}, reward, False, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
|
||||
Vendored
+5
-10
@@ -121,7 +121,7 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
|
||||
action = self.normalizer(action, "actions", forward=False)
|
||||
|
||||
# Step the environment n_action_steps times
|
||||
obs, sparse_reward, dense_reward, done, info = self._inner_step(action)
|
||||
obs, sparse_reward, dense_reward, info = self._inner_step(action)
|
||||
if self.sparse_reward:
|
||||
reward = sparse_reward.clone().cpu().numpy()
|
||||
else:
|
||||
@@ -129,17 +129,14 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
|
||||
|
||||
# Only mark the environment as done if it times out, ignore done from inner steps
|
||||
truncated = self.env.env_steps >= self.max_env_steps
|
||||
done = truncated
|
||||
|
||||
nobs: np.ndarray = self.process_obs(obs)
|
||||
done: np.ndarray = done.squeeze().cpu().numpy()
|
||||
truncated: np.ndarray = truncated.squeeze().cpu().numpy()
|
||||
terminated: np.ndarray = np.zeros_like(truncated, dtype=bool)
|
||||
|
||||
return {"state": nobs}, reward, done, info
|
||||
return {"state": nobs}, reward, terminated, truncated, info
|
||||
|
||||
def _inner_step(self, action_chunk: torch.Tensor):
|
||||
dones = torch.zeros(
|
||||
action_chunk.shape[0], dtype=torch.bool, device=action_chunk.device
|
||||
)
|
||||
dense_reward = torch.zeros(action_chunk.shape[0], device=action_chunk.device)
|
||||
sparse_reward = torch.zeros(action_chunk.shape[0], device=action_chunk.device)
|
||||
for i in range(self.n_action_steps):
|
||||
@@ -156,10 +153,8 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
|
||||
# assign "permanent" rewards
|
||||
dense_reward += self.best_reward
|
||||
|
||||
dones = dones | done.squeeze()
|
||||
|
||||
obs = stack_last_n_obs_dict(self.obs, self.n_obs_steps)
|
||||
return obs, sparse_reward, dense_reward, dones, info
|
||||
return obs, sparse_reward, dense_reward, info
|
||||
|
||||
def process_obs(self, obs: torch.Tensor) -> np.ndarray:
|
||||
# Convert the robot state to have 6D pose
|
||||
|
||||
+2
-2
@@ -57,12 +57,12 @@ class MujocoLocomotionLowdimWrapper(gym.Env):
|
||||
def normalize_obs(self, obs):
|
||||
return 2 * ((obs - self.obs_min) / (self.obs_max - self.obs_min + 1e-6) - 0.5)
|
||||
|
||||
def unnormaliza_action(self, action):
|
||||
def unnormalize_action(self, action):
|
||||
action = (action + 1) / 2 # [-1, 1] -> [0, 1]
|
||||
return action * (self.action_max - self.action_min) + self.action_min
|
||||
|
||||
def step(self, action):
|
||||
raw_action = self.unnormaliza_action(action)
|
||||
raw_action = self.unnormalize_action(action)
|
||||
raw_obs, reward, done, info = self.env.step(raw_action)
|
||||
|
||||
# normalize
|
||||
|
||||
Vendored
+25
-9
@@ -138,22 +138,32 @@ class MultiStep(gym.Wrapper):
|
||||
"""
|
||||
if action.ndim == 1: # in case action_steps = 1
|
||||
action = action[None]
|
||||
truncated = False
|
||||
terminated = False
|
||||
for act_step, act in enumerate(action):
|
||||
self.cnt += 1
|
||||
|
||||
if len(self.done) > 0 and self.done[-1]:
|
||||
# termination
|
||||
if terminated or truncated:
|
||||
break
|
||||
|
||||
# done does not differentiate terminal and truncation
|
||||
observation, reward, done, info = self.env.step(act)
|
||||
|
||||
self.obs.append(observation)
|
||||
self.action.append(act)
|
||||
self.reward.append(reward)
|
||||
if (
|
||||
self.max_episode_steps is not None
|
||||
) and self.cnt >= self.max_episode_steps:
|
||||
# truncation
|
||||
done = True
|
||||
|
||||
# in gym, timelimit wrapper is automatically used given env._spec.max_episode_steps
|
||||
if "TimeLimit.truncated" not in info:
|
||||
if done:
|
||||
terminated = True
|
||||
elif (
|
||||
self.max_episode_steps is not None
|
||||
) and self.cnt >= self.max_episode_steps:
|
||||
truncated = True
|
||||
else:
|
||||
truncated = info["TimeLimit.truncated"]
|
||||
terminated = done
|
||||
done = truncated or terminated
|
||||
self.done.append(done)
|
||||
self._add_info(info)
|
||||
observation = self._get_obs(self.n_obs_steps)
|
||||
@@ -165,6 +175,12 @@ class MultiStep(gym.Wrapper):
|
||||
|
||||
# In mujoco case, done can happen within the loop above
|
||||
if self.reset_within_step and self.done[-1]:
|
||||
|
||||
# need to save old observation in the case of truncation only, for bootstrapping
|
||||
if truncated:
|
||||
info["final_obs"] = observation
|
||||
|
||||
# reset
|
||||
observation = (
|
||||
self.reset()
|
||||
) # TODO: arguments? this cannot handle video recording right now since needs to pass in options
|
||||
@@ -173,7 +189,7 @@ class MultiStep(gym.Wrapper):
|
||||
# reset reward and done for next step
|
||||
self.reward = list()
|
||||
self.done = list()
|
||||
return observation, reward, done, info
|
||||
return observation, reward, terminated, truncated, info
|
||||
|
||||
def _get_obs(self, n_steps=1):
|
||||
"""
|
||||
|
||||
+3
-1
@@ -1,6 +1,8 @@
|
||||
"""
|
||||
Environment wrapper for Robomimic environments with image observations.
|
||||
|
||||
Also return done=False since we do not terminate episode early.
|
||||
|
||||
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/env/robomimic/robomimic_image_wrapper.py
|
||||
|
||||
"""
|
||||
@@ -158,7 +160,7 @@ class RobomimicImageWrapper(gym.Env):
|
||||
video_img = self.render(mode="rgb_array")
|
||||
self.video_writer.append_data(video_img)
|
||||
|
||||
return obs, reward, done, info
|
||||
return obs, reward, False, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
|
||||
+4
-2
@@ -1,6 +1,8 @@
|
||||
"""
|
||||
Environment wrapper for Robomimic environments with state observations.
|
||||
|
||||
Also return done=False since we do not terminate episode early.
|
||||
|
||||
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/env/robomimic/robomimic_lowdim_wrapper.py
|
||||
|
||||
For consistency, we will use Dict{} for the observation space, with the key "state" for the state observation.
|
||||
@@ -65,7 +67,7 @@ class RobomimicLowdimWrapper(gym.Env):
|
||||
low=low,
|
||||
high=high,
|
||||
shape=low.shape,
|
||||
dtype=low.dtype,
|
||||
dtype=np.float32,
|
||||
)
|
||||
|
||||
def normalize_obs(self, obs):
|
||||
@@ -129,7 +131,7 @@ class RobomimicLowdimWrapper(gym.Env):
|
||||
video_img = self.render(mode="rgb_array")
|
||||
self.video_writer.append_data(video_img)
|
||||
|
||||
return obs, reward, done, info
|
||||
return obs, reward, False, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
|
||||
Reference in New Issue
Block a user