v0.5 to main (#10)
* v0.5 (#9) * update idql configs * update awr configs * update dipo configs * update qsm configs * update dqm configs * update project version to 0.5.0
This commit is contained in:
Vendored
+8
-6
@@ -184,8 +184,8 @@ def make_async(
|
||||
# Create a fake env whose sole purpose is to provide
|
||||
# obs/action spaces and metadata.
|
||||
env = gym.Env()
|
||||
observation_space = spaces.Dict()
|
||||
if shape_meta is not None: # rn only for images
|
||||
observation_space = spaces.Dict()
|
||||
for key, value in shape_meta["obs"].items():
|
||||
shape = value["shape"]
|
||||
if key.endswith("rgb"):
|
||||
@@ -194,18 +194,20 @@ def make_async(
|
||||
min_value, max_value = -1, 1
|
||||
else:
|
||||
raise RuntimeError(f"Unsupported type {key}")
|
||||
this_space = spaces.Box(
|
||||
observation_space[key] = spaces.Box(
|
||||
low=min_value,
|
||||
high=max_value,
|
||||
shape=shape,
|
||||
dtype=np.float32,
|
||||
)
|
||||
observation_space[key] = this_space
|
||||
env.observation_space = observation_space
|
||||
else:
|
||||
env.observation_space = gym.spaces.Box(
|
||||
-1, 1, shape=(obs_dim,), dtype=np.float64
|
||||
observation_space["state"] = gym.spaces.Box(
|
||||
-1,
|
||||
1,
|
||||
shape=(obs_dim,),
|
||||
dtype=np.float32,
|
||||
)
|
||||
env.observation_space = observation_space
|
||||
env.action_space = gym.spaces.Box(-1, 1, shape=(action_dim,), dtype=np.int64)
|
||||
env.metadata = {
|
||||
"render.modes": ["human", "rgb_array", "depth_array"],
|
||||
|
||||
Vendored
+16
-15
@@ -1,7 +1,10 @@
|
||||
"""
|
||||
From gym==0.22.0
|
||||
|
||||
Disable auto-reset after done.
|
||||
Use terminated/truncated instead of done.
|
||||
|
||||
Disable auto-reset after done. Reset in MultiStepWrapper instead.
|
||||
|
||||
Add reset_arg() that allows all environments with different options.
|
||||
Add reset_one_arg() that allows resetting a single environment with options.
|
||||
Add render().
|
||||
@@ -398,8 +401,11 @@ class AsyncVectorEnv(VectorEnv):
|
||||
rewards : :obj:`np.ndarray`, dtype :obj:`np.float_`
|
||||
A vector of rewards from the vectorized environment.
|
||||
|
||||
dones : :obj:`np.ndarray`, dtype :obj:`np.bool_`
|
||||
A vector whose entries indicate whether the episode has ended.
|
||||
terminates : :obj:`np.ndarray`, dtype :obj:`np.bool_`
|
||||
A vector whose entries indicate whether the episode has terminated (failed).
|
||||
|
||||
truncates : :obj:`np.ndarray`, dtype :obj:`np.bool_`
|
||||
A vector whose entries indicate whether the episode has been truncated (max episode length).
|
||||
|
||||
infos : list of dict
|
||||
A list of auxiliary diagnostic information dicts from sub-environments.
|
||||
@@ -432,7 +438,7 @@ class AsyncVectorEnv(VectorEnv):
|
||||
results, successes = zip(*[pipe.recv() for pipe in self.parent_pipes])
|
||||
self._raise_if_errors(successes)
|
||||
self._state = AsyncState.DEFAULT
|
||||
observations_list, rewards, dones, infos = zip(*results)
|
||||
observations_list, rewards, terminates, truncates, infos = zip(*results)
|
||||
|
||||
if not self.shared_memory:
|
||||
self.observations = concatenate(
|
||||
@@ -444,7 +450,8 @@ class AsyncVectorEnv(VectorEnv):
|
||||
return (
|
||||
deepcopy(self.observations) if self.copy else self.observations,
|
||||
np.array(rewards),
|
||||
np.array(dones, dtype=np.bool_),
|
||||
np.array(terminates, dtype=np.bool_),
|
||||
np.array(truncates, dtype=np.bool_),
|
||||
infos,
|
||||
)
|
||||
|
||||
@@ -717,11 +724,8 @@ def _worker(index, env_fn, pipe, parent_pipe, shared_memory, error_queue):
|
||||
pipe.send((observation, True))
|
||||
|
||||
elif command == "step":
|
||||
observation, reward, done, info = env.step(data)
|
||||
# if done:
|
||||
# info["terminal_observation"] = observation
|
||||
# observation = env.reset()
|
||||
pipe.send(((observation, reward, done, info), True))
|
||||
observation, reward, terminated, truncated, info = env.step(data)
|
||||
pipe.send(((observation, reward, terminated, truncated, info), True))
|
||||
elif command == "seed":
|
||||
env.seed(data)
|
||||
pipe.send((None, True))
|
||||
@@ -789,14 +793,11 @@ def _worker_shared_memory(index, env_fn, pipe, parent_pipe, shared_memory, error
|
||||
)
|
||||
pipe.send((None, True))
|
||||
elif command == "step":
|
||||
observation, reward, done, info = env.step(data)
|
||||
# if done:
|
||||
# info["terminal_observation"] = observation
|
||||
# observation = env.reset()
|
||||
observation, reward, terminated, truncated, info = env.step(data)
|
||||
write_to_shared_memory(
|
||||
observation_space, index, observation, shared_memory
|
||||
)
|
||||
pipe.send(((None, reward, done, info), True))
|
||||
pipe.send(((None, reward, terminated, truncated, info), True))
|
||||
elif command == "seed":
|
||||
env.seed(data)
|
||||
pipe.send((None, True))
|
||||
|
||||
Vendored
+13
-7
@@ -73,7 +73,8 @@ class SyncVectorEnv(VectorEnv):
|
||||
self.single_observation_space, n=self.num_envs, fn=np.zeros
|
||||
)
|
||||
self._rewards = np.zeros((self.num_envs,), dtype=np.float64)
|
||||
self._dones = np.zeros((self.num_envs,), dtype=np.bool_)
|
||||
self._terminates = np.zeros((self.num_envs,), dtype=np.bool_)
|
||||
self._truncates = np.zeros((self.num_envs,), dtype=np.bool_)
|
||||
self._actions = None
|
||||
|
||||
def seed(self, seed=None):
|
||||
@@ -99,7 +100,8 @@ class SyncVectorEnv(VectorEnv):
|
||||
seed = [seed + i for i in range(self.num_envs)]
|
||||
assert len(seed) == self.num_envs
|
||||
|
||||
self._dones[:] = False
|
||||
self._terminates[:] = False
|
||||
self._truncates[:] = False
|
||||
observations = []
|
||||
data_list = []
|
||||
for env, single_seed in zip(self.envs, seed):
|
||||
@@ -136,10 +138,13 @@ class SyncVectorEnv(VectorEnv):
|
||||
def step_wait(self):
|
||||
observations, infos = [], []
|
||||
for i, (env, action) in enumerate(zip(self.envs, self._actions)):
|
||||
observation, self._rewards[i], self._dones[i], info = env.step(action)
|
||||
if self._dones[i]:
|
||||
info["terminal_observation"] = observation
|
||||
observation = env.reset()
|
||||
(
|
||||
observation,
|
||||
self._rewards[i],
|
||||
self._terminates[i],
|
||||
self._truncates[i],
|
||||
info,
|
||||
) = env.step(action)
|
||||
observations.append(observation)
|
||||
infos.append(info)
|
||||
self.observations = concatenate(
|
||||
@@ -149,7 +154,8 @@ class SyncVectorEnv(VectorEnv):
|
||||
return (
|
||||
deepcopy(self.observations) if self.copy else self.observations,
|
||||
np.copy(self._rewards),
|
||||
np.copy(self._dones),
|
||||
np.copy(self._terminates),
|
||||
np.copy(self._truncates),
|
||||
infos,
|
||||
)
|
||||
|
||||
|
||||
Vendored
+5
-2
@@ -102,8 +102,11 @@ class VectorEnv(gym.Env):
|
||||
rewards : :obj:`np.ndarray`, dtype :obj:`np.float_`
|
||||
A vector of rewards from the vectorized environment.
|
||||
|
||||
dones : :obj:`np.ndarray`, dtype :obj:`np.bool_`
|
||||
A vector whose entries indicate whether the episode has ended.
|
||||
terminated : :obj:`np.ndarray`, dtype :obj:`np.bool_`
|
||||
A vector whose entries indicate whether the episode has terminated (failed).
|
||||
|
||||
truncated : :obj:`np.ndarray`, dtype :obj:`np.bool_`
|
||||
A vector whose entries indicate whether the episode has been truncated (max episode length).
|
||||
|
||||
infos : list of dict
|
||||
A list of auxiliary diagnostic information dicts from sub-environments.
|
||||
|
||||
Vendored
+3
-1
@@ -1,6 +1,8 @@
|
||||
"""
|
||||
Environment wrapper for D3IL environments with state observations.
|
||||
|
||||
Also return done=False since we do not terminate episode early.
|
||||
|
||||
For consistency, we will use Dict{} for the observation space, with the key "state" for the state observation.
|
||||
"""
|
||||
|
||||
@@ -73,7 +75,7 @@ class D3ilLowdimWrapper(gym.Env):
|
||||
|
||||
# normalize
|
||||
obs = self.normalize_obs(obs)
|
||||
return {"state": obs}, reward, done, info
|
||||
return {"state": obs}, reward, False, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
|
||||
Vendored
+5
-10
@@ -121,7 +121,7 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
|
||||
action = self.normalizer(action, "actions", forward=False)
|
||||
|
||||
# Step the environment n_action_steps times
|
||||
obs, sparse_reward, dense_reward, done, info = self._inner_step(action)
|
||||
obs, sparse_reward, dense_reward, info = self._inner_step(action)
|
||||
if self.sparse_reward:
|
||||
reward = sparse_reward.clone().cpu().numpy()
|
||||
else:
|
||||
@@ -129,17 +129,14 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
|
||||
|
||||
# Only mark the environment as done if it times out, ignore done from inner steps
|
||||
truncated = self.env.env_steps >= self.max_env_steps
|
||||
done = truncated
|
||||
|
||||
nobs: np.ndarray = self.process_obs(obs)
|
||||
done: np.ndarray = done.squeeze().cpu().numpy()
|
||||
truncated: np.ndarray = truncated.squeeze().cpu().numpy()
|
||||
terminated: np.ndarray = np.zeros_like(truncated, dtype=bool)
|
||||
|
||||
return {"state": nobs}, reward, done, info
|
||||
return {"state": nobs}, reward, terminated, truncated, info
|
||||
|
||||
def _inner_step(self, action_chunk: torch.Tensor):
|
||||
dones = torch.zeros(
|
||||
action_chunk.shape[0], dtype=torch.bool, device=action_chunk.device
|
||||
)
|
||||
dense_reward = torch.zeros(action_chunk.shape[0], device=action_chunk.device)
|
||||
sparse_reward = torch.zeros(action_chunk.shape[0], device=action_chunk.device)
|
||||
for i in range(self.n_action_steps):
|
||||
@@ -156,10 +153,8 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
|
||||
# assign "permanent" rewards
|
||||
dense_reward += self.best_reward
|
||||
|
||||
dones = dones | done.squeeze()
|
||||
|
||||
obs = stack_last_n_obs_dict(self.obs, self.n_obs_steps)
|
||||
return obs, sparse_reward, dense_reward, dones, info
|
||||
return obs, sparse_reward, dense_reward, info
|
||||
|
||||
def process_obs(self, obs: torch.Tensor) -> np.ndarray:
|
||||
# Convert the robot state to have 6D pose
|
||||
|
||||
+2
-2
@@ -57,12 +57,12 @@ class MujocoLocomotionLowdimWrapper(gym.Env):
|
||||
def normalize_obs(self, obs):
|
||||
return 2 * ((obs - self.obs_min) / (self.obs_max - self.obs_min + 1e-6) - 0.5)
|
||||
|
||||
def unnormaliza_action(self, action):
|
||||
def unnormalize_action(self, action):
|
||||
action = (action + 1) / 2 # [-1, 1] -> [0, 1]
|
||||
return action * (self.action_max - self.action_min) + self.action_min
|
||||
|
||||
def step(self, action):
|
||||
raw_action = self.unnormaliza_action(action)
|
||||
raw_action = self.unnormalize_action(action)
|
||||
raw_obs, reward, done, info = self.env.step(raw_action)
|
||||
|
||||
# normalize
|
||||
|
||||
Vendored
+25
-9
@@ -138,22 +138,32 @@ class MultiStep(gym.Wrapper):
|
||||
"""
|
||||
if action.ndim == 1: # in case action_steps = 1
|
||||
action = action[None]
|
||||
truncated = False
|
||||
terminated = False
|
||||
for act_step, act in enumerate(action):
|
||||
self.cnt += 1
|
||||
|
||||
if len(self.done) > 0 and self.done[-1]:
|
||||
# termination
|
||||
if terminated or truncated:
|
||||
break
|
||||
|
||||
# done does not differentiate terminal and truncation
|
||||
observation, reward, done, info = self.env.step(act)
|
||||
|
||||
self.obs.append(observation)
|
||||
self.action.append(act)
|
||||
self.reward.append(reward)
|
||||
if (
|
||||
self.max_episode_steps is not None
|
||||
) and self.cnt >= self.max_episode_steps:
|
||||
# truncation
|
||||
done = True
|
||||
|
||||
# in gym, timelimit wrapper is automatically used given env._spec.max_episode_steps
|
||||
if "TimeLimit.truncated" not in info:
|
||||
if done:
|
||||
terminated = True
|
||||
elif (
|
||||
self.max_episode_steps is not None
|
||||
) and self.cnt >= self.max_episode_steps:
|
||||
truncated = True
|
||||
else:
|
||||
truncated = info["TimeLimit.truncated"]
|
||||
terminated = done
|
||||
done = truncated or terminated
|
||||
self.done.append(done)
|
||||
self._add_info(info)
|
||||
observation = self._get_obs(self.n_obs_steps)
|
||||
@@ -165,6 +175,12 @@ class MultiStep(gym.Wrapper):
|
||||
|
||||
# In mujoco case, done can happen within the loop above
|
||||
if self.reset_within_step and self.done[-1]:
|
||||
|
||||
# need to save old observation in the case of truncation only, for bootstrapping
|
||||
if truncated:
|
||||
info["final_obs"] = observation
|
||||
|
||||
# reset
|
||||
observation = (
|
||||
self.reset()
|
||||
) # TODO: arguments? this cannot handle video recording right now since needs to pass in options
|
||||
@@ -173,7 +189,7 @@ class MultiStep(gym.Wrapper):
|
||||
# reset reward and done for next step
|
||||
self.reward = list()
|
||||
self.done = list()
|
||||
return observation, reward, done, info
|
||||
return observation, reward, terminated, truncated, info
|
||||
|
||||
def _get_obs(self, n_steps=1):
|
||||
"""
|
||||
|
||||
+3
-1
@@ -1,6 +1,8 @@
|
||||
"""
|
||||
Environment wrapper for Robomimic environments with image observations.
|
||||
|
||||
Also return done=False since we do not terminate episode early.
|
||||
|
||||
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/env/robomimic/robomimic_image_wrapper.py
|
||||
|
||||
"""
|
||||
@@ -158,7 +160,7 @@ class RobomimicImageWrapper(gym.Env):
|
||||
video_img = self.render(mode="rgb_array")
|
||||
self.video_writer.append_data(video_img)
|
||||
|
||||
return obs, reward, done, info
|
||||
return obs, reward, False, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
|
||||
+4
-2
@@ -1,6 +1,8 @@
|
||||
"""
|
||||
Environment wrapper for Robomimic environments with state observations.
|
||||
|
||||
Also return done=False since we do not terminate episode early.
|
||||
|
||||
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/env/robomimic/robomimic_lowdim_wrapper.py
|
||||
|
||||
For consistency, we will use Dict{} for the observation space, with the key "state" for the state observation.
|
||||
@@ -65,7 +67,7 @@ class RobomimicLowdimWrapper(gym.Env):
|
||||
low=low,
|
||||
high=high,
|
||||
shape=low.shape,
|
||||
dtype=low.dtype,
|
||||
dtype=np.float32,
|
||||
)
|
||||
|
||||
def normalize_obs(self, obs):
|
||||
@@ -129,7 +131,7 @@ class RobomimicLowdimWrapper(gym.Env):
|
||||
video_img = self.render(mode="rgb_array")
|
||||
self.video_writer.append_data(video_img)
|
||||
|
||||
return obs, reward, done, info
|
||||
return obs, reward, False, info
|
||||
|
||||
def render(self, mode="rgb_array"):
|
||||
h, w = self.render_hw
|
||||
|
||||
Reference in New Issue
Block a user