v0.5 to main (#10)

* v0.5 (#9)

* update idql configs

* update awr configs

* update dipo configs

* update qsm configs

* update dqm configs

* update project version to 0.5.0
This commit is contained in:
Allen Z. Ren
2024-10-07 16:35:13 -04:00
committed by GitHub
parent dd14c5887c
commit e0842e71dc
267 changed files with 6769 additions and 1645 deletions
+8 -6
View File
@@ -184,8 +184,8 @@ def make_async(
# Create a fake env whose sole purpose is to provide
# obs/action spaces and metadata.
env = gym.Env()
observation_space = spaces.Dict()
if shape_meta is not None: # rn only for images
observation_space = spaces.Dict()
for key, value in shape_meta["obs"].items():
shape = value["shape"]
if key.endswith("rgb"):
@@ -194,18 +194,20 @@ def make_async(
min_value, max_value = -1, 1
else:
raise RuntimeError(f"Unsupported type {key}")
this_space = spaces.Box(
observation_space[key] = spaces.Box(
low=min_value,
high=max_value,
shape=shape,
dtype=np.float32,
)
observation_space[key] = this_space
env.observation_space = observation_space
else:
env.observation_space = gym.spaces.Box(
-1, 1, shape=(obs_dim,), dtype=np.float64
observation_space["state"] = gym.spaces.Box(
-1,
1,
shape=(obs_dim,),
dtype=np.float32,
)
env.observation_space = observation_space
env.action_space = gym.spaces.Box(-1, 1, shape=(action_dim,), dtype=np.int64)
env.metadata = {
"render.modes": ["human", "rgb_array", "depth_array"],
+16 -15
View File
@@ -1,7 +1,10 @@
"""
From gym==0.22.0
Disable auto-reset after done.
Use terminated/truncated instead of done.
Disable auto-reset after done. Reset in MultiStepWrapper instead.
Add reset_arg() that allows all environments with different options.
Add reset_one_arg() that allows resetting a single environment with options.
Add render().
@@ -398,8 +401,11 @@ class AsyncVectorEnv(VectorEnv):
rewards : :obj:`np.ndarray`, dtype :obj:`np.float_`
A vector of rewards from the vectorized environment.
dones : :obj:`np.ndarray`, dtype :obj:`np.bool_`
A vector whose entries indicate whether the episode has ended.
terminates : :obj:`np.ndarray`, dtype :obj:`np.bool_`
A vector whose entries indicate whether the episode has terminated (failed).
truncates : :obj:`np.ndarray`, dtype :obj:`np.bool_`
A vector whose entries indicate whether the episode has been truncated (max episode length).
infos : list of dict
A list of auxiliary diagnostic information dicts from sub-environments.
@@ -432,7 +438,7 @@ class AsyncVectorEnv(VectorEnv):
results, successes = zip(*[pipe.recv() for pipe in self.parent_pipes])
self._raise_if_errors(successes)
self._state = AsyncState.DEFAULT
observations_list, rewards, dones, infos = zip(*results)
observations_list, rewards, terminates, truncates, infos = zip(*results)
if not self.shared_memory:
self.observations = concatenate(
@@ -444,7 +450,8 @@ class AsyncVectorEnv(VectorEnv):
return (
deepcopy(self.observations) if self.copy else self.observations,
np.array(rewards),
np.array(dones, dtype=np.bool_),
np.array(terminates, dtype=np.bool_),
np.array(truncates, dtype=np.bool_),
infos,
)
@@ -717,11 +724,8 @@ def _worker(index, env_fn, pipe, parent_pipe, shared_memory, error_queue):
pipe.send((observation, True))
elif command == "step":
observation, reward, done, info = env.step(data)
# if done:
# info["terminal_observation"] = observation
# observation = env.reset()
pipe.send(((observation, reward, done, info), True))
observation, reward, terminated, truncated, info = env.step(data)
pipe.send(((observation, reward, terminated, truncated, info), True))
elif command == "seed":
env.seed(data)
pipe.send((None, True))
@@ -789,14 +793,11 @@ def _worker_shared_memory(index, env_fn, pipe, parent_pipe, shared_memory, error
)
pipe.send((None, True))
elif command == "step":
observation, reward, done, info = env.step(data)
# if done:
# info["terminal_observation"] = observation
# observation = env.reset()
observation, reward, terminated, truncated, info = env.step(data)
write_to_shared_memory(
observation_space, index, observation, shared_memory
)
pipe.send(((None, reward, done, info), True))
pipe.send(((None, reward, terminated, truncated, info), True))
elif command == "seed":
env.seed(data)
pipe.send((None, True))
+13 -7
View File
@@ -73,7 +73,8 @@ class SyncVectorEnv(VectorEnv):
self.single_observation_space, n=self.num_envs, fn=np.zeros
)
self._rewards = np.zeros((self.num_envs,), dtype=np.float64)
self._dones = np.zeros((self.num_envs,), dtype=np.bool_)
self._terminates = np.zeros((self.num_envs,), dtype=np.bool_)
self._truncates = np.zeros((self.num_envs,), dtype=np.bool_)
self._actions = None
def seed(self, seed=None):
@@ -99,7 +100,8 @@ class SyncVectorEnv(VectorEnv):
seed = [seed + i for i in range(self.num_envs)]
assert len(seed) == self.num_envs
self._dones[:] = False
self._terminates[:] = False
self._truncates[:] = False
observations = []
data_list = []
for env, single_seed in zip(self.envs, seed):
@@ -136,10 +138,13 @@ class SyncVectorEnv(VectorEnv):
def step_wait(self):
observations, infos = [], []
for i, (env, action) in enumerate(zip(self.envs, self._actions)):
observation, self._rewards[i], self._dones[i], info = env.step(action)
if self._dones[i]:
info["terminal_observation"] = observation
observation = env.reset()
(
observation,
self._rewards[i],
self._terminates[i],
self._truncates[i],
info,
) = env.step(action)
observations.append(observation)
infos.append(info)
self.observations = concatenate(
@@ -149,7 +154,8 @@ class SyncVectorEnv(VectorEnv):
return (
deepcopy(self.observations) if self.copy else self.observations,
np.copy(self._rewards),
np.copy(self._dones),
np.copy(self._terminates),
np.copy(self._truncates),
infos,
)
+5 -2
View File
@@ -102,8 +102,11 @@ class VectorEnv(gym.Env):
rewards : :obj:`np.ndarray`, dtype :obj:`np.float_`
A vector of rewards from the vectorized environment.
dones : :obj:`np.ndarray`, dtype :obj:`np.bool_`
A vector whose entries indicate whether the episode has ended.
terminated : :obj:`np.ndarray`, dtype :obj:`np.bool_`
A vector whose entries indicate whether the episode has terminated (failed).
truncated : :obj:`np.ndarray`, dtype :obj:`np.bool_`
A vector whose entries indicate whether the episode has been truncated (max episode length).
infos : list of dict
A list of auxiliary diagnostic information dicts from sub-environments.
+3 -1
View File
@@ -1,6 +1,8 @@
"""
Environment wrapper for D3IL environments with state observations.
Also return done=False since we do not terminate episode early.
For consistency, we will use Dict{} for the observation space, with the key "state" for the state observation.
"""
@@ -73,7 +75,7 @@ class D3ilLowdimWrapper(gym.Env):
# normalize
obs = self.normalize_obs(obs)
return {"state": obs}, reward, done, info
return {"state": obs}, reward, False, info
def render(self, mode="rgb_array"):
h, w = self.render_hw
+5 -10
View File
@@ -121,7 +121,7 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
action = self.normalizer(action, "actions", forward=False)
# Step the environment n_action_steps times
obs, sparse_reward, dense_reward, done, info = self._inner_step(action)
obs, sparse_reward, dense_reward, info = self._inner_step(action)
if self.sparse_reward:
reward = sparse_reward.clone().cpu().numpy()
else:
@@ -129,17 +129,14 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
# Only mark the environment as done if it times out, ignore done from inner steps
truncated = self.env.env_steps >= self.max_env_steps
done = truncated
nobs: np.ndarray = self.process_obs(obs)
done: np.ndarray = done.squeeze().cpu().numpy()
truncated: np.ndarray = truncated.squeeze().cpu().numpy()
terminated: np.ndarray = np.zeros_like(truncated, dtype=bool)
return {"state": nobs}, reward, done, info
return {"state": nobs}, reward, terminated, truncated, info
def _inner_step(self, action_chunk: torch.Tensor):
dones = torch.zeros(
action_chunk.shape[0], dtype=torch.bool, device=action_chunk.device
)
dense_reward = torch.zeros(action_chunk.shape[0], device=action_chunk.device)
sparse_reward = torch.zeros(action_chunk.shape[0], device=action_chunk.device)
for i in range(self.n_action_steps):
@@ -156,10 +153,8 @@ class FurnitureRLSimEnvMultiStepWrapper(gym.Wrapper):
# assign "permanent" rewards
dense_reward += self.best_reward
dones = dones | done.squeeze()
obs = stack_last_n_obs_dict(self.obs, self.n_obs_steps)
return obs, sparse_reward, dense_reward, dones, info
return obs, sparse_reward, dense_reward, info
def process_obs(self, obs: torch.Tensor) -> np.ndarray:
# Convert the robot state to have 6D pose
+2 -2
View File
@@ -57,12 +57,12 @@ class MujocoLocomotionLowdimWrapper(gym.Env):
def normalize_obs(self, obs):
return 2 * ((obs - self.obs_min) / (self.obs_max - self.obs_min + 1e-6) - 0.5)
def unnormaliza_action(self, action):
def unnormalize_action(self, action):
action = (action + 1) / 2 # [-1, 1] -> [0, 1]
return action * (self.action_max - self.action_min) + self.action_min
def step(self, action):
raw_action = self.unnormaliza_action(action)
raw_action = self.unnormalize_action(action)
raw_obs, reward, done, info = self.env.step(raw_action)
# normalize
+25 -9
View File
@@ -138,22 +138,32 @@ class MultiStep(gym.Wrapper):
"""
if action.ndim == 1: # in case action_steps = 1
action = action[None]
truncated = False
terminated = False
for act_step, act in enumerate(action):
self.cnt += 1
if len(self.done) > 0 and self.done[-1]:
# termination
if terminated or truncated:
break
# done does not differentiate terminal and truncation
observation, reward, done, info = self.env.step(act)
self.obs.append(observation)
self.action.append(act)
self.reward.append(reward)
if (
self.max_episode_steps is not None
) and self.cnt >= self.max_episode_steps:
# truncation
done = True
# in gym, timelimit wrapper is automatically used given env._spec.max_episode_steps
if "TimeLimit.truncated" not in info:
if done:
terminated = True
elif (
self.max_episode_steps is not None
) and self.cnt >= self.max_episode_steps:
truncated = True
else:
truncated = info["TimeLimit.truncated"]
terminated = done
done = truncated or terminated
self.done.append(done)
self._add_info(info)
observation = self._get_obs(self.n_obs_steps)
@@ -165,6 +175,12 @@ class MultiStep(gym.Wrapper):
# In mujoco case, done can happen within the loop above
if self.reset_within_step and self.done[-1]:
# need to save old observation in the case of truncation only, for bootstrapping
if truncated:
info["final_obs"] = observation
# reset
observation = (
self.reset()
) # TODO: arguments? this cannot handle video recording right now since needs to pass in options
@@ -173,7 +189,7 @@ class MultiStep(gym.Wrapper):
# reset reward and done for next step
self.reward = list()
self.done = list()
return observation, reward, done, info
return observation, reward, terminated, truncated, info
def _get_obs(self, n_steps=1):
"""
+3 -1
View File
@@ -1,6 +1,8 @@
"""
Environment wrapper for Robomimic environments with image observations.
Also return done=False since we do not terminate episode early.
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/env/robomimic/robomimic_image_wrapper.py
"""
@@ -158,7 +160,7 @@ class RobomimicImageWrapper(gym.Env):
video_img = self.render(mode="rgb_array")
self.video_writer.append_data(video_img)
return obs, reward, done, info
return obs, reward, False, info
def render(self, mode="rgb_array"):
h, w = self.render_hw
+4 -2
View File
@@ -1,6 +1,8 @@
"""
Environment wrapper for Robomimic environments with state observations.
Also return done=False since we do not terminate episode early.
Modified from https://github.com/real-stanford/diffusion_policy/blob/main/diffusion_policy/env/robomimic/robomimic_lowdim_wrapper.py
For consistency, we will use Dict{} for the observation space, with the key "state" for the state observation.
@@ -65,7 +67,7 @@ class RobomimicLowdimWrapper(gym.Env):
low=low,
high=high,
shape=low.shape,
dtype=low.dtype,
dtype=np.float32,
)
def normalize_obs(self, obs):
@@ -129,7 +131,7 @@ class RobomimicLowdimWrapper(gym.Env):
video_img = self.render(mode="rgb_array")
self.video_writer.append_data(video_img)
return obs, reward, done, info
return obs, reward, False, info
def render(self, mode="rgb_array"):
h, w = self.render_hw