v0.5 to main (#10)
* v0.5 (#9) * update idql configs * update awr configs * update dipo configs * update qsm configs * update dqm configs * update project version to 0.5.0
This commit is contained in:
@@ -1,3 +1,18 @@
|
||||
## Data processing scripts
|
||||
|
||||
These are some scripts used for processing the raw datasets from the benchmarks. We already pre-processed them and provide the final datasets. These scripts are for information only.
|
||||
These are some scripts used for processing the raw datasets from the benchmarks. We already pre-processed them and provide the final datasets.
|
||||
|
||||
Gym and robomimic data
|
||||
```console
|
||||
python script/dataset/get_d4rl_dataset.py --env_name=hopper-medium-v2 --save_dir=data/gym/hopper-medium-v2
|
||||
python script/dataset/process_robomimic_dataset.py --load_path=../robomimic_raw_data/lift_low_dim_v141.hdf5 --save_dir=data/robomimic/lift --normalize
|
||||
```
|
||||
|
||||
Raw robomimic data can be downloaded with a clone of the repository and then
|
||||
```console
|
||||
cd ~/robomimic/robomimic/scripts
|
||||
python download_datasets.py --tasks all --dataset_types mh --hdf5_types low_dim # state-only policy
|
||||
python download_datasets.py --tasks all --dataset_types mh --hdf5_types raw # pixel-based policy
|
||||
# for pixel, replay the trajectories to extract image observations
|
||||
python robomimic/scripts/dataset_states_to_obs.py --done_mode 2 --dataset datasets/can/mh/demo_v141.hdf5 --output_name image_v141.hdf5 --camera_names robot0_eye_in_hand --camera_height 96 --camera_width 96 --exclude-next-obs --n 100
|
||||
```
|
||||
+106
-148
@@ -1,95 +1,79 @@
|
||||
"""
|
||||
Download D4RL dataset and save it into our custom format so it can be loaded for diffusion training.
|
||||
|
||||
Download D4RL dataset and save it into our custom format for diffusion training.
|
||||
"""
|
||||
|
||||
import os
|
||||
import logging
|
||||
import gym
|
||||
import random
|
||||
from copy import deepcopy
|
||||
import numpy as np
|
||||
from tqdm import tqdm
|
||||
import pickle
|
||||
|
||||
import d4rl.gym_mujoco # Import required to register environments
|
||||
from copy import deepcopy
|
||||
|
||||
|
||||
def make_dataset(env_name, save_dir, save_name_prefix, val_split, logger):
|
||||
# Create the environment
|
||||
env = gym.make(env_name)
|
||||
|
||||
# d4rl abides by the OpenAI gym interface
|
||||
env.reset()
|
||||
env.step(env.action_space.sample())
|
||||
|
||||
# Each task is associated with a dataset
|
||||
# dataset contains observations, actions, rewards, terminals, and infos
|
||||
env.step(
|
||||
env.action_space.sample()
|
||||
) # Interact with the environment to initialize it
|
||||
dataset = env.get_dataset()
|
||||
|
||||
# rename observations to states
|
||||
dataset["states"] = dataset.pop("observations")
|
||||
|
||||
logger.info("\n========== Basic Info ===========")
|
||||
logger.info(f"Keys in the dataset: {dataset.keys()}")
|
||||
logger.info(f"Observation shape: {dataset['observations'].shape}")
|
||||
logger.info(f"State shape: {dataset['states'].shape}")
|
||||
logger.info(f"Action shape: {dataset['actions'].shape}")
|
||||
|
||||
# determine trajectories from terminals and timeouts
|
||||
terminal_indices = np.argwhere(dataset["terminals"])[:, 0]
|
||||
timeout_indices = np.argwhere(dataset["timeouts"])[:, 0]
|
||||
obs_dim = dataset["observations"].shape[1]
|
||||
action_dim = dataset["actions"].shape[1]
|
||||
done_indices = np.concatenate([terminal_indices, timeout_indices])
|
||||
done_indices = np.sort(done_indices)
|
||||
traj_lengths = []
|
||||
prev_index = 0
|
||||
for i in tqdm(range(len(done_indices))):
|
||||
# get episode length
|
||||
cur_index = done_indices[i]
|
||||
traj_lengths.append(cur_index - prev_index + 1)
|
||||
prev_index = cur_index + 1
|
||||
obs_min = np.min(dataset["observations"], axis=0)
|
||||
obs_max = np.max(dataset["observations"], axis=0)
|
||||
done_indices = np.sort(np.concatenate([terminal_indices, timeout_indices]))
|
||||
traj_lengths = np.diff(np.concatenate([[0], done_indices + 1]))
|
||||
|
||||
obs_min = np.min(dataset["states"], axis=0)
|
||||
obs_max = np.max(dataset["states"], axis=0)
|
||||
action_min = np.min(dataset["actions"], axis=0)
|
||||
action_max = np.max(dataset["actions"], axis=0)
|
||||
max_episode_steps = max(traj_lengths)
|
||||
logger.info("total transitions: {}".format(np.sum(traj_lengths)))
|
||||
logger.info("total trajectories: {}".format(len(traj_lengths)))
|
||||
logger.info(
|
||||
f"traj length mean/std: {np.mean(traj_lengths)}, {np.std(traj_lengths)}"
|
||||
)
|
||||
logger.info(f"traj length min/max: {np.min(traj_lengths)}, {np.max(traj_lengths)}")
|
||||
logger.info(f"obs min: {obs_min}")
|
||||
logger.info(f"obs max: {obs_max}")
|
||||
logger.info(f"action min: {action_min}")
|
||||
logger.info(f"action max: {action_max}")
|
||||
|
||||
# Subsample episodes by taking the first ones
|
||||
logger.info(f"Total transitions: {np.sum(traj_lengths)}")
|
||||
logger.info(f"Total trajectories: {len(traj_lengths)}")
|
||||
logger.info(
|
||||
f"Trajectory length mean/std: {np.mean(traj_lengths)}, {np.std(traj_lengths)}"
|
||||
)
|
||||
logger.info(
|
||||
f"Trajectory length min/max: {np.min(traj_lengths)}, {np.max(traj_lengths)}"
|
||||
)
|
||||
logger.info(f"obs min: {obs_min}, obs max: {obs_max}")
|
||||
logger.info(f"action min: {action_min}, action max: {action_max}")
|
||||
|
||||
# Subsample episodes if needed
|
||||
if args.max_episodes > 0:
|
||||
traj_lengths = traj_lengths[: args.max_episodes]
|
||||
done_indices = done_indices[: args.max_episodes]
|
||||
max_episode_steps = max(traj_lengths)
|
||||
|
||||
# split indices in train and val
|
||||
# Split into train and validation sets
|
||||
num_traj = len(traj_lengths)
|
||||
num_train = int(num_traj * (1 - val_split))
|
||||
train_indices = random.sample(range(num_traj), k=num_train)
|
||||
|
||||
# do over all indices
|
||||
out_train = {}
|
||||
keys = [
|
||||
"observations",
|
||||
"actions",
|
||||
"rewards",
|
||||
]
|
||||
out_train["observations"] = np.empty(
|
||||
(0, max_episode_steps, dataset["observations"].shape[-1])
|
||||
)
|
||||
out_train["actions"] = np.empty(
|
||||
(0, max_episode_steps, dataset["actions"].shape[-1])
|
||||
)
|
||||
out_train["rewards"] = np.empty((0, max_episode_steps))
|
||||
out_train["traj_length"] = []
|
||||
# Prepare data containers for train and validation sets
|
||||
out_train = {
|
||||
"states": [],
|
||||
"actions": [],
|
||||
"rewards": [],
|
||||
"terminals": [],
|
||||
"traj_lengths": [],
|
||||
}
|
||||
out_val = deepcopy(out_train)
|
||||
prev_index = 0
|
||||
train_episode_reward_all = []
|
||||
val_episode_reward_all = []
|
||||
for i in tqdm(range(len(done_indices))):
|
||||
for i, cur_index in tqdm(enumerate(done_indices), total=len(done_indices)):
|
||||
if i in train_indices:
|
||||
out = out_train
|
||||
episode_reward_all = train_episode_reward_all
|
||||
@@ -97,57 +81,65 @@ def make_dataset(env_name, save_dir, save_name_prefix, val_split, logger):
|
||||
out = out_val
|
||||
episode_reward_all = val_episode_reward_all
|
||||
|
||||
# get episode length
|
||||
cur_index = done_indices[i]
|
||||
# Get the trajectory length and slice
|
||||
traj_length = cur_index - prev_index + 1
|
||||
trajectory = {
|
||||
key: dataset[key][prev_index : cur_index + 1]
|
||||
for key in ["states", "actions", "rewards", "terminals"]
|
||||
}
|
||||
|
||||
# Skip if the episode has no reward
|
||||
if np.sum(dataset["rewards"][prev_index : cur_index + 1]) > 0:
|
||||
out["traj_length"].append(traj_length)
|
||||
# Skip if there is no reward in the episode
|
||||
if np.sum(trajectory["rewards"]) > 0:
|
||||
# Scale observations and actions
|
||||
trajectory["states"] = (
|
||||
2 * (trajectory["states"] - obs_min) / (obs_max - obs_min + 1e-6) - 1
|
||||
)
|
||||
trajectory["actions"] = (
|
||||
2
|
||||
* (trajectory["actions"] - action_min)
|
||||
/ (action_max - action_min + 1e-6)
|
||||
- 1
|
||||
)
|
||||
|
||||
# apply padding to make all episodes have the same max steps
|
||||
for key in keys:
|
||||
traj = dataset[key][prev_index : cur_index + 1]
|
||||
|
||||
# also scale
|
||||
if key == "observations":
|
||||
traj = 2 * (traj - obs_min) / (obs_max - obs_min + 1e-6) - 1
|
||||
elif key == "actions":
|
||||
traj = (
|
||||
2 * (traj - action_min) / (action_max - action_min + 1e-6) - 1
|
||||
)
|
||||
|
||||
if traj.ndim == 1:
|
||||
traj = np.pad(
|
||||
traj,
|
||||
(0, max_episode_steps - len(traj)),
|
||||
mode="constant",
|
||||
constant_values=0,
|
||||
)
|
||||
else:
|
||||
traj = np.pad(
|
||||
traj,
|
||||
((0, max_episode_steps - traj.shape[0]), (0, 0)),
|
||||
mode="constant",
|
||||
constant_values=0,
|
||||
)
|
||||
out[key] = np.vstack((out[key], traj[None]))
|
||||
|
||||
# check reward
|
||||
episode_reward_all.append(np.sum(out["rewards"][-1]))
|
||||
for key in ["states", "actions", "rewards", "terminals"]:
|
||||
out[key].append(trajectory[key])
|
||||
out["traj_lengths"].append(traj_length)
|
||||
episode_reward_all.append(np.sum(trajectory["rewards"]))
|
||||
else:
|
||||
print(f"skipping {i} / {len(done_indices)}")
|
||||
logger.info(f"Skipping trajectory {i} due to zero rewards.")
|
||||
|
||||
# update prev index
|
||||
prev_index = cur_index + 1
|
||||
|
||||
# Save to np file
|
||||
save_train_path = os.path.join(save_dir, save_name_prefix + "train.npz")
|
||||
save_val_path = os.path.join(save_dir, save_name_prefix + "val.npz")
|
||||
with open(save_train_path, "wb") as f:
|
||||
pickle.dump(out_train, f)
|
||||
with open(save_val_path, "wb") as f:
|
||||
pickle.dump(out_val, f)
|
||||
# Concatenate trajectories
|
||||
for key in ["states", "actions", "rewards", "terminals"]:
|
||||
out_train[key] = np.concatenate(out_train[key], axis=0)
|
||||
|
||||
# Only concatenate validation set if it exists
|
||||
if val_split > 0:
|
||||
out_val[key] = np.concatenate(out_val[key], axis=0)
|
||||
|
||||
# Save train dataset to npz files
|
||||
train_save_path = os.path.join(save_dir, save_name_prefix + "train.npz")
|
||||
np.savez_compressed(
|
||||
train_save_path,
|
||||
states=np.array(out_train["states"]),
|
||||
actions=np.array(out_train["actions"]),
|
||||
rewards=np.array(out_train["rewards"]),
|
||||
terminals=np.array(out_train["terminals"]),
|
||||
traj_lengths=np.array(out_train["traj_lengths"]),
|
||||
)
|
||||
|
||||
# Save validation dataset to npz files
|
||||
val_save_path = os.path.join(save_dir, save_name_prefix + "val.npz")
|
||||
np.savez_compressed(
|
||||
val_save_path,
|
||||
states=np.array(out_val["states"]),
|
||||
actions=np.array(out_val["actions"]),
|
||||
rewards=np.array(out_val["rewards"]),
|
||||
terminals=np.array(out_val["terminals"]),
|
||||
traj_lengths=np.array(out_val["traj_lengths"]),
|
||||
)
|
||||
|
||||
normalization_save_path = os.path.join(
|
||||
save_dir, save_name_prefix + "normalization.npz"
|
||||
)
|
||||
@@ -159,83 +151,49 @@ def make_dataset(env_name, save_dir, save_name_prefix, val_split, logger):
|
||||
action_max=action_max,
|
||||
)
|
||||
|
||||
# debug
|
||||
# Logging summary statistics
|
||||
logger.info("\n========== Final ===========")
|
||||
logger.info(
|
||||
f"Train - Number of episodes and transitions: {len(out_train['traj_length'])}, {np.sum(out_train['traj_length'])}"
|
||||
f"Train - Trajectories: {len(out_train['traj_lengths'])}, Transitions: {np.sum(out_train['traj_lengths'])}"
|
||||
)
|
||||
logger.info(
|
||||
f"Val - Number of episodes and transitions: {len(out_val['traj_length'])}, {np.sum(out_val['traj_length'])}"
|
||||
f"Val - Trajectories: {len(out_val['traj_lengths'])}, Transitions: {np.sum(out_val['traj_lengths'])}"
|
||||
)
|
||||
logger.info(
|
||||
f"Train - Mean/Std trajectory length: {np.mean(out_train['traj_length'])}, {np.std(out_train['traj_length'])}"
|
||||
f"Train - Mean/Std trajectory length: {np.mean(out_train['traj_lengths'])}, {np.std(out_train['traj_lengths'])}"
|
||||
)
|
||||
logger.info(
|
||||
f"Train - Max/Min trajectory length: {np.max(out_train['traj_length'])}, {np.min(out_train['traj_length'])}"
|
||||
(
|
||||
logger.info(
|
||||
f"Val - Mean/Std trajectory length: {np.mean(out_val['traj_lengths'])}, {np.std(out_val['traj_lengths'])}"
|
||||
)
|
||||
if val_split > 0
|
||||
else None
|
||||
)
|
||||
if val_split > 0:
|
||||
logger.info(
|
||||
f"Val - Mean/Std trajectory length: {np.mean(out_val['traj_length'])}, {np.std(out_val['traj_length'])}"
|
||||
)
|
||||
logger.info(
|
||||
f"Val - Max/Min trajectory length: {np.max(out_val['traj_length'])}, {np.min(out_val['traj_length'])}"
|
||||
)
|
||||
logger.info(
|
||||
f"Train - Mean/Std episode reward: {np.mean(train_episode_reward_all)}, {np.std(train_episode_reward_all)}"
|
||||
)
|
||||
if val_split > 0:
|
||||
logger.info(
|
||||
f"Val - Mean/Std episode reward: {np.mean(val_episode_reward_all)}, {np.std(val_episode_reward_all)}"
|
||||
)
|
||||
for obs_dim_ind in range(obs_dim):
|
||||
obs = out_train["observations"][:, :, obs_dim_ind]
|
||||
logger.info(
|
||||
f"Train - Obs dim {obs_dim_ind+1} mean {np.mean(obs)} std {np.std(obs)} min {np.min(obs)} max {np.max(obs)}"
|
||||
)
|
||||
for action_dim_ind in range(action_dim):
|
||||
action = out_train["actions"][:, :, action_dim_ind]
|
||||
logger.info(
|
||||
f"Train - Action dim {action_dim_ind+1} mean {np.mean(action)} std {np.std(action)} min {np.min(action)} max {np.max(action)}"
|
||||
)
|
||||
if val_split > 0:
|
||||
for obs_dim_ind in range(obs_dim):
|
||||
obs = out_val["observations"][:, :, obs_dim_ind]
|
||||
logger.info(
|
||||
f"Val - Obs dim {obs_dim_ind+1} mean {np.mean(obs)} std {np.std(obs)} min {np.min(obs)} max {np.max(obs)}"
|
||||
)
|
||||
for action_dim_ind in range(action_dim):
|
||||
action = out_val["actions"][:, :, action_dim_ind]
|
||||
logger.info(
|
||||
f"Val - Action dim {action_dim_ind+1} mean {np.mean(action)} std {np.std(action)} min {np.min(action)} max {np.max(action)}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
import datetime
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--env_name", type=str, default="hopper-medium-v2")
|
||||
parser.add_argument("--save_dir", type=str, default=".")
|
||||
parser.add_argument("--save_name_prefix", type=str, default="")
|
||||
parser.add_argument("--val_split", type=float, default="0.2")
|
||||
parser.add_argument("--max_episodes", type=int, default="-1")
|
||||
parser.add_argument("--val_split", type=float, default=0)
|
||||
parser.add_argument("--max_episodes", type=int, default=-1)
|
||||
args = parser.parse_args()
|
||||
|
||||
import datetime
|
||||
|
||||
# import logging.config
|
||||
if args.max_episodes > 0:
|
||||
args.save_name_prefix += f"max_episodes_{args.max_episodes}_"
|
||||
os.makedirs(args.save_dir, exist_ok=True)
|
||||
log_path = os.path.join(
|
||||
args.save_dir,
|
||||
args.save_name_prefix
|
||||
+ f"_{datetime.datetime.now().strftime('%Y_%m_%d_%H_%M_%S')}.log",
|
||||
)
|
||||
|
||||
logger = logging.getLogger("get_D4RL_dataset")
|
||||
logger.setLevel(logging.INFO)
|
||||
file_handler = logging.FileHandler(log_path)
|
||||
file_handler.setLevel(logging.INFO) # Set the minimum level for this handler
|
||||
file_handler.setLevel(logging.INFO)
|
||||
formatter = logging.Formatter(
|
||||
"%(asctime)s - %(name)s - %(levelname)s - %(message)s"
|
||||
)
|
||||
|
||||
@@ -3,6 +3,8 @@ Process robomimic dataset and save it into our custom format so it can be loaded
|
||||
|
||||
Using some code from robomimic/robomimic/scripts/get_dataset_info.py
|
||||
|
||||
Since we do not terminate episode early and cumulate reward when the goal is reached, we set terminals to all False.
|
||||
|
||||
can-mh:
|
||||
total transitions: 62756
|
||||
total trajectories: 300
|
||||
@@ -76,31 +78,20 @@ robomimic dataset normalizes action to [-1, 1], observation roughly? to [-1, 1].
|
||||
|
||||
"""
|
||||
|
||||
|
||||
import numpy as np
|
||||
from tqdm import tqdm
|
||||
import pickle
|
||||
|
||||
try:
|
||||
import h5py # not included in pyproject.toml
|
||||
except:
|
||||
print("Installing h5py")
|
||||
os.system("pip install h5py")
|
||||
import h5py
|
||||
import os
|
||||
import random
|
||||
from copy import deepcopy
|
||||
import logging
|
||||
|
||||
|
||||
def make_dataset(
|
||||
load_path,
|
||||
save_dir,
|
||||
save_name_prefix,
|
||||
val_split,
|
||||
normalize,
|
||||
):
|
||||
def make_dataset(load_path, save_dir, save_name_prefix, val_split, normalize):
|
||||
# Load hdf5 file from load_path
|
||||
with h5py.File(load_path, "r") as f:
|
||||
# put demonstration list in increasing episode order
|
||||
# Sort demonstrations in increasing episode order
|
||||
demos = sorted(list(f["data"].keys()))
|
||||
inds = np.argsort([int(elem[5:]) for elem in demos])
|
||||
demos = [demos[i] for i in inds]
|
||||
@@ -108,7 +99,7 @@ def make_dataset(
|
||||
if args.max_episodes > 0:
|
||||
demos = demos[: args.max_episodes]
|
||||
|
||||
# From generate_paper_configs.py: default observation is eef pose, gripper finger position, and object information, all of which are low-dim.
|
||||
# Default low-dimensional observation keys
|
||||
low_dim_obs_names = [
|
||||
"robot0_eef_pos",
|
||||
"robot0_eef_quat",
|
||||
@@ -120,23 +111,28 @@ def make_dataset(
|
||||
"robot1_eef_quat",
|
||||
"robot1_gripper_qpos",
|
||||
]
|
||||
if args.cameras is None: # state-only
|
||||
if args.cameras is None:
|
||||
low_dim_obs_names.append("object")
|
||||
|
||||
# Calculate dimensions for observations and actions
|
||||
obs_dim = 0
|
||||
for low_dim_obs_name in low_dim_obs_names:
|
||||
dim = f["data/demo_0/obs/{}".format(low_dim_obs_name)].shape[1]
|
||||
dim = f[f"data/demo_0/obs/{low_dim_obs_name}"].shape[1]
|
||||
obs_dim += dim
|
||||
logging.info(f"Using {low_dim_obs_name} with dim {dim} for observation")
|
||||
|
||||
action_dim = f["data/demo_0/actions"].shape[1]
|
||||
logging.info(f"Total low-dim observation dim: {obs_dim}")
|
||||
logging.info(f"Action dim: {action_dim}")
|
||||
|
||||
# get basic stats
|
||||
# Initialize variables for tracking trajectory statistics
|
||||
traj_lengths = []
|
||||
obs_min = np.zeros((obs_dim))
|
||||
obs_max = np.zeros((obs_dim))
|
||||
action_min = np.zeros((action_dim))
|
||||
action_max = np.zeros((action_dim))
|
||||
|
||||
# Process each demo
|
||||
for ep in demos:
|
||||
traj_lengths.append(f[f"data/{ep}/actions"].shape[0])
|
||||
obs = np.hstack(
|
||||
@@ -145,96 +141,47 @@ def make_dataset(
|
||||
for low_dim_obs_name in low_dim_obs_names
|
||||
]
|
||||
)
|
||||
actions = f[f"data/{ep}/actions"]
|
||||
actions = f[f"data/{ep}/actions"][()]
|
||||
obs_min = np.minimum(obs_min, np.min(obs, axis=0))
|
||||
obs_max = np.maximum(obs_max, np.max(obs, axis=0))
|
||||
action_min = np.minimum(action_min, np.min(actions, axis=0))
|
||||
action_max = np.maximum(action_max, np.max(actions, axis=0))
|
||||
traj_lengths = np.array(traj_lengths)
|
||||
max_traj_length = np.max(traj_lengths)
|
||||
|
||||
# report statistics on the data
|
||||
traj_lengths = np.array(traj_lengths)
|
||||
|
||||
# Report statistics
|
||||
logging.info("===== Basic stats =====")
|
||||
logging.info("total transitions: {}".format(np.sum(traj_lengths)))
|
||||
logging.info("total trajectories: {}".format(traj_lengths.shape[0]))
|
||||
logging.info(f"Total transitions: {np.sum(traj_lengths)}")
|
||||
logging.info(f"Total trajectories: {len(traj_lengths)}")
|
||||
logging.info(
|
||||
f"traj length mean/std: {np.mean(traj_lengths)}, {np.std(traj_lengths)}"
|
||||
f"Traj length mean/std: {np.mean(traj_lengths)}, {np.std(traj_lengths)}"
|
||||
)
|
||||
logging.info(
|
||||
f"traj length min/max: {np.min(traj_lengths)}, {np.max(traj_lengths)}"
|
||||
f"Traj length min/max: {np.min(traj_lengths)}, {np.max(traj_lengths)}"
|
||||
)
|
||||
logging.info(f"obs min: {obs_min}")
|
||||
logging.info(f"obs max: {obs_max}")
|
||||
logging.info(f"action min: {action_min}")
|
||||
logging.info(f"action max: {action_max}")
|
||||
|
||||
# deal with images
|
||||
if args.cameras is not None:
|
||||
img_shapes = []
|
||||
img_names = [] # not necessary but keep old implementation
|
||||
for camera in args.cameras:
|
||||
if f"{camera}_image" in f["data/demo_0/obs"]:
|
||||
img_shape = f["data/demo_0/obs/{}_image".format(camera)].shape[1:]
|
||||
img_shapes.append(img_shape)
|
||||
img_names.append(f"{camera}_image")
|
||||
# ensure all images have the same height and width
|
||||
assert all(
|
||||
[
|
||||
img_shape[0] == img_shapes[0][0]
|
||||
and img_shape[1] == img_shapes[0][1]
|
||||
for img_shape in img_shapes
|
||||
]
|
||||
)
|
||||
combined_img_shape = (
|
||||
img_shapes[0][0],
|
||||
img_shapes[0][1],
|
||||
sum([img_shape[2] for img_shape in img_shapes]),
|
||||
)
|
||||
logging.info(f"Image shapes: {img_shapes}")
|
||||
|
||||
# split indices in train and val
|
||||
# Split indices into train and validation sets
|
||||
num_traj = len(traj_lengths)
|
||||
num_train = int(num_traj * (1 - val_split))
|
||||
train_indices = random.sample(range(num_traj), k=num_train)
|
||||
|
||||
# do over all indices
|
||||
out_train = {}
|
||||
keys = [
|
||||
"observations",
|
||||
"actions",
|
||||
"rewards",
|
||||
]
|
||||
if args.cameras is not None:
|
||||
keys.append("images")
|
||||
out_train["observations"] = np.empty((0, max_traj_length, obs_dim))
|
||||
out_train["actions"] = np.empty((0, max_traj_length, action_dim))
|
||||
out_train["rewards"] = np.empty((0, max_traj_length))
|
||||
out_train["traj_length"] = []
|
||||
if args.cameras is not None:
|
||||
out_train["images"] = np.empty(
|
||||
(
|
||||
0,
|
||||
max_traj_length,
|
||||
*combined_img_shape,
|
||||
),
|
||||
dtype=np.uint8,
|
||||
)
|
||||
# Initialize output dictionaries for train and val sets
|
||||
out_train = {"states": [], "actions": [], "rewards": [], "traj_lengths": []}
|
||||
out_val = deepcopy(out_train)
|
||||
train_episode_reward_all = []
|
||||
val_episode_reward_all = []
|
||||
|
||||
# Process each demo
|
||||
for i in tqdm(range(len(demos))):
|
||||
ep = demos[i]
|
||||
if i in train_indices:
|
||||
out = out_train
|
||||
else:
|
||||
out = out_val
|
||||
out = out_train if i in train_indices else out_val
|
||||
|
||||
# get episode length
|
||||
# Get trajectory data
|
||||
traj_length = f[f"data/{ep}"].attrs["num_samples"]
|
||||
out["traj_length"].append(traj_length)
|
||||
# print("Episode:", i, "Trajectory length:", traj_length)
|
||||
out["traj_lengths"].append(traj_length)
|
||||
|
||||
# extract
|
||||
raw_actions = f[f"data/{ep}/actions"][()]
|
||||
rewards = f[f"data/{ep}/rewards"][()]
|
||||
raw_obs = np.hstack(
|
||||
@@ -242,9 +189,9 @@ def make_dataset(
|
||||
f[f"data/{ep}/obs/{low_dim_obs_name}"][()]
|
||||
for low_dim_obs_name in low_dim_obs_names
|
||||
]
|
||||
) # not normalized
|
||||
)
|
||||
|
||||
# scale to [-1, 1] for both ob and action
|
||||
# Normalize if specified
|
||||
if normalize:
|
||||
obs = 2 * (raw_obs - obs_min) / (obs_max - obs_min + 1e-6) - 1
|
||||
actions = (
|
||||
@@ -255,128 +202,60 @@ def make_dataset(
|
||||
obs = raw_obs
|
||||
actions = raw_actions
|
||||
|
||||
data_traj = {
|
||||
"observations": obs,
|
||||
"actions": actions,
|
||||
"rewards": rewards,
|
||||
}
|
||||
if args.cameras is not None: # no normalization
|
||||
data_traj["images"] = np.concatenate(
|
||||
(
|
||||
[
|
||||
f["data/{}/obs/{}".format(ep, img_name)][()]
|
||||
for img_name in img_names
|
||||
]
|
||||
),
|
||||
axis=-1,
|
||||
)
|
||||
# Store trajectories in output dictionary
|
||||
out["states"].append(obs)
|
||||
out["actions"].append(actions)
|
||||
out["rewards"].append(rewards)
|
||||
|
||||
# apply padding to make all episodes have the same max steps
|
||||
# later when we load this dataset, we will use the traj_length to slice the data
|
||||
for key in keys:
|
||||
traj = data_traj[key]
|
||||
if traj.ndim == 1:
|
||||
pad_width = (0, max_traj_length - len(traj))
|
||||
elif traj.ndim == 2:
|
||||
pad_width = ((0, max_traj_length - traj.shape[0]), (0, 0))
|
||||
elif traj.ndim == 4:
|
||||
pad_width = (
|
||||
(0, max_traj_length - traj.shape[0]),
|
||||
(0, 0),
|
||||
(0, 0),
|
||||
(0, 0),
|
||||
)
|
||||
else:
|
||||
raise ValueError("Unsupported dimension")
|
||||
traj = np.pad(
|
||||
traj,
|
||||
pad_width,
|
||||
mode="constant",
|
||||
constant_values=0,
|
||||
)
|
||||
out[key] = np.vstack((out[key], traj[None]))
|
||||
# Concatenate trajectories (no padding)
|
||||
for key in ["states", "actions", "rewards"]:
|
||||
out_train[key] = np.concatenate(out_train[key], axis=0)
|
||||
|
||||
# check reward
|
||||
if i in train_indices:
|
||||
train_episode_reward_all.append(np.sum(data_traj["rewards"]))
|
||||
else:
|
||||
val_episode_reward_all.append(np.sum(data_traj["rewards"]))
|
||||
# Only concatenate validation set if it exists
|
||||
if val_split > 0:
|
||||
out_val[key] = np.concatenate(out_val[key], axis=0)
|
||||
|
||||
# Save to np file
|
||||
save_train_path = os.path.join(save_dir, save_name_prefix + "train.npz")
|
||||
save_val_path = os.path.join(save_dir, save_name_prefix + "val.npz")
|
||||
with open(save_train_path, "wb") as f:
|
||||
pickle.dump(out_train, f)
|
||||
with open(save_val_path, "wb") as f:
|
||||
pickle.dump(out_val, f)
|
||||
if normalize:
|
||||
normalization_save_path = os.path.join(
|
||||
save_dir, save_name_prefix + "normalization.npz"
|
||||
)
|
||||
np.savez(
|
||||
normalization_save_path,
|
||||
obs_min=obs_min,
|
||||
obs_max=obs_max,
|
||||
action_min=action_min,
|
||||
action_max=action_max,
|
||||
# Save datasets as npz files
|
||||
train_save_path = os.path.join(save_dir, save_name_prefix + "train.npz")
|
||||
np.savez_compressed(
|
||||
train_save_path,
|
||||
states=np.array(out_train["states"]),
|
||||
actions=np.array(out_train["actions"]),
|
||||
rewards=np.array(out_train["rewards"]),
|
||||
terminals=np.array([False] * len(out_train["states"])),
|
||||
traj_lengths=np.array(out_train["traj_lengths"]),
|
||||
)
|
||||
|
||||
# debug
|
||||
logging.info("\n========== Final ===========")
|
||||
logging.info(
|
||||
f"Train - Number of episodes and transitions: {len(out_train['traj_length'])}, {np.sum(out_train['traj_length'])}"
|
||||
)
|
||||
logging.info(
|
||||
f"Val - Number of episodes and transitions: {len(out_val['traj_length'])}, {np.sum(out_val['traj_length'])}"
|
||||
)
|
||||
logging.info(
|
||||
f"Train - Mean/Std trajectory length: {np.mean(out_train['traj_length'])}, {np.std(out_train['traj_length'])}"
|
||||
)
|
||||
logging.info(
|
||||
f"Train - Max/Min trajectory length: {np.max(out_train['traj_length'])}, {np.min(out_train['traj_length'])}"
|
||||
)
|
||||
logging.info(
|
||||
f"Train - Mean/Std episode reward: {np.mean(train_episode_reward_all)}, {np.std(train_episode_reward_all)}"
|
||||
)
|
||||
if val_split > 0:
|
||||
logging.info(
|
||||
f"Val - Mean/Std trajectory length: {np.mean(out_val['traj_length'])}, {np.std(out_val['traj_length'])}"
|
||||
val_save_path = os.path.join(save_dir, save_name_prefix + "val.npz")
|
||||
np.savez_compressed(
|
||||
val_save_path,
|
||||
states=np.array(out_val["states"]),
|
||||
actions=np.array(out_val["actions"]),
|
||||
rewards=np.array(out_val["rewards"]),
|
||||
terminals=np.array([False] * len(out_val["states"])),
|
||||
traj_lengths=np.array(out_val["traj_lengths"]),
|
||||
)
|
||||
logging.info(
|
||||
f"Val - Max/Min trajectory length: {np.max(out_val['traj_length'])}, {np.min(out_val['traj_length'])}"
|
||||
)
|
||||
logging.info(
|
||||
f"Val - Mean/Std episode reward: {np.mean(val_episode_reward_all)}, {np.std(val_episode_reward_all)}"
|
||||
)
|
||||
for obs_dim_ind in range(obs_dim):
|
||||
obs = out_train["observations"][:, :, obs_dim_ind]
|
||||
logging.info(
|
||||
f"Train - Obs dim {obs_dim_ind+1} mean {np.mean(obs)} std {np.std(obs)} min {np.min(obs)} max {np.max(obs)}"
|
||||
)
|
||||
for action_dim_ind in range(action_dim):
|
||||
action = out_train["actions"][:, :, action_dim_ind]
|
||||
logging.info(
|
||||
f"Train - Action dim {action_dim_ind+1} mean {np.mean(action)} std {np.std(action)} min {np.min(action)} max {np.max(action)}"
|
||||
)
|
||||
if val_split > 0:
|
||||
for obs_dim_ind in range(obs_dim):
|
||||
obs = out_val["observations"][:, :, obs_dim_ind]
|
||||
logging.info(
|
||||
f"Val - Obs dim {obs_dim_ind+1} mean {np.mean(obs)} std {np.std(obs)} min {np.min(obs)} max {np.max(obs)}"
|
||||
|
||||
# Save normalization stats if required
|
||||
if normalize:
|
||||
normalization_save_path = os.path.join(
|
||||
save_dir, save_name_prefix + "normalization.npz"
|
||||
)
|
||||
for action_dim_ind in range(action_dim):
|
||||
action = out_val["actions"][:, :, action_dim_ind]
|
||||
logging.info(
|
||||
f"Val - Action dim {action_dim_ind+1} mean {np.mean(action)} std {np.std(action)} min {np.min(action)} max {np.max(action)}"
|
||||
np.savez_compressed(
|
||||
normalization_save_path,
|
||||
obs_min=obs_min,
|
||||
obs_max=obs_max,
|
||||
action_min=action_min,
|
||||
action_max=action_max,
|
||||
)
|
||||
# logging.info("Train - Observation shape:", out_train["observations"].shape)
|
||||
# logging.info("Train - Action shape:", out_train["actions"].shape)
|
||||
# logging.info("Train - Reward shape:", out_train["rewards"].shape)
|
||||
# logging.info("Val - Observation shape:", out_val["observations"].shape)
|
||||
# logging.info("Val - Action shape:", out_val["actions"].shape)
|
||||
# logging.info("Val - Reward shape:", out_val["rewards"].shape)
|
||||
# if use_img:
|
||||
# logging.info("Image shapes:", img_shapes)
|
||||
|
||||
# Logging final information
|
||||
logging.info(
|
||||
f"Train - Trajectories: {len(out_train['traj_lengths'])}, Transitions: {np.sum(out_train['traj_lengths'])}"
|
||||
)
|
||||
logging.info(
|
||||
f"Val - Trajectories: {len(out_val['traj_lengths'])}, Transitions: {np.sum(out_val['traj_lengths'])}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
@@ -386,7 +265,7 @@ if __name__ == "__main__":
|
||||
parser.add_argument("--load_path", type=str, default=".")
|
||||
parser.add_argument("--save_dir", type=str, default=".")
|
||||
parser.add_argument("--save_name_prefix", type=str, default="")
|
||||
parser.add_argument("--val_split", type=float, default="0.2")
|
||||
parser.add_argument("--val_split", type=float, default="0")
|
||||
parser.add_argument("--max_episodes", type=int, default="-1")
|
||||
parser.add_argument("--normalize", action="store_true")
|
||||
parser.add_argument("--cameras", nargs="*", default=None)
|
||||
@@ -394,9 +273,6 @@ if __name__ == "__main__":
|
||||
|
||||
import datetime
|
||||
|
||||
if args.max_episodes > 0:
|
||||
args.save_name_prefix += f"max_episodes_{args.max_episodes}_"
|
||||
|
||||
os.makedirs(args.save_dir, exist_ok=True)
|
||||
log_path = os.path.join(
|
||||
args.save_dir,
|
||||
|
||||
Reference in New Issue
Block a user