v0.5 to main (#10)

* v0.5 (#9)

* update idql configs

* update awr configs

* update dipo configs

* update qsm configs

* update dqm configs

* update project version to 0.5.0
This commit is contained in:
Allen Z. Ren
2024-10-07 16:35:13 -04:00
committed by GitHub
parent dd14c5887c
commit e0842e71dc
267 changed files with 6769 additions and 1645 deletions
+16 -1
View File
@@ -1,3 +1,18 @@
## Data processing scripts
These are some scripts used for processing the raw datasets from the benchmarks. We already pre-processed them and provide the final datasets. These scripts are for information only.
These are some scripts used for processing the raw datasets from the benchmarks. We already pre-processed them and provide the final datasets.
Gym and robomimic data
```console
python script/dataset/get_d4rl_dataset.py --env_name=hopper-medium-v2 --save_dir=data/gym/hopper-medium-v2
python script/dataset/process_robomimic_dataset.py --load_path=../robomimic_raw_data/lift_low_dim_v141.hdf5 --save_dir=data/robomimic/lift --normalize
```
Raw robomimic data can be downloaded with a clone of the repository and then
```console
cd ~/robomimic/robomimic/scripts
python download_datasets.py --tasks all --dataset_types mh --hdf5_types low_dim # state-only policy
python download_datasets.py --tasks all --dataset_types mh --hdf5_types raw # pixel-based policy
# for pixel, replay the trajectories to extract image observations
python robomimic/scripts/dataset_states_to_obs.py --done_mode 2 --dataset datasets/can/mh/demo_v141.hdf5 --output_name image_v141.hdf5 --camera_names robot0_eye_in_hand --camera_height 96 --camera_width 96 --exclude-next-obs --n 100
```
+106 -148
View File
@@ -1,95 +1,79 @@
"""
Download D4RL dataset and save it into our custom format so it can be loaded for diffusion training.
Download D4RL dataset and save it into our custom format for diffusion training.
"""
import os
import logging
import gym
import random
from copy import deepcopy
import numpy as np
from tqdm import tqdm
import pickle
import d4rl.gym_mujoco # Import required to register environments
from copy import deepcopy
def make_dataset(env_name, save_dir, save_name_prefix, val_split, logger):
# Create the environment
env = gym.make(env_name)
# d4rl abides by the OpenAI gym interface
env.reset()
env.step(env.action_space.sample())
# Each task is associated with a dataset
# dataset contains observations, actions, rewards, terminals, and infos
env.step(
env.action_space.sample()
) # Interact with the environment to initialize it
dataset = env.get_dataset()
# rename observations to states
dataset["states"] = dataset.pop("observations")
logger.info("\n========== Basic Info ===========")
logger.info(f"Keys in the dataset: {dataset.keys()}")
logger.info(f"Observation shape: {dataset['observations'].shape}")
logger.info(f"State shape: {dataset['states'].shape}")
logger.info(f"Action shape: {dataset['actions'].shape}")
# determine trajectories from terminals and timeouts
terminal_indices = np.argwhere(dataset["terminals"])[:, 0]
timeout_indices = np.argwhere(dataset["timeouts"])[:, 0]
obs_dim = dataset["observations"].shape[1]
action_dim = dataset["actions"].shape[1]
done_indices = np.concatenate([terminal_indices, timeout_indices])
done_indices = np.sort(done_indices)
traj_lengths = []
prev_index = 0
for i in tqdm(range(len(done_indices))):
# get episode length
cur_index = done_indices[i]
traj_lengths.append(cur_index - prev_index + 1)
prev_index = cur_index + 1
obs_min = np.min(dataset["observations"], axis=0)
obs_max = np.max(dataset["observations"], axis=0)
done_indices = np.sort(np.concatenate([terminal_indices, timeout_indices]))
traj_lengths = np.diff(np.concatenate([[0], done_indices + 1]))
obs_min = np.min(dataset["states"], axis=0)
obs_max = np.max(dataset["states"], axis=0)
action_min = np.min(dataset["actions"], axis=0)
action_max = np.max(dataset["actions"], axis=0)
max_episode_steps = max(traj_lengths)
logger.info("total transitions: {}".format(np.sum(traj_lengths)))
logger.info("total trajectories: {}".format(len(traj_lengths)))
logger.info(
f"traj length mean/std: {np.mean(traj_lengths)}, {np.std(traj_lengths)}"
)
logger.info(f"traj length min/max: {np.min(traj_lengths)}, {np.max(traj_lengths)}")
logger.info(f"obs min: {obs_min}")
logger.info(f"obs max: {obs_max}")
logger.info(f"action min: {action_min}")
logger.info(f"action max: {action_max}")
# Subsample episodes by taking the first ones
logger.info(f"Total transitions: {np.sum(traj_lengths)}")
logger.info(f"Total trajectories: {len(traj_lengths)}")
logger.info(
f"Trajectory length mean/std: {np.mean(traj_lengths)}, {np.std(traj_lengths)}"
)
logger.info(
f"Trajectory length min/max: {np.min(traj_lengths)}, {np.max(traj_lengths)}"
)
logger.info(f"obs min: {obs_min}, obs max: {obs_max}")
logger.info(f"action min: {action_min}, action max: {action_max}")
# Subsample episodes if needed
if args.max_episodes > 0:
traj_lengths = traj_lengths[: args.max_episodes]
done_indices = done_indices[: args.max_episodes]
max_episode_steps = max(traj_lengths)
# split indices in train and val
# Split into train and validation sets
num_traj = len(traj_lengths)
num_train = int(num_traj * (1 - val_split))
train_indices = random.sample(range(num_traj), k=num_train)
# do over all indices
out_train = {}
keys = [
"observations",
"actions",
"rewards",
]
out_train["observations"] = np.empty(
(0, max_episode_steps, dataset["observations"].shape[-1])
)
out_train["actions"] = np.empty(
(0, max_episode_steps, dataset["actions"].shape[-1])
)
out_train["rewards"] = np.empty((0, max_episode_steps))
out_train["traj_length"] = []
# Prepare data containers for train and validation sets
out_train = {
"states": [],
"actions": [],
"rewards": [],
"terminals": [],
"traj_lengths": [],
}
out_val = deepcopy(out_train)
prev_index = 0
train_episode_reward_all = []
val_episode_reward_all = []
for i in tqdm(range(len(done_indices))):
for i, cur_index in tqdm(enumerate(done_indices), total=len(done_indices)):
if i in train_indices:
out = out_train
episode_reward_all = train_episode_reward_all
@@ -97,57 +81,65 @@ def make_dataset(env_name, save_dir, save_name_prefix, val_split, logger):
out = out_val
episode_reward_all = val_episode_reward_all
# get episode length
cur_index = done_indices[i]
# Get the trajectory length and slice
traj_length = cur_index - prev_index + 1
trajectory = {
key: dataset[key][prev_index : cur_index + 1]
for key in ["states", "actions", "rewards", "terminals"]
}
# Skip if the episode has no reward
if np.sum(dataset["rewards"][prev_index : cur_index + 1]) > 0:
out["traj_length"].append(traj_length)
# Skip if there is no reward in the episode
if np.sum(trajectory["rewards"]) > 0:
# Scale observations and actions
trajectory["states"] = (
2 * (trajectory["states"] - obs_min) / (obs_max - obs_min + 1e-6) - 1
)
trajectory["actions"] = (
2
* (trajectory["actions"] - action_min)
/ (action_max - action_min + 1e-6)
- 1
)
# apply padding to make all episodes have the same max steps
for key in keys:
traj = dataset[key][prev_index : cur_index + 1]
# also scale
if key == "observations":
traj = 2 * (traj - obs_min) / (obs_max - obs_min + 1e-6) - 1
elif key == "actions":
traj = (
2 * (traj - action_min) / (action_max - action_min + 1e-6) - 1
)
if traj.ndim == 1:
traj = np.pad(
traj,
(0, max_episode_steps - len(traj)),
mode="constant",
constant_values=0,
)
else:
traj = np.pad(
traj,
((0, max_episode_steps - traj.shape[0]), (0, 0)),
mode="constant",
constant_values=0,
)
out[key] = np.vstack((out[key], traj[None]))
# check reward
episode_reward_all.append(np.sum(out["rewards"][-1]))
for key in ["states", "actions", "rewards", "terminals"]:
out[key].append(trajectory[key])
out["traj_lengths"].append(traj_length)
episode_reward_all.append(np.sum(trajectory["rewards"]))
else:
print(f"skipping {i} / {len(done_indices)}")
logger.info(f"Skipping trajectory {i} due to zero rewards.")
# update prev index
prev_index = cur_index + 1
# Save to np file
save_train_path = os.path.join(save_dir, save_name_prefix + "train.npz")
save_val_path = os.path.join(save_dir, save_name_prefix + "val.npz")
with open(save_train_path, "wb") as f:
pickle.dump(out_train, f)
with open(save_val_path, "wb") as f:
pickle.dump(out_val, f)
# Concatenate trajectories
for key in ["states", "actions", "rewards", "terminals"]:
out_train[key] = np.concatenate(out_train[key], axis=0)
# Only concatenate validation set if it exists
if val_split > 0:
out_val[key] = np.concatenate(out_val[key], axis=0)
# Save train dataset to npz files
train_save_path = os.path.join(save_dir, save_name_prefix + "train.npz")
np.savez_compressed(
train_save_path,
states=np.array(out_train["states"]),
actions=np.array(out_train["actions"]),
rewards=np.array(out_train["rewards"]),
terminals=np.array(out_train["terminals"]),
traj_lengths=np.array(out_train["traj_lengths"]),
)
# Save validation dataset to npz files
val_save_path = os.path.join(save_dir, save_name_prefix + "val.npz")
np.savez_compressed(
val_save_path,
states=np.array(out_val["states"]),
actions=np.array(out_val["actions"]),
rewards=np.array(out_val["rewards"]),
terminals=np.array(out_val["terminals"]),
traj_lengths=np.array(out_val["traj_lengths"]),
)
normalization_save_path = os.path.join(
save_dir, save_name_prefix + "normalization.npz"
)
@@ -159,83 +151,49 @@ def make_dataset(env_name, save_dir, save_name_prefix, val_split, logger):
action_max=action_max,
)
# debug
# Logging summary statistics
logger.info("\n========== Final ===========")
logger.info(
f"Train - Number of episodes and transitions: {len(out_train['traj_length'])}, {np.sum(out_train['traj_length'])}"
f"Train - Trajectories: {len(out_train['traj_lengths'])}, Transitions: {np.sum(out_train['traj_lengths'])}"
)
logger.info(
f"Val - Number of episodes and transitions: {len(out_val['traj_length'])}, {np.sum(out_val['traj_length'])}"
f"Val - Trajectories: {len(out_val['traj_lengths'])}, Transitions: {np.sum(out_val['traj_lengths'])}"
)
logger.info(
f"Train - Mean/Std trajectory length: {np.mean(out_train['traj_length'])}, {np.std(out_train['traj_length'])}"
f"Train - Mean/Std trajectory length: {np.mean(out_train['traj_lengths'])}, {np.std(out_train['traj_lengths'])}"
)
logger.info(
f"Train - Max/Min trajectory length: {np.max(out_train['traj_length'])}, {np.min(out_train['traj_length'])}"
(
logger.info(
f"Val - Mean/Std trajectory length: {np.mean(out_val['traj_lengths'])}, {np.std(out_val['traj_lengths'])}"
)
if val_split > 0
else None
)
if val_split > 0:
logger.info(
f"Val - Mean/Std trajectory length: {np.mean(out_val['traj_length'])}, {np.std(out_val['traj_length'])}"
)
logger.info(
f"Val - Max/Min trajectory length: {np.max(out_val['traj_length'])}, {np.min(out_val['traj_length'])}"
)
logger.info(
f"Train - Mean/Std episode reward: {np.mean(train_episode_reward_all)}, {np.std(train_episode_reward_all)}"
)
if val_split > 0:
logger.info(
f"Val - Mean/Std episode reward: {np.mean(val_episode_reward_all)}, {np.std(val_episode_reward_all)}"
)
for obs_dim_ind in range(obs_dim):
obs = out_train["observations"][:, :, obs_dim_ind]
logger.info(
f"Train - Obs dim {obs_dim_ind+1} mean {np.mean(obs)} std {np.std(obs)} min {np.min(obs)} max {np.max(obs)}"
)
for action_dim_ind in range(action_dim):
action = out_train["actions"][:, :, action_dim_ind]
logger.info(
f"Train - Action dim {action_dim_ind+1} mean {np.mean(action)} std {np.std(action)} min {np.min(action)} max {np.max(action)}"
)
if val_split > 0:
for obs_dim_ind in range(obs_dim):
obs = out_val["observations"][:, :, obs_dim_ind]
logger.info(
f"Val - Obs dim {obs_dim_ind+1} mean {np.mean(obs)} std {np.std(obs)} min {np.min(obs)} max {np.max(obs)}"
)
for action_dim_ind in range(action_dim):
action = out_val["actions"][:, :, action_dim_ind]
logger.info(
f"Val - Action dim {action_dim_ind+1} mean {np.mean(action)} std {np.std(action)} min {np.min(action)} max {np.max(action)}"
)
if __name__ == "__main__":
import argparse
import datetime
parser = argparse.ArgumentParser()
parser.add_argument("--env_name", type=str, default="hopper-medium-v2")
parser.add_argument("--save_dir", type=str, default=".")
parser.add_argument("--save_name_prefix", type=str, default="")
parser.add_argument("--val_split", type=float, default="0.2")
parser.add_argument("--max_episodes", type=int, default="-1")
parser.add_argument("--val_split", type=float, default=0)
parser.add_argument("--max_episodes", type=int, default=-1)
args = parser.parse_args()
import datetime
# import logging.config
if args.max_episodes > 0:
args.save_name_prefix += f"max_episodes_{args.max_episodes}_"
os.makedirs(args.save_dir, exist_ok=True)
log_path = os.path.join(
args.save_dir,
args.save_name_prefix
+ f"_{datetime.datetime.now().strftime('%Y_%m_%d_%H_%M_%S')}.log",
)
logger = logging.getLogger("get_D4RL_dataset")
logger.setLevel(logging.INFO)
file_handler = logging.FileHandler(log_path)
file_handler.setLevel(logging.INFO) # Set the minimum level for this handler
file_handler.setLevel(logging.INFO)
formatter = logging.Formatter(
"%(asctime)s - %(name)s - %(levelname)s - %(message)s"
)
+80 -204
View File
@@ -3,6 +3,8 @@ Process robomimic dataset and save it into our custom format so it can be loaded
Using some code from robomimic/robomimic/scripts/get_dataset_info.py
Since we do not terminate episode early and cumulate reward when the goal is reached, we set terminals to all False.
can-mh:
total transitions: 62756
total trajectories: 300
@@ -76,31 +78,20 @@ robomimic dataset normalizes action to [-1, 1], observation roughly? to [-1, 1].
"""
import numpy as np
from tqdm import tqdm
import pickle
try:
import h5py # not included in pyproject.toml
except:
print("Installing h5py")
os.system("pip install h5py")
import h5py
import os
import random
from copy import deepcopy
import logging
def make_dataset(
load_path,
save_dir,
save_name_prefix,
val_split,
normalize,
):
def make_dataset(load_path, save_dir, save_name_prefix, val_split, normalize):
# Load hdf5 file from load_path
with h5py.File(load_path, "r") as f:
# put demonstration list in increasing episode order
# Sort demonstrations in increasing episode order
demos = sorted(list(f["data"].keys()))
inds = np.argsort([int(elem[5:]) for elem in demos])
demos = [demos[i] for i in inds]
@@ -108,7 +99,7 @@ def make_dataset(
if args.max_episodes > 0:
demos = demos[: args.max_episodes]
# From generate_paper_configs.py: default observation is eef pose, gripper finger position, and object information, all of which are low-dim.
# Default low-dimensional observation keys
low_dim_obs_names = [
"robot0_eef_pos",
"robot0_eef_quat",
@@ -120,23 +111,28 @@ def make_dataset(
"robot1_eef_quat",
"robot1_gripper_qpos",
]
if args.cameras is None: # state-only
if args.cameras is None:
low_dim_obs_names.append("object")
# Calculate dimensions for observations and actions
obs_dim = 0
for low_dim_obs_name in low_dim_obs_names:
dim = f["data/demo_0/obs/{}".format(low_dim_obs_name)].shape[1]
dim = f[f"data/demo_0/obs/{low_dim_obs_name}"].shape[1]
obs_dim += dim
logging.info(f"Using {low_dim_obs_name} with dim {dim} for observation")
action_dim = f["data/demo_0/actions"].shape[1]
logging.info(f"Total low-dim observation dim: {obs_dim}")
logging.info(f"Action dim: {action_dim}")
# get basic stats
# Initialize variables for tracking trajectory statistics
traj_lengths = []
obs_min = np.zeros((obs_dim))
obs_max = np.zeros((obs_dim))
action_min = np.zeros((action_dim))
action_max = np.zeros((action_dim))
# Process each demo
for ep in demos:
traj_lengths.append(f[f"data/{ep}/actions"].shape[0])
obs = np.hstack(
@@ -145,96 +141,47 @@ def make_dataset(
for low_dim_obs_name in low_dim_obs_names
]
)
actions = f[f"data/{ep}/actions"]
actions = f[f"data/{ep}/actions"][()]
obs_min = np.minimum(obs_min, np.min(obs, axis=0))
obs_max = np.maximum(obs_max, np.max(obs, axis=0))
action_min = np.minimum(action_min, np.min(actions, axis=0))
action_max = np.maximum(action_max, np.max(actions, axis=0))
traj_lengths = np.array(traj_lengths)
max_traj_length = np.max(traj_lengths)
# report statistics on the data
traj_lengths = np.array(traj_lengths)
# Report statistics
logging.info("===== Basic stats =====")
logging.info("total transitions: {}".format(np.sum(traj_lengths)))
logging.info("total trajectories: {}".format(traj_lengths.shape[0]))
logging.info(f"Total transitions: {np.sum(traj_lengths)}")
logging.info(f"Total trajectories: {len(traj_lengths)}")
logging.info(
f"traj length mean/std: {np.mean(traj_lengths)}, {np.std(traj_lengths)}"
f"Traj length mean/std: {np.mean(traj_lengths)}, {np.std(traj_lengths)}"
)
logging.info(
f"traj length min/max: {np.min(traj_lengths)}, {np.max(traj_lengths)}"
f"Traj length min/max: {np.min(traj_lengths)}, {np.max(traj_lengths)}"
)
logging.info(f"obs min: {obs_min}")
logging.info(f"obs max: {obs_max}")
logging.info(f"action min: {action_min}")
logging.info(f"action max: {action_max}")
# deal with images
if args.cameras is not None:
img_shapes = []
img_names = [] # not necessary but keep old implementation
for camera in args.cameras:
if f"{camera}_image" in f["data/demo_0/obs"]:
img_shape = f["data/demo_0/obs/{}_image".format(camera)].shape[1:]
img_shapes.append(img_shape)
img_names.append(f"{camera}_image")
# ensure all images have the same height and width
assert all(
[
img_shape[0] == img_shapes[0][0]
and img_shape[1] == img_shapes[0][1]
for img_shape in img_shapes
]
)
combined_img_shape = (
img_shapes[0][0],
img_shapes[0][1],
sum([img_shape[2] for img_shape in img_shapes]),
)
logging.info(f"Image shapes: {img_shapes}")
# split indices in train and val
# Split indices into train and validation sets
num_traj = len(traj_lengths)
num_train = int(num_traj * (1 - val_split))
train_indices = random.sample(range(num_traj), k=num_train)
# do over all indices
out_train = {}
keys = [
"observations",
"actions",
"rewards",
]
if args.cameras is not None:
keys.append("images")
out_train["observations"] = np.empty((0, max_traj_length, obs_dim))
out_train["actions"] = np.empty((0, max_traj_length, action_dim))
out_train["rewards"] = np.empty((0, max_traj_length))
out_train["traj_length"] = []
if args.cameras is not None:
out_train["images"] = np.empty(
(
0,
max_traj_length,
*combined_img_shape,
),
dtype=np.uint8,
)
# Initialize output dictionaries for train and val sets
out_train = {"states": [], "actions": [], "rewards": [], "traj_lengths": []}
out_val = deepcopy(out_train)
train_episode_reward_all = []
val_episode_reward_all = []
# Process each demo
for i in tqdm(range(len(demos))):
ep = demos[i]
if i in train_indices:
out = out_train
else:
out = out_val
out = out_train if i in train_indices else out_val
# get episode length
# Get trajectory data
traj_length = f[f"data/{ep}"].attrs["num_samples"]
out["traj_length"].append(traj_length)
# print("Episode:", i, "Trajectory length:", traj_length)
out["traj_lengths"].append(traj_length)
# extract
raw_actions = f[f"data/{ep}/actions"][()]
rewards = f[f"data/{ep}/rewards"][()]
raw_obs = np.hstack(
@@ -242,9 +189,9 @@ def make_dataset(
f[f"data/{ep}/obs/{low_dim_obs_name}"][()]
for low_dim_obs_name in low_dim_obs_names
]
) # not normalized
)
# scale to [-1, 1] for both ob and action
# Normalize if specified
if normalize:
obs = 2 * (raw_obs - obs_min) / (obs_max - obs_min + 1e-6) - 1
actions = (
@@ -255,128 +202,60 @@ def make_dataset(
obs = raw_obs
actions = raw_actions
data_traj = {
"observations": obs,
"actions": actions,
"rewards": rewards,
}
if args.cameras is not None: # no normalization
data_traj["images"] = np.concatenate(
(
[
f["data/{}/obs/{}".format(ep, img_name)][()]
for img_name in img_names
]
),
axis=-1,
)
# Store trajectories in output dictionary
out["states"].append(obs)
out["actions"].append(actions)
out["rewards"].append(rewards)
# apply padding to make all episodes have the same max steps
# later when we load this dataset, we will use the traj_length to slice the data
for key in keys:
traj = data_traj[key]
if traj.ndim == 1:
pad_width = (0, max_traj_length - len(traj))
elif traj.ndim == 2:
pad_width = ((0, max_traj_length - traj.shape[0]), (0, 0))
elif traj.ndim == 4:
pad_width = (
(0, max_traj_length - traj.shape[0]),
(0, 0),
(0, 0),
(0, 0),
)
else:
raise ValueError("Unsupported dimension")
traj = np.pad(
traj,
pad_width,
mode="constant",
constant_values=0,
)
out[key] = np.vstack((out[key], traj[None]))
# Concatenate trajectories (no padding)
for key in ["states", "actions", "rewards"]:
out_train[key] = np.concatenate(out_train[key], axis=0)
# check reward
if i in train_indices:
train_episode_reward_all.append(np.sum(data_traj["rewards"]))
else:
val_episode_reward_all.append(np.sum(data_traj["rewards"]))
# Only concatenate validation set if it exists
if val_split > 0:
out_val[key] = np.concatenate(out_val[key], axis=0)
# Save to np file
save_train_path = os.path.join(save_dir, save_name_prefix + "train.npz")
save_val_path = os.path.join(save_dir, save_name_prefix + "val.npz")
with open(save_train_path, "wb") as f:
pickle.dump(out_train, f)
with open(save_val_path, "wb") as f:
pickle.dump(out_val, f)
if normalize:
normalization_save_path = os.path.join(
save_dir, save_name_prefix + "normalization.npz"
)
np.savez(
normalization_save_path,
obs_min=obs_min,
obs_max=obs_max,
action_min=action_min,
action_max=action_max,
# Save datasets as npz files
train_save_path = os.path.join(save_dir, save_name_prefix + "train.npz")
np.savez_compressed(
train_save_path,
states=np.array(out_train["states"]),
actions=np.array(out_train["actions"]),
rewards=np.array(out_train["rewards"]),
terminals=np.array([False] * len(out_train["states"])),
traj_lengths=np.array(out_train["traj_lengths"]),
)
# debug
logging.info("\n========== Final ===========")
logging.info(
f"Train - Number of episodes and transitions: {len(out_train['traj_length'])}, {np.sum(out_train['traj_length'])}"
)
logging.info(
f"Val - Number of episodes and transitions: {len(out_val['traj_length'])}, {np.sum(out_val['traj_length'])}"
)
logging.info(
f"Train - Mean/Std trajectory length: {np.mean(out_train['traj_length'])}, {np.std(out_train['traj_length'])}"
)
logging.info(
f"Train - Max/Min trajectory length: {np.max(out_train['traj_length'])}, {np.min(out_train['traj_length'])}"
)
logging.info(
f"Train - Mean/Std episode reward: {np.mean(train_episode_reward_all)}, {np.std(train_episode_reward_all)}"
)
if val_split > 0:
logging.info(
f"Val - Mean/Std trajectory length: {np.mean(out_val['traj_length'])}, {np.std(out_val['traj_length'])}"
val_save_path = os.path.join(save_dir, save_name_prefix + "val.npz")
np.savez_compressed(
val_save_path,
states=np.array(out_val["states"]),
actions=np.array(out_val["actions"]),
rewards=np.array(out_val["rewards"]),
terminals=np.array([False] * len(out_val["states"])),
traj_lengths=np.array(out_val["traj_lengths"]),
)
logging.info(
f"Val - Max/Min trajectory length: {np.max(out_val['traj_length'])}, {np.min(out_val['traj_length'])}"
)
logging.info(
f"Val - Mean/Std episode reward: {np.mean(val_episode_reward_all)}, {np.std(val_episode_reward_all)}"
)
for obs_dim_ind in range(obs_dim):
obs = out_train["observations"][:, :, obs_dim_ind]
logging.info(
f"Train - Obs dim {obs_dim_ind+1} mean {np.mean(obs)} std {np.std(obs)} min {np.min(obs)} max {np.max(obs)}"
)
for action_dim_ind in range(action_dim):
action = out_train["actions"][:, :, action_dim_ind]
logging.info(
f"Train - Action dim {action_dim_ind+1} mean {np.mean(action)} std {np.std(action)} min {np.min(action)} max {np.max(action)}"
)
if val_split > 0:
for obs_dim_ind in range(obs_dim):
obs = out_val["observations"][:, :, obs_dim_ind]
logging.info(
f"Val - Obs dim {obs_dim_ind+1} mean {np.mean(obs)} std {np.std(obs)} min {np.min(obs)} max {np.max(obs)}"
# Save normalization stats if required
if normalize:
normalization_save_path = os.path.join(
save_dir, save_name_prefix + "normalization.npz"
)
for action_dim_ind in range(action_dim):
action = out_val["actions"][:, :, action_dim_ind]
logging.info(
f"Val - Action dim {action_dim_ind+1} mean {np.mean(action)} std {np.std(action)} min {np.min(action)} max {np.max(action)}"
np.savez_compressed(
normalization_save_path,
obs_min=obs_min,
obs_max=obs_max,
action_min=action_min,
action_max=action_max,
)
# logging.info("Train - Observation shape:", out_train["observations"].shape)
# logging.info("Train - Action shape:", out_train["actions"].shape)
# logging.info("Train - Reward shape:", out_train["rewards"].shape)
# logging.info("Val - Observation shape:", out_val["observations"].shape)
# logging.info("Val - Action shape:", out_val["actions"].shape)
# logging.info("Val - Reward shape:", out_val["rewards"].shape)
# if use_img:
# logging.info("Image shapes:", img_shapes)
# Logging final information
logging.info(
f"Train - Trajectories: {len(out_train['traj_lengths'])}, Transitions: {np.sum(out_train['traj_lengths'])}"
)
logging.info(
f"Val - Trajectories: {len(out_val['traj_lengths'])}, Transitions: {np.sum(out_val['traj_lengths'])}"
)
if __name__ == "__main__":
@@ -386,7 +265,7 @@ if __name__ == "__main__":
parser.add_argument("--load_path", type=str, default=".")
parser.add_argument("--save_dir", type=str, default=".")
parser.add_argument("--save_name_prefix", type=str, default="")
parser.add_argument("--val_split", type=float, default="0.2")
parser.add_argument("--val_split", type=float, default="0")
parser.add_argument("--max_episodes", type=int, default="-1")
parser.add_argument("--normalize", action="store_true")
parser.add_argument("--cameras", nargs="*", default=None)
@@ -394,9 +273,6 @@ if __name__ == "__main__":
import datetime
if args.max_episodes > 0:
args.save_name_prefix += f"max_episodes_{args.max_episodes}_"
os.makedirs(args.save_dir, exist_ok=True)
log_path = os.path.join(
args.save_dir,