Paper code basis
This commit is contained in:
Vendored
+8
@@ -0,0 +1,8 @@
|
||||
type: brax
|
||||
name:
|
||||
max_episode_steps: 1000
|
||||
reward_scaling: 0.1
|
||||
terminate: true
|
||||
|
||||
vmin: 0
|
||||
vmax: 150
|
||||
Vendored
+8
@@ -0,0 +1,8 @@
|
||||
type: brax
|
||||
name:
|
||||
max_episode_steps: 1000
|
||||
reward_scaling: 0.1
|
||||
terminate: true
|
||||
|
||||
vmin: 0
|
||||
vmax: 200
|
||||
Vendored
+3
@@ -0,0 +1,3 @@
|
||||
type: isaaclab
|
||||
name:
|
||||
action_bounds: [-1, 1]
|
||||
Vendored
+16
@@ -0,0 +1,16 @@
|
||||
type: maniskill
|
||||
name:
|
||||
reconfiguration_freq: 1
|
||||
partial_reset: true
|
||||
asymmetric_obs: false
|
||||
max_episode_steps: 50
|
||||
stochastic_eval: true
|
||||
has_final_obs: true
|
||||
|
||||
env_kwargs:
|
||||
obs_mode: state
|
||||
render_mode: rgb_array
|
||||
sim_backend: physx_cuda
|
||||
|
||||
vmin: -15
|
||||
vmax: 15
|
||||
Vendored
+9
@@ -0,0 +1,9 @@
|
||||
type: mjx
|
||||
name:
|
||||
max_episode_steps: 1000
|
||||
reward_scaling: 1.0
|
||||
terminate: false
|
||||
asymmetric_observation: false
|
||||
|
||||
vmin: 0
|
||||
vmax: 150
|
||||
Vendored
+10
@@ -0,0 +1,10 @@
|
||||
type: mjx
|
||||
name:
|
||||
max_episode_steps: 1000
|
||||
reward_scaling: 1.0
|
||||
terminate: false
|
||||
push_distractions: false
|
||||
asymmetric_observation: true
|
||||
|
||||
vmin: -10
|
||||
vmax: 10
|
||||
@@ -0,0 +1,5 @@
|
||||
lmbda: 0.95
|
||||
|
||||
num_epochs: 4
|
||||
|
||||
aux_loss_mult: 1.0
|
||||
@@ -0,0 +1,5 @@
|
||||
num_envs: 1024
|
||||
num_steps: 128
|
||||
num_mini_batches: 64
|
||||
num_epochs: 8
|
||||
kl_bound: 0.1
|
||||
@@ -0,0 +1,5 @@
|
||||
num_envs: 1024
|
||||
num_steps: 64
|
||||
num_mini_batches: 32
|
||||
num_epochs: 8
|
||||
kl_bound: 0.1
|
||||
@@ -0,0 +1,5 @@
|
||||
num_envs: 1024
|
||||
num_steps: 32
|
||||
num_mini_batches: 16
|
||||
num_epochs: 8
|
||||
kl_bound: 0.1
|
||||
@@ -0,0 +1,8 @@
|
||||
gamma: 0.97
|
||||
critic_hidden_dim: 1024
|
||||
|
||||
num_envs: 1024
|
||||
num_steps: 128
|
||||
num_mini_batches: 16
|
||||
num_epochs: 8
|
||||
kl_bound: 0.1
|
||||
@@ -0,0 +1,8 @@
|
||||
gamma: 0.97
|
||||
critic_hidden_dim: 1024
|
||||
|
||||
num_envs: 1024
|
||||
num_steps: 32
|
||||
num_mini_batches: 4
|
||||
num_epochs: 8
|
||||
kl_bound: 0.1
|
||||
@@ -0,0 +1,7 @@
|
||||
amp_enabled: false
|
||||
amp_device: "cuda"
|
||||
cuda: true
|
||||
amp_dtype: f32
|
||||
torch_deterministic: false
|
||||
device_rank: 0
|
||||
compile: true
|
||||
@@ -0,0 +1,35 @@
|
||||
defaults:
|
||||
- env: brax
|
||||
- experiment_overrides: default
|
||||
- trial_spec: default
|
||||
- _self_
|
||||
|
||||
hyperparameters:
|
||||
lr: 3e-4
|
||||
gamma: 0.99
|
||||
lmbda: 0.95
|
||||
clip_ratio: 0.2
|
||||
value_coef: 0.5
|
||||
entropy_coef: 0.0
|
||||
total_time_steps: 50_000_000
|
||||
num_steps: 64
|
||||
num_mini_batches: 32
|
||||
num_envs: 2048
|
||||
num_epochs: 16
|
||||
max_grad_norm: 0.5
|
||||
normalize_advantages: True
|
||||
normalize_env: True
|
||||
anneal_lr: False
|
||||
num_eval: 20
|
||||
max_episode_steps: 1000
|
||||
name: "ppo"
|
||||
tags: ["ppo_baseline_retuned"]
|
||||
seed: 0
|
||||
num_seeds: 1
|
||||
tune: false
|
||||
checkpoint_dir: null
|
||||
trials: 8
|
||||
wandb:
|
||||
mode: "online" # set to online to activate wandb
|
||||
entity: "viper_svg"
|
||||
project: "online_sac"
|
||||
@@ -0,0 +1,89 @@
|
||||
defaults:
|
||||
- env: brax
|
||||
- experiment_overrides: default
|
||||
- trial_spec: default
|
||||
- platform: torch
|
||||
- _self_
|
||||
|
||||
hyperparameters:
|
||||
# env and run settings (mostly don't touch)
|
||||
total_time_steps: 50_000_000
|
||||
normalize_env: true
|
||||
max_episode_steps: 1000
|
||||
eval_interval: 2
|
||||
num_eval: 20
|
||||
|
||||
# optimization settings (seem very stable)
|
||||
lr: 3e-4
|
||||
anneal_lr: false
|
||||
max_grad_norm: 0.5
|
||||
polyak: 1.0 # maybe ablate ?
|
||||
|
||||
# problem discount settings (need tuning)
|
||||
gamma: 0.99
|
||||
lmbda: 0.95
|
||||
lmbda_min: 0.50 # irrelevant if no exploration noise is added
|
||||
|
||||
# batch settings (need tuning for MJX humanoid)
|
||||
num_steps: 128
|
||||
num_mini_batches: 128
|
||||
num_envs: 1024
|
||||
num_epochs: 4
|
||||
|
||||
# exploration settings (currently not touched)
|
||||
exploration_noise_max: 1.0
|
||||
exploration_noise_min: 1.0
|
||||
exploration_base_envs: 0
|
||||
|
||||
# critic architecture settings (need to be increased for MJX humanoid)
|
||||
critic_hidden_dim: 512
|
||||
actor_hidden_dim: 512
|
||||
vmin: ${env.vmin}
|
||||
vmax: ${env.vmax}
|
||||
num_bins: 151
|
||||
hl_gauss: true
|
||||
use_critic_norm: true
|
||||
num_critic_encoder_layers: 2
|
||||
num_critic_head_layers: 2
|
||||
num_critic_pred_layers: 2
|
||||
use_simplical_embedding: False
|
||||
|
||||
# actor architecture settings (seem stable)
|
||||
use_actor_norm: true
|
||||
num_actor_layers: 3
|
||||
actor_min_std: 0.0
|
||||
|
||||
# actor & critic loss settings (seem remarkably stable)
|
||||
## kl settings
|
||||
kl_start: 0.01
|
||||
kl_bound: 0.1 # switched to tighter bounds for MJX
|
||||
reduce_kl: true
|
||||
reverse_kl: false # previous default "false"
|
||||
update_kl_lagrangian: true
|
||||
actor_kl_clip_mode: "clipped" # "full", "clipped", "kl_relu_clipped", "kl_bound_clipped", "value"
|
||||
## entropy settings
|
||||
ent_start: 0.01
|
||||
ent_target_mult: 0.5
|
||||
update_entropy_lagrangian: true
|
||||
## auxiliary loss settings
|
||||
aux_loss_mult: 1.0
|
||||
|
||||
|
||||
measure_burnin: 3
|
||||
|
||||
|
||||
name: "sac"
|
||||
seed: 0
|
||||
num_seeds: 1
|
||||
tune: false
|
||||
checkpoint_dir: null
|
||||
num_trials: 10
|
||||
tags: ["experimental"]
|
||||
wandb:
|
||||
mode: "online" # set to online to activate wandb
|
||||
entity: "viper_svg"
|
||||
project: "online_sac"
|
||||
|
||||
hydra:
|
||||
job:
|
||||
chdir: True
|
||||
Reference in New Issue
Block a user