Fix 6 critical bugs in REPPO repository preventing execution
- Fix missing MUON optimizer by replacing with optax.adam - Fix Hydra configuration parameter paths (env.name instead of env_name) - Fix BraxGymnaxWrapper method signatures to accept params argument - Fix training loop division by zero with proper total_time_steps - Fix incorrect algorithm name in wandb (reppo instead of sac) - Fix JAX key batching error in BraxGymnaxWrapper reset method - Add comprehensive HoReKa SLURM integration with wandb logging - Update README with detailed bug documentation and fixes
This commit is contained in:
@@ -42,11 +42,14 @@ echo "Experiment type: $EXPERIMENT_TYPE"
|
||||
# Run the experiment
|
||||
python reppo_alg/jaxrl/reppo.py \
|
||||
env=brax \
|
||||
env_name=$ENV_NAME \
|
||||
experiment_override=$EXPERIMENT_TYPE \
|
||||
env.name=$ENV_NAME \
|
||||
hyperparameters.num_envs=1024 \
|
||||
hyperparameters.num_steps=128 \
|
||||
hyperparameters.num_mini_batches=128 \
|
||||
hyperparameters.num_epochs=4 \
|
||||
hyperparameters.total_time_steps=50000000 \
|
||||
wandb.mode=online \
|
||||
wandb.entity=${WANDB_ENTITY} \
|
||||
wandb.project=$WANDB_PROJECT \
|
||||
wandb.name="reppo_${ENV_NAME}_${EXPERIMENT_TYPE}_${SLURM_JOB_ID}"
|
||||
wandb.project=$WANDB_PROJECT
|
||||
|
||||
echo "Training completed!"
|
||||
Executable
+55
@@ -0,0 +1,55 @@
|
||||
#!/bin/bash
|
||||
#SBATCH --job-name=reppo_dev_test
|
||||
#SBATCH --account=hk-project-p0022232
|
||||
#SBATCH --partition=dev_accelerated
|
||||
#SBATCH --gres=gpu:1
|
||||
#SBATCH --nodes=1
|
||||
#SBATCH --ntasks-per-node=1
|
||||
#SBATCH --cpus-per-task=4
|
||||
#SBATCH --time=00:30:00
|
||||
#SBATCH --mem=16G
|
||||
#SBATCH --output=logs/reppo_dev_%j.out
|
||||
#SBATCH --error=logs/reppo_dev_%j.err
|
||||
|
||||
# Load required modules
|
||||
module load devel/cuda/12.4
|
||||
|
||||
# Set environment variables
|
||||
export WANDB_MODE=online
|
||||
export WANDB_PROJECT=reppo_dev_test
|
||||
export WANDB_API_KEY=01fbfaf5e2f64bedd68febedfcaa7e3bbd54952c
|
||||
export WANDB_ENTITY=dominik_roth
|
||||
|
||||
# Change to project directory
|
||||
cd /hkfs/home/project/hk-project-robolear/ys1087/Projects/reppo
|
||||
|
||||
# Activate virtual environment
|
||||
source .venv/bin/activate
|
||||
|
||||
# Run quick test with Brax (faster than ManiSkill)
|
||||
echo "Starting REPPO dev test..."
|
||||
echo "Job ID: $SLURM_JOB_ID"
|
||||
echo "Node: $SLURM_NODELIST"
|
||||
echo "GPU: $CUDA_VISIBLE_DEVICES"
|
||||
|
||||
# Use small data for quick test
|
||||
ENV_NAME=${ENV_NAME:-ant}
|
||||
EXPERIMENT_TYPE=${EXPERIMENT_TYPE:-mjx_dmc_small_data}
|
||||
|
||||
echo "Environment: $ENV_NAME"
|
||||
echo "Experiment type: $EXPERIMENT_TYPE"
|
||||
|
||||
# Run the experiment
|
||||
python reppo_alg/jaxrl/reppo.py \
|
||||
env=brax \
|
||||
env.name=$ENV_NAME \
|
||||
hyperparameters.num_envs=256 \
|
||||
hyperparameters.num_steps=32 \
|
||||
hyperparameters.num_mini_batches=8 \
|
||||
hyperparameters.num_epochs=4 \
|
||||
hyperparameters.total_time_steps=1000000 \
|
||||
wandb.mode=online \
|
||||
wandb.entity=$WANDB_ENTITY \
|
||||
wandb.project=$WANDB_PROJECT
|
||||
|
||||
echo "Dev test completed!"
|
||||
@@ -42,11 +42,14 @@ echo "Experiment type: $EXPERIMENT_TYPE"
|
||||
# Run the experiment
|
||||
python reppo_alg/jaxrl/reppo.py \
|
||||
env=maniskill \
|
||||
env_name=$ENV_NAME \
|
||||
experiment_override=$EXPERIMENT_TYPE \
|
||||
env.name=$ENV_NAME \
|
||||
hyperparameters.num_envs=512 \
|
||||
hyperparameters.num_steps=64 \
|
||||
hyperparameters.num_mini_batches=64 \
|
||||
hyperparameters.num_epochs=4 \
|
||||
hyperparameters.total_time_steps=10000000 \
|
||||
wandb.mode=online \
|
||||
wandb.entity=${WANDB_ENTITY} \
|
||||
wandb.project=$WANDB_PROJECT \
|
||||
wandb.name="reppo_${ENV_NAME}_${EXPERIMENT_TYPE}_${SLURM_JOB_ID}"
|
||||
wandb.project=$WANDB_PROJECT
|
||||
|
||||
echo "Training completed!"
|
||||
Reference in New Issue
Block a user