diff --git a/configs/default_ppo.yaml b/configs/default_ppo.yaml deleted file mode 100644 index 1b06c3d..0000000 --- a/configs/default_ppo.yaml +++ /dev/null @@ -1,48 +0,0 @@ -# PPO Training Configuration Template -# -# This file provides an example configuration for PPO training. -# Copy this file and modify it for your specific experiments. -# -# Usage: -# python src/train.py --config-path configs/my_config.yaml -# Or override specific parameters: -# python src/train.py --learning-rate 0.001 --num-envs 32 - -# Experiment settings -exp_name: "brittle_star_ppo" -seed: 1 - -# Tracking settings -track: false # Set to true to enable WandB logging -wandb_project_name: "PPO-Modularity" -wandb_entity: "SEL3-2026-Groep-4" # Set to your WandB username or team name - -# Model saving -save_model: true -checkpoint_frequency: 100 # Save checkpoint every N iterations (0 = no checkpoints) - -# Environment settings -num_envs: 16 - -# Training hyperparameters -total_timesteps: 10000000 -learning_rate: 0.00025 -num_steps: 128 -anneal_lr: true - -# PPO specific -gamma: 0.99 -gae_lambda: 0.95 -num_minibatches: 4 -update_epochs: 4 -norm_adv: true -clip_coef: 0.1 -clip_vloss: true -ent_coef: 0.01 -vf_coef: 0.5 -max_grad_norm: 0.5 -target_kl: null - -# Hardware -cuda: true -torch_deterministic: true diff --git a/configs/dev_test.yaml b/configs/dev_test.yaml deleted file mode 100644 index b64d330..0000000 --- a/configs/dev_test.yaml +++ /dev/null @@ -1,42 +0,0 @@ -# Quick Development/Testing Configuration -# -# Fast configuration for development and testing with short runs. - -# Experiment settings -exp_name: "brittle_star_dev_test" -seed: 123 - -# Tracking settings - IMPORTANT: Set your own wandb_entity! -track: true -wandb_project_name: "PPO-Modularity-Dev" -wandb_entity: "SEL3-2026-Groep-4" # ⚠️ SET THIS TO YOUR WANDB USERNAME OR TEAM - -# Model saving -save_model: true -checkpoint_frequency: 10 # More frequent checkpoints for testing - -# Environment settings -num_envs: 4 # Smaller for faster iteration - -# Training hyperparameters - Fast/testing -total_timesteps: 100000 # Short run for testing -learning_rate: 0.001 # Higher learning rate for faster learning -num_steps: 64 # Shorter rollouts -anneal_lr: true - -# PPO specific - Optimized for quick results -gamma: 0.99 -gae_lambda: 0.95 -num_minibatches: 2 -update_epochs: 2 # Fewer epochs for speed -norm_adv: true -clip_coef: 0.1 -clip_vloss: true -ent_coef: 0.02 # Higher entropy for exploration -vf_coef: 0.5 -max_grad_norm: 0.5 -target_kl: null - -# Hardware -cuda: true -torch_deterministic: true \ No newline at end of file diff --git a/configs/example.yaml b/configs/example.yaml deleted file mode 100644 index baeb1b0..0000000 --- a/configs/example.yaml +++ /dev/null @@ -1,4 +0,0 @@ -morphology: - segments_per_arm: [4, 4] - use_p_control: true - use_torque_control: false diff --git a/configs/hpc/smoke_test.yaml b/configs/hpc/smoke_test.yaml deleted file mode 100644 index 1dd2fc0..0000000 --- a/configs/hpc/smoke_test.yaml +++ /dev/null @@ -1,11 +0,0 @@ -# Minimal config to verify HPC setup is functional. -# Run with: python scripts/train.py --config-path configs/hpc/smoke_test.yaml -exp_name: "hpc_smoke_test" -seed: 0 -track: false # Test WandB integration -capture_video: false # No rendering for smoke test -save_model: true # Test the end-of-training save routine -num_envs: 512 -total_timesteps: 65536 -num_steps: 128 -cuda: true diff --git a/configs/hpc/wandb_expand.yaml b/configs/hpc/wandb_expand.yaml deleted file mode 100644 index 432160f..0000000 --- a/configs/hpc/wandb_expand.yaml +++ /dev/null @@ -1,10 +0,0 @@ -exp_name: "explained_var_fun_more_steps" -seed: 42 -track: true -wandb_project_name: "LET-THERE-BE-MORE-LOGGING" -wandb_entity: "SEL3-2026-Groep-4" - -num_envs: 16 -num_steps: 256 -total_timesteps: 50000 -cuda: true \ No newline at end of file diff --git a/configs/hpc/wandb_test.yaml b/configs/hpc/wandb_test.yaml deleted file mode 100644 index bf2229d..0000000 --- a/configs/hpc/wandb_test.yaml +++ /dev/null @@ -1,11 +0,0 @@ -# Configuration to verify WandB online tracking -exp_name: "hpc_wandb_verification" -seed: 42 -track: true # Enabled for testing WandB -wandb_project_name: "PPO-Modularity" -wandb_entity: "SEL3-2026-Groep-4" - -num_envs: 128 -total_timesteps: 50000 # Short run for quick verification -num_steps: 128 -cuda: true diff --git a/configs/personal_template.yaml b/configs/personal_template.yaml deleted file mode 100644 index 67caef7..0000000 --- a/configs/personal_template.yaml +++ /dev/null @@ -1,40 +0,0 @@ -# Personal Configuration Example for Team Member -# -# Copy this template and customize for your personal experiments - -# Experiment settings - PERSONALIZE THESE -exp_name: "YOUR_NAME_experiment_v1" # ⚠️ Change YOUR_NAME -seed: 42 - -# WandB settings - ⚠️ IMPORTANT: Set your credentials! -track: true # Enable WandB tracking -wandb_project_name: "PPO-Modularity" -wandb_entity: "SEL3-2026-Groep-4" # ⚠️ CHANGE THIS to your WandB username/team - -# Quick experiment settings (modify as needed) -total_timesteps: 500000 # 500K for quick results -num_envs: 8 -learning_rate: 0.0005 -num_steps: 128 - -# Model saving -save_model: true -checkpoint_frequency: 25 # Save checkpoints frequently - -# Standard PPO settings (usually don't need to change) -gamma: 0.99 -gae_lambda: 0.95 -num_minibatches: 4 -update_epochs: 4 -norm_adv: true -clip_coef: 0.2 -clip_vloss: true -ent_coef: 0.01 -vf_coef: 0.5 -max_grad_norm: 0.5 -target_kl: null -anneal_lr: true - -# Hardware -cuda: true -torch_deterministic: true \ No newline at end of file diff --git a/configs/production_training.yaml b/configs/production_training.yaml deleted file mode 100644 index 6dce29b..0000000 --- a/configs/production_training.yaml +++ /dev/null @@ -1,42 +0,0 @@ -# Production Training Configuration -# -# Full-scale training configuration for production runs -# with wandb logging enabled. - -# Experiment settings -exp_name: "brittle_star_production_training" -seed: 42 - -# Tracking -track: true -capture_video: false -wandb_project_name: "PPO-Modularity" -wandb_entity: "SEL3-2026-Groep-4" - -# Model saving -save_model: true -checkpoint_frequency: 100 # Save checkpoint every 100 iterations - -# Environment settings -num_envs: 512 - -# Training hyperparameters -total_timesteps: 50000000 -num_steps: 256 -num_minibatches: 4 -update_epochs: 4 - -learning_rate: 2.5e-4 -anneal_lr: true -gamma: 0.99 -gae_lambda: 0.95 -clip_coef: 0.1 -clip_vloss: true -ent_coef: 0.01 -vf_coef: 0.5 -max_grad_norm: 0.5 -target_kl: null - -# Hardware -cuda: true -torch_deterministic: true \ No newline at end of file diff --git a/scripts/hpc/train.pbs b/scripts/hpc/train.pbs index b9c8519..60ba4d8 100644 --- a/scripts/hpc/train.pbs +++ b/scripts/hpc/train.pbs @@ -63,11 +63,11 @@ elif [ -f "$PBS_O_WORKDIR/.env" ]; then export $(grep -v '^#' "$PBS_O_WORKDIR/.env" | xargs) fi -# TODO Once experiments get serious, change the config +# Run training using Hydra overrides python scripts/train.py \ - --env-config-path configs/hpc/wandb_expand.yaml \ - --hyperparameter-config-path configs/hpc/wandb_expand.yaml \ - --run-dir "$SCRATCH_RUNDIR" + hydra.run.dir="$SCRATCH_RUNDIR" \ + ppo=stable \ + logging=wandb_enabled echo ">>> Staging out results to $DATA_RUNDIR..." cp -r "$SCRATCH_RUNDIR/." "$DATA_RUNDIR/" diff --git a/src/brittle_star_project/dataclasses/PPOArgs.py b/src/brittle_star_project/dataclasses/PPOArgs.py deleted file mode 100644 index 036b44f..0000000 --- a/src/brittle_star_project/dataclasses/PPOArgs.py +++ /dev/null @@ -1,116 +0,0 @@ -from dataclasses import dataclass - -import jax - - -@jax.tree_util.register_dataclass -@dataclass -class PPOArgs: - """ - source: https://github.com/vwxyzjn/cleanrl/blob/master/cleanrl/ppo_atari_envpool_xla_jax_scan.py - """ - - # path to environment config file, if None, use default config - env_config_path: str | None = None - - # path to hyperparameter config file (yaml), if None, use default config - hyperparameter_config_path: str | None = None - - # the name of this experiment - exp_name: str = "brittle_star_ppo" - - # the directory to save the experiment results - run_dir: str | None = None - - # seed of the experiment - seed: int = 1 - - # if toggled, `torch.backends.cudnn.deterministic=False` - torch_deterministic: bool = True - - # if toggled, cuda will be enabled by default - cuda: bool = True - - # if toggled, this experiment will be tracked with Weights and Biases - track: bool = False - - # the wandb's project name - wandb_project_name: str = "PPO-Modularity" - - # the entity (team) of wandb's project - wandb_entity: str | None = "SEL3-2026-Groep-4" - - # whether to capture videos of the agent performances (check out `videos` folder) - capture_video: bool = False - - # whether to save model into the `runs/{run_name}` folder - save_model: bool = True - - # checkpoint frequency (in iterations, 0 = no intermediate checkpoints) - checkpoint_frequency: int = 100 - - # whether to upload the saved model to huggingface - upload_model: bool = False - - # the user or org name of the model repository from the Hugging Face Hub - hf_entity: str = "" - - # ==== Algorithm specific dataclasses ==== - - # total timesteps of the experiments - total_timesteps: int = 10000000 - - # the learning rate of the optimizer - learning_rate: float = 2.5e-4 - - # the number of parallel game environments - num_envs: int = 100 - - # the number of steps to run in each environment per policy rollout - num_steps: int = 128 - - # Toggle learning rate annealing for policy and value networks - anneal_lr: bool = True - - # the discount factor gamma - gamma: float = 0.99 - - # the lambda for the general advantage estimation - gae_lambda: float = 0.95 - - # the number of mini-batches - num_minibatches: int = 4 - - # the K epochs to update the policy - update_epochs: int = 4 - - # Toggles advantages normalization - norm_adv: bool = True - - # the surrogate clipping coefficient - clip_coef: float = 0.1 - - # Toggles whether or not to use a clipped loss for the value function, as per the paper. - clip_vloss: bool = True - - # coefficient of the entropy - ent_coef: float = 0.01 - - # coefficient of the value function - vf_coef: float = 0.5 - - # the maximum norm for the gradient clipping - max_grad_norm: float = 0.5 - - # the target KL divergence threshold - target_kl: float | None = None - - # ==== to be filled in runtime ==== - # the batch size (computed in runtime) - batch_size: int = 0 - - # the mini-batch size (computed in runtime) - minibatch_size: int = 0 - - # the number of iterations (computed in runtime) - num_iterations: int = 0 diff --git a/src/brittle_star_project/dataclasses/__init__.py b/src/brittle_star_project/dataclasses/__init__.py index 3501b2b..f2d2ad1 100644 --- a/src/brittle_star_project/dataclasses/__init__.py +++ b/src/brittle_star_project/dataclasses/__init__.py @@ -1,8 +1,6 @@ -from .PPOArgs import PPOArgs from .EpisodeStatistics import EpisodeStatistics __all__ = [ - "PPOArgs", "EpisodeStatistics", ]