# Stable PPO Configuration # Standard hyperparams with lower LR and larger batch. learning_rate: 0.0001 total_timesteps: 10000000 num_envs: 100 num_steps: 256 anneal_lr: true gamma: 0.99 gae_lambda: 0.95 num_minibatches: 8 update_epochs: 4 norm_adv: true clip_coef: 0.1 clip_vloss: true ent_coef: 0.01 vf_coef: 0.5 max_grad_norm: 0.5 target_kl: null