# Fast PPO Configuration # Lower timestep count for quick iterations/testing. learning_rate: 0.0005 total_timesteps: 1024 num_envs: 32 num_steps: 32 anneal_lr: true gamma: 0.99 gae_lambda: 0.95 num_minibatches: 4 update_epochs: 4 norm_adv: true clip_coef: 0.2 clip_vloss: true ent_coef: 0.01 vf_coef: 0.5 max_grad_norm: 0.5 target_kl: null