diff --git a/scripts/hpc/install.sh b/scripts/hpc/install.sh index 543b2de..25063de 100644 --- a/scripts/hpc/install.sh +++ b/scripts/hpc/install.sh @@ -6,22 +6,40 @@ # bash scripts/hpc/install.sh # # Prerequisites: -# - Connected to VSC via the web portal (HPC Login → Shell tmux) -# - Project cloned to $VSC_DATA or $VSC_HOME -# - Run from the project root directory +# - Connected to VSC via the web portal (HPC Login → Interactive Apps > Shell tmux) +# - Set cluster to "donphan (interactive/debug)" +# - Project cloned into $VSC_HOME (e.g. git clone && cd ) set -euo pipefail +# --------------------------------------------------------------------------- +# Redirect caches away from $VSC_HOME (very limited at ~3 GB) to scratch. +# This must be done before any pip/module activity. +# --------------------------------------------------------------------------- +export XDG_CACHE_HOME="$VSC_SCRATCH/.cache" +export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv" +export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib" +mkdir -p "$XDG_CACHE_HOME" "$UV_CACHE_DIR" "$MPLCONFIGDIR" + +echo "==> Loading vsc-venv module..." module load vsc-venv +echo "==> Activating virtual environment (created per-cluster in \$VSC_DATA)..." +# vsc-venv transparently creates and manages a per-cluster venv in $VSC_DATA, +# loading the modules listed in env/hpc/modules.txt and pip-installing anything +# in env/hpc/requirements.txt that is not already satisfied. source vsc-venv --activate \ --modules env/hpc/modules.txt \ --requirements env/hpc/requirements.txt +echo "==> Registering Jupyter kernel..." python -m ipykernel install --user --name="sel3_${VSC_INSTITUTE_CLUSTER}" \ --display-name "SEL3 (${VSC_INSTITUTE_CLUSTER})" echo "" echo "✅ Environment ready. Kernel: sel3_${VSC_INSTITUTE_CLUSTER}" -echo " Activate in future sessions with:" +echo " To activate in future sessions:" echo " module load vsc-venv && source vsc-venv --activate --modules env/hpc/modules.txt" +echo "" +echo " Remember to also set cache dirs in any interactive session:" +echo " export XDG_CACHE_HOME=\$VSC_SCRATCH/.cache" diff --git a/scripts/hpc/train.pbs b/scripts/hpc/train.pbs index 981ea96..6f1d23a 100644 --- a/scripts/hpc/train.pbs +++ b/scripts/hpc/train.pbs @@ -1,29 +1,80 @@ #!/bin/bash # scripts/hpc/train.pbs — Submit PPO training as a batch job with qsub. # -# Usage (from project root): +# Usage (from project root after cloning into $VSC_HOME): # qsub scripts/hpc/train.pbs # -# To target a specific GPU cluster (default: joltik): +# To target a different GPU cluster (default: joltik): # module swap cluster/accelgor && qsub scripts/hpc/train.pbs +# +# Available GPU clusters: joltik, accelgor, litleo +# Debug / interactive: doduo (CPU) or donphan (interactive GPU) +# --------------------------------------------------------------------------- +# PBS job directives +# --------------------------------------------------------------------------- #PBS -N brittlestar-ppo #PBS -l nodes=1:ppn=8:gpus=1 #PBS -l walltime=24:00:00 #PBS -l mem=32gb -#PBS -o runs/pbs_${PBS_JOBID}.out -#PBS -e runs/pbs_${PBS_JOBID}.err +# Logs are redirected inside the script to $VSC_DATA so they persist. +#PBS -o /dev/null +#PBS -e /dev/null +# --------------------------------------------------------------------------- set -euo pipefail +# Always change to the directory from which qsub was called (best practice). cd "$PBS_O_WORKDIR" +# --------------------------------------------------------------------------- +# Storage routing (see https://docs.hpc.ugent.be/Linux/running_jobs_with_input_output_data/) +# +# $VSC_HOME (~3 GB) — project source code only; no writes during jobs +# $VSC_DATA (~25 GB) — persistent run artefacts (models, final logs) +# $VSC_SCRATCH (large) — fast I/O during the job; caches; intermediate files +# --------------------------------------------------------------------------- + +# Redirect Python / pip / uv caches away from $VSC_HOME to scratch (fast I/O). +export XDG_CACHE_HOME="$VSC_SCRATCH/.cache" +export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv" +export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib" +mkdir -p "$XDG_CACHE_HOME" "$UV_CACHE_DIR" "$MPLCONFIGDIR" + +# Run outputs: write to $VSC_SCRATCH during the job (fast), then copy to +# $VSC_DATA at the end for long-term storage. +RUN_ID="brittlestar_${PBS_JOBID}" +SCRATCH_RUNDIR="$VSC_SCRATCH/runs/$RUN_ID" +DATA_RUNDIR="$VSC_DATA/runs/$RUN_ID" +mkdir -p "$SCRATCH_RUNDIR" "$DATA_RUNDIR" + +# PBS stdout/stderr: redirect into the data directory for persistence. +exec 1> "$DATA_RUNDIR/job.out" 2> "$DATA_RUNDIR/job.err" + +# --------------------------------------------------------------------------- +# Environment +# --------------------------------------------------------------------------- module load vsc-venv source vsc-venv --activate --modules env/hpc/modules.txt -# EGL is required for headless MuJoCo operation — both for the physics -# simulation loop and for rendering videos to disk. Without this, MuJoCo -# attempts to open an X11 display and fails on compute nodes. +# EGL is required for headless MuJoCo — both the physics sim loop and +# on-demand video rendering to disk. Without this, MuJoCo tries to open an +# X11 display and fails on compute nodes. export MUJOCO_GL=egl -python src/train.py --config configs/production_training.yaml +# Tell WandB to store its local run cache on scratch (fast, large). +export WANDB_DIR="$SCRATCH_RUNDIR" + +# --------------------------------------------------------------------------- +# Run +# --------------------------------------------------------------------------- +python src/train.py \ + --config configs/production_training.yaml \ + --run_dir "$SCRATCH_RUNDIR" + +# --------------------------------------------------------------------------- +# Stage out: copy results to persistent VSC_DATA storage. +# --------------------------------------------------------------------------- +echo "Copying results from scratch to \$VSC_DATA..." +cp -r "$SCRATCH_RUNDIR/." "$DATA_RUNDIR/" +echo "Done. Results at: $DATA_RUNDIR"