feat(hpc): route caches to $VSC_SCRATCH, outputs to $VSC_DATA
- Install: redirect XDG_CACHE_HOME, UV_CACHE_DIR, MPLCONFIGDIR to
$VSC_SCRATCH before any pip/module activity ($VSC_HOME is only ~3 GB)
- train.pbs:
- Set same cache env vars at job start
- Write run outputs and WandB cache to $VSC_SCRATCH during the run (fast I/O)
- Copy results to $VSC_DATA/runs/<jobid>/ on completion for persistence
- Redirect PBS stdout/stderr to $VSC_DATA/runs/<jobid>/job.{out,err}
- Add cd $PBS_O_WORKDIR as first step (HPC best practice)
This commit is contained in:
parent
8b068086c8
commit
7bcfa66685
2 changed files with 81 additions and 12 deletions
|
|
@ -6,22 +6,40 @@
|
||||||
# bash scripts/hpc/install.sh
|
# bash scripts/hpc/install.sh
|
||||||
#
|
#
|
||||||
# Prerequisites:
|
# Prerequisites:
|
||||||
# - Connected to VSC via the web portal (HPC Login → Shell tmux)
|
# - Connected to VSC via the web portal (HPC Login → Interactive Apps > Shell tmux)
|
||||||
# - Project cloned to $VSC_DATA or $VSC_HOME
|
# - Set cluster to "donphan (interactive/debug)"
|
||||||
# - Run from the project root directory
|
# - Project cloned into $VSC_HOME (e.g. git clone <repo> && cd <repo>)
|
||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Redirect caches away from $VSC_HOME (very limited at ~3 GB) to scratch.
|
||||||
|
# This must be done before any pip/module activity.
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
export XDG_CACHE_HOME="$VSC_SCRATCH/.cache"
|
||||||
|
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
||||||
|
export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib"
|
||||||
|
mkdir -p "$XDG_CACHE_HOME" "$UV_CACHE_DIR" "$MPLCONFIGDIR"
|
||||||
|
|
||||||
|
echo "==> Loading vsc-venv module..."
|
||||||
module load vsc-venv
|
module load vsc-venv
|
||||||
|
|
||||||
|
echo "==> Activating virtual environment (created per-cluster in \$VSC_DATA)..."
|
||||||
|
# vsc-venv transparently creates and manages a per-cluster venv in $VSC_DATA,
|
||||||
|
# loading the modules listed in env/hpc/modules.txt and pip-installing anything
|
||||||
|
# in env/hpc/requirements.txt that is not already satisfied.
|
||||||
source vsc-venv --activate \
|
source vsc-venv --activate \
|
||||||
--modules env/hpc/modules.txt \
|
--modules env/hpc/modules.txt \
|
||||||
--requirements env/hpc/requirements.txt
|
--requirements env/hpc/requirements.txt
|
||||||
|
|
||||||
|
echo "==> Registering Jupyter kernel..."
|
||||||
python -m ipykernel install --user --name="sel3_${VSC_INSTITUTE_CLUSTER}" \
|
python -m ipykernel install --user --name="sel3_${VSC_INSTITUTE_CLUSTER}" \
|
||||||
--display-name "SEL3 (${VSC_INSTITUTE_CLUSTER})"
|
--display-name "SEL3 (${VSC_INSTITUTE_CLUSTER})"
|
||||||
|
|
||||||
echo ""
|
echo ""
|
||||||
echo "✅ Environment ready. Kernel: sel3_${VSC_INSTITUTE_CLUSTER}"
|
echo "✅ Environment ready. Kernel: sel3_${VSC_INSTITUTE_CLUSTER}"
|
||||||
echo " Activate in future sessions with:"
|
echo " To activate in future sessions:"
|
||||||
echo " module load vsc-venv && source vsc-venv --activate --modules env/hpc/modules.txt"
|
echo " module load vsc-venv && source vsc-venv --activate --modules env/hpc/modules.txt"
|
||||||
|
echo ""
|
||||||
|
echo " Remember to also set cache dirs in any interactive session:"
|
||||||
|
echo " export XDG_CACHE_HOME=\$VSC_SCRATCH/.cache"
|
||||||
|
|
|
||||||
|
|
@ -1,29 +1,80 @@
|
||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
# scripts/hpc/train.pbs — Submit PPO training as a batch job with qsub.
|
# scripts/hpc/train.pbs — Submit PPO training as a batch job with qsub.
|
||||||
#
|
#
|
||||||
# Usage (from project root):
|
# Usage (from project root after cloning into $VSC_HOME):
|
||||||
# qsub scripts/hpc/train.pbs
|
# qsub scripts/hpc/train.pbs
|
||||||
#
|
#
|
||||||
# To target a specific GPU cluster (default: joltik):
|
# To target a different GPU cluster (default: joltik):
|
||||||
# module swap cluster/accelgor && qsub scripts/hpc/train.pbs
|
# module swap cluster/accelgor && qsub scripts/hpc/train.pbs
|
||||||
|
#
|
||||||
|
# Available GPU clusters: joltik, accelgor, litleo
|
||||||
|
# Debug / interactive: doduo (CPU) or donphan (interactive GPU)
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# PBS job directives
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
#PBS -N brittlestar-ppo
|
#PBS -N brittlestar-ppo
|
||||||
#PBS -l nodes=1:ppn=8:gpus=1
|
#PBS -l nodes=1:ppn=8:gpus=1
|
||||||
#PBS -l walltime=24:00:00
|
#PBS -l walltime=24:00:00
|
||||||
#PBS -l mem=32gb
|
#PBS -l mem=32gb
|
||||||
#PBS -o runs/pbs_${PBS_JOBID}.out
|
# Logs are redirected inside the script to $VSC_DATA so they persist.
|
||||||
#PBS -e runs/pbs_${PBS_JOBID}.err
|
#PBS -o /dev/null
|
||||||
|
#PBS -e /dev/null
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
|
# Always change to the directory from which qsub was called (best practice).
|
||||||
cd "$PBS_O_WORKDIR"
|
cd "$PBS_O_WORKDIR"
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Storage routing (see https://docs.hpc.ugent.be/Linux/running_jobs_with_input_output_data/)
|
||||||
|
#
|
||||||
|
# $VSC_HOME (~3 GB) — project source code only; no writes during jobs
|
||||||
|
# $VSC_DATA (~25 GB) — persistent run artefacts (models, final logs)
|
||||||
|
# $VSC_SCRATCH (large) — fast I/O during the job; caches; intermediate files
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
# Redirect Python / pip / uv caches away from $VSC_HOME to scratch (fast I/O).
|
||||||
|
export XDG_CACHE_HOME="$VSC_SCRATCH/.cache"
|
||||||
|
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
||||||
|
export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib"
|
||||||
|
mkdir -p "$XDG_CACHE_HOME" "$UV_CACHE_DIR" "$MPLCONFIGDIR"
|
||||||
|
|
||||||
|
# Run outputs: write to $VSC_SCRATCH during the job (fast), then copy to
|
||||||
|
# $VSC_DATA at the end for long-term storage.
|
||||||
|
RUN_ID="brittlestar_${PBS_JOBID}"
|
||||||
|
SCRATCH_RUNDIR="$VSC_SCRATCH/runs/$RUN_ID"
|
||||||
|
DATA_RUNDIR="$VSC_DATA/runs/$RUN_ID"
|
||||||
|
mkdir -p "$SCRATCH_RUNDIR" "$DATA_RUNDIR"
|
||||||
|
|
||||||
|
# PBS stdout/stderr: redirect into the data directory for persistence.
|
||||||
|
exec 1> "$DATA_RUNDIR/job.out" 2> "$DATA_RUNDIR/job.err"
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Environment
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
module load vsc-venv
|
module load vsc-venv
|
||||||
source vsc-venv --activate --modules env/hpc/modules.txt
|
source vsc-venv --activate --modules env/hpc/modules.txt
|
||||||
|
|
||||||
# EGL is required for headless MuJoCo operation — both for the physics
|
# EGL is required for headless MuJoCo — both the physics sim loop and
|
||||||
# simulation loop and for rendering videos to disk. Without this, MuJoCo
|
# on-demand video rendering to disk. Without this, MuJoCo tries to open an
|
||||||
# attempts to open an X11 display and fails on compute nodes.
|
# X11 display and fails on compute nodes.
|
||||||
export MUJOCO_GL=egl
|
export MUJOCO_GL=egl
|
||||||
|
|
||||||
python src/train.py --config configs/production_training.yaml
|
# Tell WandB to store its local run cache on scratch (fast, large).
|
||||||
|
export WANDB_DIR="$SCRATCH_RUNDIR"
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Run
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
python src/train.py \
|
||||||
|
--config configs/production_training.yaml \
|
||||||
|
--run_dir "$SCRATCH_RUNDIR"
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Stage out: copy results to persistent VSC_DATA storage.
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
echo "Copying results from scratch to \$VSC_DATA..."
|
||||||
|
cp -r "$SCRATCH_RUNDIR/." "$DATA_RUNDIR/"
|
||||||
|
echo "Done. Results at: $DATA_RUNDIR"
|
||||||
|
|
|
||||||
Reference in a new issue