feat(hpc): route caches to $VSC_SCRATCH, outputs to $VSC_DATA
- Install: redirect XDG_CACHE_HOME, UV_CACHE_DIR, MPLCONFIGDIR to
$VSC_SCRATCH before any pip/module activity ($VSC_HOME is only ~3 GB)
- train.pbs:
- Set same cache env vars at job start
- Write run outputs and WandB cache to $VSC_SCRATCH during the run (fast I/O)
- Copy results to $VSC_DATA/runs/<jobid>/ on completion for persistence
- Redirect PBS stdout/stderr to $VSC_DATA/runs/<jobid>/job.{out,err}
- Add cd $PBS_O_WORKDIR as first step (HPC best practice)
This commit is contained in:
parent
8b068086c8
commit
7bcfa66685
2 changed files with 81 additions and 12 deletions
|
|
@ -6,22 +6,40 @@
|
|||
# bash scripts/hpc/install.sh
|
||||
#
|
||||
# Prerequisites:
|
||||
# - Connected to VSC via the web portal (HPC Login → Shell tmux)
|
||||
# - Project cloned to $VSC_DATA or $VSC_HOME
|
||||
# - Run from the project root directory
|
||||
# - Connected to VSC via the web portal (HPC Login → Interactive Apps > Shell tmux)
|
||||
# - Set cluster to "donphan (interactive/debug)"
|
||||
# - Project cloned into $VSC_HOME (e.g. git clone <repo> && cd <repo>)
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Redirect caches away from $VSC_HOME (very limited at ~3 GB) to scratch.
|
||||
# This must be done before any pip/module activity.
|
||||
# ---------------------------------------------------------------------------
|
||||
export XDG_CACHE_HOME="$VSC_SCRATCH/.cache"
|
||||
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
||||
export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib"
|
||||
mkdir -p "$XDG_CACHE_HOME" "$UV_CACHE_DIR" "$MPLCONFIGDIR"
|
||||
|
||||
echo "==> Loading vsc-venv module..."
|
||||
module load vsc-venv
|
||||
|
||||
echo "==> Activating virtual environment (created per-cluster in \$VSC_DATA)..."
|
||||
# vsc-venv transparently creates and manages a per-cluster venv in $VSC_DATA,
|
||||
# loading the modules listed in env/hpc/modules.txt and pip-installing anything
|
||||
# in env/hpc/requirements.txt that is not already satisfied.
|
||||
source vsc-venv --activate \
|
||||
--modules env/hpc/modules.txt \
|
||||
--requirements env/hpc/requirements.txt
|
||||
|
||||
echo "==> Registering Jupyter kernel..."
|
||||
python -m ipykernel install --user --name="sel3_${VSC_INSTITUTE_CLUSTER}" \
|
||||
--display-name "SEL3 (${VSC_INSTITUTE_CLUSTER})"
|
||||
|
||||
echo ""
|
||||
echo "✅ Environment ready. Kernel: sel3_${VSC_INSTITUTE_CLUSTER}"
|
||||
echo " Activate in future sessions with:"
|
||||
echo " To activate in future sessions:"
|
||||
echo " module load vsc-venv && source vsc-venv --activate --modules env/hpc/modules.txt"
|
||||
echo ""
|
||||
echo " Remember to also set cache dirs in any interactive session:"
|
||||
echo " export XDG_CACHE_HOME=\$VSC_SCRATCH/.cache"
|
||||
|
|
|
|||
|
|
@ -1,29 +1,80 @@
|
|||
#!/bin/bash
|
||||
# scripts/hpc/train.pbs — Submit PPO training as a batch job with qsub.
|
||||
#
|
||||
# Usage (from project root):
|
||||
# Usage (from project root after cloning into $VSC_HOME):
|
||||
# qsub scripts/hpc/train.pbs
|
||||
#
|
||||
# To target a specific GPU cluster (default: joltik):
|
||||
# To target a different GPU cluster (default: joltik):
|
||||
# module swap cluster/accelgor && qsub scripts/hpc/train.pbs
|
||||
#
|
||||
# Available GPU clusters: joltik, accelgor, litleo
|
||||
# Debug / interactive: doduo (CPU) or donphan (interactive GPU)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# PBS job directives
|
||||
# ---------------------------------------------------------------------------
|
||||
#PBS -N brittlestar-ppo
|
||||
#PBS -l nodes=1:ppn=8:gpus=1
|
||||
#PBS -l walltime=24:00:00
|
||||
#PBS -l mem=32gb
|
||||
#PBS -o runs/pbs_${PBS_JOBID}.out
|
||||
#PBS -e runs/pbs_${PBS_JOBID}.err
|
||||
# Logs are redirected inside the script to $VSC_DATA so they persist.
|
||||
#PBS -o /dev/null
|
||||
#PBS -e /dev/null
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Always change to the directory from which qsub was called (best practice).
|
||||
cd "$PBS_O_WORKDIR"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Storage routing (see https://docs.hpc.ugent.be/Linux/running_jobs_with_input_output_data/)
|
||||
#
|
||||
# $VSC_HOME (~3 GB) — project source code only; no writes during jobs
|
||||
# $VSC_DATA (~25 GB) — persistent run artefacts (models, final logs)
|
||||
# $VSC_SCRATCH (large) — fast I/O during the job; caches; intermediate files
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Redirect Python / pip / uv caches away from $VSC_HOME to scratch (fast I/O).
|
||||
export XDG_CACHE_HOME="$VSC_SCRATCH/.cache"
|
||||
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
||||
export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib"
|
||||
mkdir -p "$XDG_CACHE_HOME" "$UV_CACHE_DIR" "$MPLCONFIGDIR"
|
||||
|
||||
# Run outputs: write to $VSC_SCRATCH during the job (fast), then copy to
|
||||
# $VSC_DATA at the end for long-term storage.
|
||||
RUN_ID="brittlestar_${PBS_JOBID}"
|
||||
SCRATCH_RUNDIR="$VSC_SCRATCH/runs/$RUN_ID"
|
||||
DATA_RUNDIR="$VSC_DATA/runs/$RUN_ID"
|
||||
mkdir -p "$SCRATCH_RUNDIR" "$DATA_RUNDIR"
|
||||
|
||||
# PBS stdout/stderr: redirect into the data directory for persistence.
|
||||
exec 1> "$DATA_RUNDIR/job.out" 2> "$DATA_RUNDIR/job.err"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Environment
|
||||
# ---------------------------------------------------------------------------
|
||||
module load vsc-venv
|
||||
source vsc-venv --activate --modules env/hpc/modules.txt
|
||||
|
||||
# EGL is required for headless MuJoCo operation — both for the physics
|
||||
# simulation loop and for rendering videos to disk. Without this, MuJoCo
|
||||
# attempts to open an X11 display and fails on compute nodes.
|
||||
# EGL is required for headless MuJoCo — both the physics sim loop and
|
||||
# on-demand video rendering to disk. Without this, MuJoCo tries to open an
|
||||
# X11 display and fails on compute nodes.
|
||||
export MUJOCO_GL=egl
|
||||
|
||||
python src/train.py --config configs/production_training.yaml
|
||||
# Tell WandB to store its local run cache on scratch (fast, large).
|
||||
export WANDB_DIR="$SCRATCH_RUNDIR"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Run
|
||||
# ---------------------------------------------------------------------------
|
||||
python src/train.py \
|
||||
--config configs/production_training.yaml \
|
||||
--run_dir "$SCRATCH_RUNDIR"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage out: copy results to persistent VSC_DATA storage.
|
||||
# ---------------------------------------------------------------------------
|
||||
echo "Copying results from scratch to \$VSC_DATA..."
|
||||
cp -r "$SCRATCH_RUNDIR/." "$DATA_RUNDIR/"
|
||||
echo "Done. Results at: $DATA_RUNDIR"
|
||||
|
|
|
|||
Reference in a new issue