refactor(hpc): use explicit PIP_CACHE_DIR/UV_CACHE_DIR, drop prose comments
Match comment style from previous cleanup commit: - No section divider bars, no echo progress lines, no prose comments on obvious commands - Replace XDG_CACHE_HOME with explicit PIP_CACHE_DIR and UV_CACHE_DIR - Drop MPLCONFIGDIR (matplotlib cache is negligible in size)
This commit is contained in:
parent
cd2e594d1b
commit
34a887cd58
3 changed files with 13 additions and 74 deletions
|
|
@ -81,9 +81,8 @@ module swap cluster/donphan
|
||||||
qsub -I -l nodes=1:ppn=4 -l walltime=1:00:00
|
qsub -I -l nodes=1:ppn=4 -l walltime=1:00:00
|
||||||
|
|
||||||
# Once inside the job — redirect caches to scratch first
|
# Once inside the job — redirect caches to scratch first
|
||||||
export XDG_CACHE_HOME="$VSC_SCRATCH/.cache"
|
export PIP_CACHE_DIR="$VSC_SCRATCH/.cache/pip"
|
||||||
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
||||||
export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib"
|
|
||||||
|
|
||||||
# Change to the project directory and activate environment
|
# Change to the project directory and activate environment
|
||||||
cd "$PBS_O_WORKDIR"
|
cd "$PBS_O_WORKDIR"
|
||||||
|
|
|
||||||
|
|
@ -1,45 +1,21 @@
|
||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
# scripts/hpc/install.sh — Run once on a login node (donphan) to set up the
|
# scripts/hpc/install.sh
|
||||||
# project's Python virtual environment using the official vsc-venv wrapper.
|
|
||||||
#
|
#
|
||||||
# Usage (from project root):
|
# Usage (from project root):
|
||||||
# bash scripts/hpc/install.sh
|
# bash scripts/hpc/install.sh
|
||||||
#
|
|
||||||
# Prerequisites:
|
|
||||||
# - Connected to VSC via the web portal (HPC Login → Interactive Apps > Shell tmux)
|
|
||||||
# - Set cluster to "donphan (interactive/debug)"
|
|
||||||
# - Project cloned into $VSC_HOME (e.g. git clone <repo> && cd <repo>)
|
|
||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Keep caches off $VSC_HOME (quota ~3 GB).
|
||||||
# Redirect caches away from $VSC_HOME (very limited at ~3 GB) to scratch.
|
export PIP_CACHE_DIR="$VSC_SCRATCH/.cache/pip"
|
||||||
# This must be done before any pip/module activity.
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
export XDG_CACHE_HOME="$VSC_SCRATCH/.cache"
|
|
||||||
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
||||||
export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib"
|
mkdir -p "$PIP_CACHE_DIR" "$UV_CACHE_DIR"
|
||||||
mkdir -p "$XDG_CACHE_HOME" "$UV_CACHE_DIR" "$MPLCONFIGDIR"
|
|
||||||
|
|
||||||
echo "==> Loading vsc-venv module..."
|
|
||||||
module load vsc-venv
|
module load vsc-venv
|
||||||
|
|
||||||
echo "==> Activating virtual environment (created per-cluster in \$VSC_DATA)..."
|
|
||||||
# vsc-venv transparently creates and manages a per-cluster venv in $VSC_DATA,
|
|
||||||
# loading the modules listed in env/hpc/modules.txt and pip-installing anything
|
|
||||||
# in env/hpc/requirements.txt that is not already satisfied.
|
|
||||||
source vsc-venv --activate \
|
source vsc-venv --activate \
|
||||||
--modules env/hpc/modules.txt \
|
--modules env/hpc/modules.txt \
|
||||||
--requirements env/hpc/requirements.txt
|
--requirements env/hpc/requirements.txt
|
||||||
|
|
||||||
echo "==> Registering Jupyter kernel..."
|
|
||||||
python -m ipykernel install --user --name="sel3_${VSC_INSTITUTE_CLUSTER}" \
|
python -m ipykernel install --user --name="sel3_${VSC_INSTITUTE_CLUSTER}" \
|
||||||
--display-name "SEL3 (${VSC_INSTITUTE_CLUSTER})"
|
--display-name "SEL3 (${VSC_INSTITUTE_CLUSTER})"
|
||||||
|
|
||||||
echo ""
|
|
||||||
echo "✅ Environment ready. Kernel: sel3_${VSC_INSTITUTE_CLUSTER}"
|
|
||||||
echo " To activate in future sessions:"
|
|
||||||
echo " module load vsc-venv && source vsc-venv --activate --modules env/hpc/modules.txt"
|
|
||||||
echo ""
|
|
||||||
echo " Remember to also set cache dirs in any interactive session:"
|
|
||||||
echo " export XDG_CACHE_HOME=\$VSC_SCRATCH/.cache"
|
|
||||||
|
|
|
||||||
|
|
@ -1,80 +1,44 @@
|
||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
# scripts/hpc/train.pbs — Submit PPO training as a batch job with qsub.
|
# scripts/hpc/train.pbs
|
||||||
#
|
#
|
||||||
# Usage (from project root after cloning into $VSC_HOME):
|
# Usage (from project root):
|
||||||
# qsub scripts/hpc/train.pbs
|
# qsub scripts/hpc/train.pbs
|
||||||
#
|
#
|
||||||
# To target a different GPU cluster (default: joltik):
|
# To target a specific GPU cluster (default: joltik):
|
||||||
# module swap cluster/accelgor && qsub scripts/hpc/train.pbs
|
# module swap cluster/accelgor && qsub scripts/hpc/train.pbs
|
||||||
#
|
|
||||||
# Available GPU clusters: joltik, accelgor, litleo
|
|
||||||
# Debug / interactive: doduo (CPU) or donphan (interactive GPU)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# PBS job directives
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
#PBS -N brittlestar-ppo
|
#PBS -N brittlestar-ppo
|
||||||
#PBS -l nodes=1:ppn=8:gpus=1
|
#PBS -l nodes=1:ppn=8:gpus=1
|
||||||
#PBS -l walltime=24:00:00
|
#PBS -l walltime=24:00:00
|
||||||
#PBS -l mem=32gb
|
#PBS -l mem=32gb
|
||||||
# Logs are redirected inside the script to $VSC_DATA so they persist.
|
|
||||||
#PBS -o /dev/null
|
#PBS -o /dev/null
|
||||||
#PBS -e /dev/null
|
#PBS -e /dev/null
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
# Always change to the directory from which qsub was called (best practice).
|
|
||||||
cd "$PBS_O_WORKDIR"
|
cd "$PBS_O_WORKDIR"
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Keep caches off $VSC_HOME (quota ~3 GB).
|
||||||
# Storage routing (see https://docs.hpc.ugent.be/Linux/running_jobs_with_input_output_data/)
|
export PIP_CACHE_DIR="$VSC_SCRATCH/.cache/pip"
|
||||||
#
|
|
||||||
# $VSC_HOME (~3 GB) — project source code only; no writes during jobs
|
|
||||||
# $VSC_DATA (~25 GB) — persistent run artefacts (models, final logs)
|
|
||||||
# $VSC_SCRATCH (large) — fast I/O during the job; caches; intermediate files
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
# Redirect Python / pip / uv caches away from $VSC_HOME to scratch (fast I/O).
|
|
||||||
export XDG_CACHE_HOME="$VSC_SCRATCH/.cache"
|
|
||||||
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv"
|
||||||
export MPLCONFIGDIR="$VSC_SCRATCH/.config/matplotlib"
|
|
||||||
mkdir -p "$XDG_CACHE_HOME" "$UV_CACHE_DIR" "$MPLCONFIGDIR"
|
|
||||||
|
|
||||||
# Run outputs: write to $VSC_SCRATCH during the job (fast), then copy to
|
# Write run outputs to scratch during the job (fast I/O),
|
||||||
# $VSC_DATA at the end for long-term storage.
|
# then stage out to $VSC_DATA at the end for persistence.
|
||||||
RUN_ID="brittlestar_${PBS_JOBID}"
|
RUN_ID="brittlestar_${PBS_JOBID}"
|
||||||
SCRATCH_RUNDIR="$VSC_SCRATCH/runs/$RUN_ID"
|
SCRATCH_RUNDIR="$VSC_SCRATCH/runs/$RUN_ID"
|
||||||
DATA_RUNDIR="$VSC_DATA/runs/$RUN_ID"
|
DATA_RUNDIR="$VSC_DATA/runs/$RUN_ID"
|
||||||
mkdir -p "$SCRATCH_RUNDIR" "$DATA_RUNDIR"
|
mkdir -p "$PIP_CACHE_DIR" "$UV_CACHE_DIR" "$SCRATCH_RUNDIR" "$DATA_RUNDIR"
|
||||||
|
|
||||||
# PBS stdout/stderr: redirect into the data directory for persistence.
|
|
||||||
exec 1> "$DATA_RUNDIR/job.out" 2> "$DATA_RUNDIR/job.err"
|
exec 1> "$DATA_RUNDIR/job.out" 2> "$DATA_RUNDIR/job.err"
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Environment
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
module load vsc-venv
|
module load vsc-venv
|
||||||
source vsc-venv --activate --modules env/hpc/modules.txt
|
source vsc-venv --activate --modules env/hpc/modules.txt
|
||||||
|
|
||||||
# EGL is required for headless MuJoCo — both the physics sim loop and
|
|
||||||
# on-demand video rendering to disk. Without this, MuJoCo tries to open an
|
|
||||||
# X11 display and fails on compute nodes.
|
|
||||||
export MUJOCO_GL=egl
|
export MUJOCO_GL=egl
|
||||||
|
|
||||||
# Tell WandB to store its local run cache on scratch (fast, large).
|
|
||||||
export WANDB_DIR="$SCRATCH_RUNDIR"
|
export WANDB_DIR="$SCRATCH_RUNDIR"
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Run
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
python src/train.py \
|
python src/train.py \
|
||||||
--config configs/production_training.yaml \
|
--config configs/production_training.yaml \
|
||||||
--run_dir "$SCRATCH_RUNDIR"
|
--run_dir "$SCRATCH_RUNDIR"
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Stage out: copy results to persistent VSC_DATA storage.
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
echo "Copying results from scratch to \$VSC_DATA..."
|
|
||||||
cp -r "$SCRATCH_RUNDIR/." "$DATA_RUNDIR/"
|
cp -r "$SCRATCH_RUNDIR/." "$DATA_RUNDIR/"
|
||||||
echo "Done. Results at: $DATA_RUNDIR"
|
|
||||||
|
|
|
||||||
Reference in a new issue