From 15319890a62b2fd5a13d8ccfbf34bd59d298c888 Mon Sep 17 00:00:00 2001 From: Tibo De Peuter Date: Sat, 4 Apr 2026 22:01:21 +0200 Subject: [PATCH] fix(hpc): streamlined install --- docs/HPC.md | 134 ++++++++++++++-------------------------- env/hpc/modules.txt | 1 + env/hpc/modules_gpu.txt | 1 - pyproject.toml | 1 + scripts/hpc/install.sh | 39 +++++------- scripts/hpc/train.pbs | 14 +++-- uv.lock | 2 + 7 files changed, 74 insertions(+), 118 deletions(-) delete mode 100644 env/hpc/modules_gpu.txt diff --git a/docs/HPC.md b/docs/HPC.md index e2af138..81a7a82 100644 --- a/docs/HPC.md +++ b/docs/HPC.md @@ -2,125 +2,83 @@ Full documentation: -## Cluster Selection - -Choose the appropriate cluster before submitting a job with `module swap cluster/`. The default login cluster is **doduo**. - -> Check the current queue load at . - ## Storage Overview -- Run outputs are written to `$VSC_SCRATCH` during the job (fast I/O) and copied to `$VSC_DATA` at the end for persistence. -- `$VSC_SCRATCH` may be purged periodically — do not use it as long-term storage. - -Check your quota: (Usage section). +- **Run Outputs**: Written to `$VSC_SCRATCH` during the job (fast I/O) and copied to `$VSC_DATA` at the end for persistence. +- **Virtual Environments**: Managed on **`$VSC_DATA`** by mirroring configuration files. This avoids the 3GB home quota without requiring symlinks in the project root. ## Initial Environment Setup -Run **once** after cloning the repository, but ensure you are logged into a **compute node** on the target cluster (e.g., `donphan` or `joltik`). The login node (`doduo`) will block the installation script to prevent architecture mismatches. - -```bash -# 1. Swap to the target cluster -module swap cluster/joltik - -# 2. Start an interactive session on a compute node -qsub -I -l nodes=1:ppn=8:gpus=1 - -# 3. Navigate to the project directory and run the install script -cd "${PBS_O_WORKDIR}" -bash scripts/hpc/install.sh - -# 4. Exit the interactive session once finished -exit -``` +Run **once** after cloning the repository. Ensure you are logged into a **compute node** on `donphan` or `joltik`. > [!IMPORTANT] -> Virtual environments are cluster-specific. If you want to switch from `joltik` to `accelgor`, you must re-run the `install.sh` script while logged into an interactive session on `accelgor`. - -## Interactive Debugging on donphan - -### Interactive shell session +> To avoid the **3GB home directory quota limit**, the installation script mirrors your configuration files to **`$VSC_DATA`** (25GB+ quota). The `vsc-venv` tool then automatically creates and manages the environment on the larger partition. ```bash -# Swap to the debug cluster (from any login node) -module swap cluster/donphan +# 1. Start an interactive session (donphan for debug, joltik for training) +qsub -I -l nodes=1:ppn=8:gpus=1 -# Request an interactive job (1 node, 4 cores) -qsub -I -l nodes=1:ppn=4 -l walltime=1:00:00 - -# Once inside the job — redirect caches to scratch first -export PIP_CACHE_DIR="$VSC_SCRATCH/.cache/pip" -export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv" - -# Change to the project directory and activate environment -cd "$PBS_O_WORKDIR" -module load vsc-venv -source vsc-venv --activate \ - --modules env/hpc/modules.txt \ - --requirements env/hpc/requirements.txt - -# Set headless rendering backend -export MUJOCO_GL=egl - -# Run the smoke test -python src/train.py --env-config-path configs/hpc/smoke_test.yaml +# 2. Run the streamlined install script +cd "${PBS_O_WORKDIR}" +bash scripts/hpc/install.sh ``` -### JupyterLab session (HPC web portal) +## Interactive Debugging -1. In the web portal go to **Interactive Apps → JupyterLab RHEL9** -2. Set the following options: +You can use the same `install.sh` script to quickly activate your environment for interactive work. - | Option | Value | - |---------------------|-------------------------------------------------| - | Cluster | `donphan (interactive/debug)` | - | Number of nodes | 1 | - | Number of cores | 4 | - | JupyterLab version | `4.2.5 GCCcore-13.3.0` | - | Custom code | *(leave blank — vsc-venv handles modules)* | +```bash +# Request an interactive job +qsub -I -l nodes=1:ppn=4 -l walltime=1:00:00 -3. Click **Launch**, wait for the session to start, then **Connect**. -4. In JupyterLab, select the kernel **`SEL3 ()`**. -5. Verify GPU access: +# Change to project directory and run install.sh to sync and activate +cd "$PBS_O_WORKDIR" +bash scripts/hpc/install.sh +``` - ```python - import jax - print(jax.default_backend()) # expected: 'gpu' - print(jax.devices()) # expected: [CudaDevice(id=0)] +### Verification Commands + +After installation, run these commands to ensure your environment is set up correctly: + +1. **Verify Location**: + ```bash + # Confirm that NO 'venvs' folder appeared in your project root + ls -d venvs 2>/dev/null # Should return 'not found' + + # Confirm the environment is on the data partition + python -c "import torch; print(torch.__file__)" + # Expected: /kyukon/data/gent/vsc... or similar ``` -> **Warning:** JAX can only be loaded by one kernel at a time. Shut down other kernels before switching notebooks. +2. **Verify GPU Access**: + ```bash + python -c "import torch; import jax; print(f'Torch CUDA: {torch.cuda.is_available()}'); print(f'JAX Devices: {jax.devices()}')" + ``` + *Expected output: `Torch CUDA: True` and `JAX Devices: [CudaDevice(id=0)]`.* + +3. **Verify Home Quota**: + ```bash + df -h ~ # Should show low usage (< 1GB typically) + ``` ## Submitting Batch Training Jobs ```bash -# Default cluster (joltik — one A100 GPU slice) +# Submit to the default cluster (joltik) qsub scripts/hpc/train.pbs -# Switch to a different GPU cluster first -module swap cluster/accelgor -qsub scripts/hpc/train.pbs +# To choose a different cluster (e.g. donphan debug) without touching code +module swap cluster/donphan && qsub scripts/hpc/train.pbs ``` -The job script automatically: -- Writes run outputs to `$VSC_SCRATCH/runs/` during the run using the `--run_dir` argument. This ensures that frequent I/O (like tensorboard logs and checkpoints) happens on the fastest available filesystem. -- Copies the final results to `$VSC_DATA/runs/` on completion for long-term persistence. -- Writes PBS stdout/stderr to `runs/brittlestar-ppo.o` / `.e` (standard PBS convention, relative to the project root). - -Monitor your jobs: - -```bash -qstat # list your jobs -qstat -f # detailed info for a specific job -qdel # cancel a job -``` +The `train.pbs` script automatically handles its own activation using the mirrored configurations on `$VSC_DATA`. ## Managing Dependencies -`env/hpc/requirements.txt` is auto-generated by CI whenever `pyproject.toml` changes. To regenerate locally: +`env/hpc/requirements.txt` is auto-generated from `pyproject.toml`. To regenerate: ```bash uv run scripts/export_hpc_requirements.py ``` -Do **not** edit `env/hpc/requirements.txt` by hand — edit `pyproject.toml` instead. +Modules listed in `env/hpc/modules.txt` are automatically excluded from the pip requirements to save space and use HPC-optimized binaries. diff --git a/env/hpc/modules.txt b/env/hpc/modules.txt index 58ca6cd..c780513 100644 --- a/env/hpc/modules.txt +++ b/env/hpc/modules.txt @@ -1,5 +1,6 @@ gfbf/2024a GCCcore/13.3.0 Python/3.12.3-GCCcore-13.3.0 +PyTorch/2.7.1-foss-2024a-CUDA-12.6.0 FFmpeg/7.0.2-GCCcore-13.3.0 PyYAML/6.0.2-GCCcore-13.3.0 diff --git a/env/hpc/modules_gpu.txt b/env/hpc/modules_gpu.txt deleted file mode 100644 index aa9f2b2..0000000 --- a/env/hpc/modules_gpu.txt +++ /dev/null @@ -1 +0,0 @@ -PyTorch/2.7.1-foss-2024a-CUDA-12.6.0 diff --git a/pyproject.toml b/pyproject.toml index 8191bca..556593b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -23,6 +23,7 @@ dependencies = [ "pyopengl-accelerate>=3.1.10", "tyro>=1.0.10", "wandb==0.24.2", + "torch>=2.4.0", ] [project.optional-dependencies] diff --git a/scripts/hpc/install.sh b/scripts/hpc/install.sh index 93c7707..9fec7c6 100644 --- a/scripts/hpc/install.sh +++ b/scripts/hpc/install.sh @@ -1,38 +1,31 @@ #!/bin/bash # scripts/hpc/install.sh # -# Usage: -# module swap cluster/donphan -# qsub -I -l nodes=1:ppn=8:gpus=1 -# (Wait for session to start, then:) +# Usage (on any compute node): # bash scripts/hpc/install.sh set -eo pipefail -# Keep caches off $VSC_HOME (quota ~3 GB). -export PIP_CACHE_DIR="$VSC_SCRATCH/.cache/pip" -export UV_CACHE_DIR="$VSC_SCRATCH/.cache/uv" -mkdir -p "$PIP_CACHE_DIR" "$UV_CACHE_DIR" - -# Ensure we are in the project root -if [ -n "$PBS_O_WORKDIR" ]; then - cd "$PBS_O_WORKDIR" -fi +# Mirror configs to $VSC_DATA to avoid home quota limits (3GB) +# vsc-venv manages environments relative to the requirements file +HPC_CONFIG_DIR="$VSC_DATA/2026SEL3-project/env/hpc" +mkdir -p "$HPC_CONFIG_DIR" +cp env/hpc/*.txt "$HPC_CONFIG_DIR/" module load vsc-venv -echo 'Synchronizing and activating environment...' +echo ">>> Synchronizing and activating environment (vsc-venv)..." source vsc-venv --activate \ - --modules env/hpc/modules.txt \ - --requirements env/hpc/requirements.txt + --modules "$HPC_CONFIG_DIR/modules.txt" \ + --requirements "$HPC_CONFIG_DIR/requirements.txt" -# Step 2: Force upgrade shared system dependencies to ensure venv precedence -echo 'Applying library overlays (NumPy, Protobuf)...' +# Overlay specific NumPy/Protobuf versions to ensure venv precedence +echo ">>> Applying library overlays (NumPy, Protobuf)..." pip install --upgrade --no-deps numpy protobuf -echo 'Installing ipykernel...' -python -m ipykernel install --user --name="sel3_${VSC_INSTITUTE_CLUSTER}" \ - --display-name "SEL3 (${VSC_INSTITUTE_CLUSTER})" - -echo 'Done' +echo '>>> Installing ipykernel...' +CLUSTER_ID="${VSC_INSTITUTE_CLUSTER:-generic}" +python -m ipykernel install --user --name="sel3_${CLUSTER_ID}" \ + --display-name "SEL3 (${CLUSTER_ID})" +echo '>>> Done' diff --git a/scripts/hpc/train.pbs b/scripts/hpc/train.pbs index c5296b5..243e337 100644 --- a/scripts/hpc/train.pbs +++ b/scripts/hpc/train.pbs @@ -3,9 +3,6 @@ # # Usage (from project root): # qsub scripts/hpc/train.pbs -# -# To target a specific GPU cluster (default: joltik): -# module swap cluster/accelgor && qsub scripts/hpc/train.pbs #PBS -N brittlestar-ppo #PBS -l nodes=1:ppn=8:gpus=1 @@ -30,18 +27,23 @@ DATA_RUNDIR="$VSC_DATA/runs/$RUN_ID" mkdir -p "$PIP_CACHE_DIR" "$UV_CACHE_DIR" "$SCRATCH_RUNDIR" "$DATA_RUNDIR" runs/ module load vsc-venv + +# Activate mirrored environment from $VSC_DATA to avoid home quota +HPC_CONFIG_DIR="$VSC_DATA/2026SEL3-project/env/hpc" set +u source vsc-venv --activate \ - --modules env/hpc/modules.txt \ - --requirements env/hpc/requirements.txt + --modules "$HPC_CONFIG_DIR/modules.txt" \ + --requirements "$HPC_CONFIG_DIR/requirements.txt" set -u +# Ensure venv versions of NumPy/Protobuf take precedence +pip install --upgrade --no-deps numpy protobuf > /dev/null 2>&1 + export MUJOCO_GL=egl export WANDB_DIR="$SCRATCH_RUNDIR" python src/train.py \ --env-config-path configs/hpc/smoke_test.yaml \ --run-dir "$SCRATCH_RUNDIR" - # ^^^ Replace with your production config, e.g. configs/production_training.yaml cp -r "$SCRATCH_RUNDIR/." "$DATA_RUNDIR/" diff --git a/uv.lock b/uv.lock index f7ebb69..7fbaacb 100644 --- a/uv.lock +++ b/uv.lock @@ -28,6 +28,7 @@ dependencies = [ { name = "protobuf" }, { name = "pyopengl" }, { name = "pyopengl-accelerate" }, + { name = "torch" }, { name = "tyro" }, { name = "wandb" }, { name = "warp-lang" }, @@ -63,6 +64,7 @@ requires-dist = [ { name = "protobuf", specifier = ">=5.0.0" }, { name = "pyopengl", specifier = ">=3.1.10" }, { name = "pyopengl-accelerate", specifier = ">=3.1.10" }, + { name = "torch", specifier = ">=2.4.0" }, { name = "tyro", specifier = ">=1.0.10" }, { name = "wandb", specifier = "==0.24.2" }, { name = "warp-lang" },