diff --git a/examples/multi_node_examples/README.md b/examples/multi_node_examples/README.md index aa787d3f..047ce9bd 100644 --- a/examples/multi_node_examples/README.md +++ b/examples/multi_node_examples/README.md @@ -4,7 +4,17 @@ To run this demo which launches a single job that trains on 2 nodes (2 gpus per 1. Log into the jumphost node of your SLURM-managed cluster. 2. Create a conda environment with Lightning and a GPU PyTorch version. -3. Submit this script. +3. Choose a script to submit + +#### DDP +Submit this job to run with distributedDataParallel (2 nodes, 2 gpus each) ```bash -sbatch job_submit.sh YourEnv +sbatch ddp_job_submit.sh YourEnv +``` + +#### DDP2 +Submit this job to run with a different implementation of distributedDataParallel. +In this version, each node acts like DataParallel but syncs across nodes like DDP. +```bash +sbatch ddp2_job_submit.sh YourEnv ``` diff --git a/examples/multi_node_examples/ddp2_job_submit.sh b/examples/multi_node_examples/ddp2_job_submit.sh new file mode 100755 index 00000000..593a4b14 --- /dev/null +++ b/examples/multi_node_examples/ddp2_job_submit.sh @@ -0,0 +1,27 @@ +#!/bin/bash -l + +# SLURM SUBMIT SCRIPT +#SBATCH --nodes=2 +#SBATCH --gres=gpu:2 +#SBATCH --ntasks-per-node=1 +#SBATCH --mem=0 +#SBATCH --time=0-02:00:00 + +# activate conda env +source activate $1 + +# ------------------------- +# debugging flags (optional) + export NCCL_DEBUG=INFO + export PYTHONFAULTHANDLER=1 + +# on your cluster you might need these: +# set the network interface +# export NCCL_SOCKET_IFNAME=^docker0,lo + +# might need the latest cuda +# module load NCCL/2.4.7-1-cuda.10.0 +# ------------------------- + +# run script from above +srun python3 multi_node_demo.py diff --git a/examples/multi_node_examples/job_submit.sh b/examples/multi_node_examples/ddp_job_submit.sh similarity index 100% rename from examples/multi_node_examples/job_submit.sh rename to examples/multi_node_examples/ddp_job_submit.sh diff --git a/examples/multi_node_examples/multi_node_ddp2_demo.py b/examples/multi_node_examples/multi_node_ddp2_demo.py new file mode 100644 index 00000000..244eb044 --- /dev/null +++ b/examples/multi_node_examples/multi_node_ddp2_demo.py @@ -0,0 +1,55 @@ +""" +Multi-node example (GPU) +""" +import os +import numpy as np +import torch + +from argparse import ArgumentParser +from pytorch_lightning import Trainer +from examples.basic_examples.lightning_module_template import LightningTemplateModel + +SEED = 2334 +torch.manual_seed(SEED) +np.random.seed(SEED) + + +def main(hparams): + """ + Main training routine specific for this project + :param hparams: + :return: + """ + # ------------------------ + # 1 INIT LIGHTNING MODEL + # ------------------------ + model = LightningTemplateModel(hparams) + + # ------------------------ + # 2 INIT TRAINER + # ------------------------ + trainer = Trainer( + gpus=2, + nb_gpu_nodes=2, + distributed_backend='ddp2' + ) + + # ------------------------ + # 3 START TRAINING + # ------------------------ + trainer.fit(model) + + +if __name__ == '__main__': + + root_dir = os.path.dirname(os.path.realpath(__file__)) + parent_parser = ArgumentParser(add_help=False) + + # each LightningModule defines arguments relevant to it + parser = LightningTemplateModel.add_model_specific_args(parent_parser, root_dir) + hyperparams = parser.parse_args() + + # --------------------- + # RUN TRAINING + # --------------------- + main(hyperparams) diff --git a/examples/multi_node_examples/multi_node_demo.py b/examples/multi_node_examples/multi_node_ddp_demo.py similarity index 100% rename from examples/multi_node_examples/multi_node_demo.py rename to examples/multi_node_examples/multi_node_ddp_demo.py