diff --git a/docs/LightningModule/RequiredTrainerInterface.md b/docs/LightningModule/RequiredTrainerInterface.md index d8ccae94..6308d569 100644 --- a/docs/LightningModule/RequiredTrainerInterface.md +++ b/docs/LightningModule/RequiredTrainerInterface.md @@ -145,7 +145,7 @@ def training_step(self, batch, batch_nb): output = { 'loss': loss, # required - 'progress': {'training_loss': loss, 'batch_nb': batch_nb} # optional + 'progress': {'training_loss': loss} # optional (MUST ALL BE TENSORS) } # return a dict diff --git a/docs/Trainer/Distributed training.md b/docs/Trainer/Distributed training.md index c895e247..d1c52bd9 100644 --- a/docs/Trainer/Distributed training.md +++ b/docs/Trainer/Distributed training.md @@ -74,6 +74,18 @@ First, install apex (if install fails, look [here](https://github.com/NVIDIA/ape ```bash $ git clone https://github.com/NVIDIA/apex $ cd apex + +# ------------------------ +# OPTIONAL: on your cluster you might need to load cuda 10 or 9 +# depending on how you installed PyTorch + +# see available modules +module avail + +# load correct cuda before install +module load cuda-10.0 +# ------------------------ + $ pip install -v --no-cache-dir --global-option="--cpp_ext" --global-option="--cuda_ext" ./ ```