diff --git a/pytorch_lightning/trainer/train_loop_mixin.py b/pytorch_lightning/trainer/train_loop_mixin.py index e51a43d9..1dad28ae 100644 --- a/pytorch_lightning/trainer/train_loop_mixin.py +++ b/pytorch_lightning/trainer/train_loop_mixin.py @@ -74,23 +74,17 @@ class TrainerTrainLoopMixin(object): # run epoch for batch_nb, batch in enumerate(self.get_train_dataloader()): self.batch_nb = batch_nb - self.global_step += 1 model = self.get_model() model.global_step = self.global_step - # stop when the flag is changed or we've gone past the amount - # requested in the batches - self.total_batch_nb += 1 - met_batch_limit = batch_nb >= self.nb_training_batches - if met_batch_limit: - break - # --------------- # RUN TRAIN STEP # --------------- output = self.run_training_batch(batch, batch_nb) batch_result, grad_norm_dic, batch_step_metrics = output + + # when returning -1 from train_step, we end epoch early early_stop_epoch = batch_result == -1 # --------------- @@ -116,10 +110,20 @@ class TrainerTrainLoopMixin(object): # logs user requested information to logger self.log_metrics(batch_step_metrics, grad_norm_dic) + self.global_step += 1 + self.total_batch_nb += 1 + # end epoch early + # stop when the flag is changed or we've gone past the amount + # requested in the batches if early_stop_epoch or self.fast_dev_run: break + # stop epoch if we limited nb batches + met_batch_limit = batch_nb >= self.nb_training_batches + if met_batch_limit: + break + # epoch end hook if self.is_function_implemented('on_epoch_end'): model = self.get_model() diff --git a/pytorch_lightning/trainer/trainer.py b/pytorch_lightning/trainer/trainer.py index 067ffe9d..19f31034 100644 --- a/pytorch_lightning/trainer/trainer.py +++ b/pytorch_lightning/trainer/trainer.py @@ -447,6 +447,10 @@ class Trainer(TrainerIOMixin, self.evaluate(model, self.get_val_dataloaders(), self.nb_sanity_val_steps, self.testing) + # clear cache before training + if self.on_gpu: + torch.cuda.empty_cache() + # CORE TRAINING LOOP self.train() diff --git a/pytorch_lightning/trainer/trainer_io.py b/pytorch_lightning/trainer/trainer_io.py index aa40e21d..5b483760 100644 --- a/pytorch_lightning/trainer/trainer_io.py +++ b/pytorch_lightning/trainer/trainer_io.py @@ -25,27 +25,37 @@ class TrainerIOMixin(object): def restore_weights(self, model): """ To restore weights we have two cases. - First, if we use the same experiment version, then restore the latest ckpt. - AFTER that, if we find weights from hpc checkpoint, then restore that. + First, attempt to restore hpc weights. If successful, don't restore + other weights. + + Otherwise, try to restore actual weights :param model: :return: """ - # restore weights if same exp version - self.restore_state_if_checkpoint_exists(model) # if script called from hpc resubmit, load weights - self.restore_hpc_weights_if_needed(model) + did_restore_hpc_weights = self.restore_hpc_weights_if_needed(model) + + if not did_restore_hpc_weights: + # restore weights if same exp version + self.restore_state_if_checkpoint_exists(model) # wait for all models to restore weights if self.use_ddp or self.use_ddp2: # wait for all processes to catch up dist.barrier() + # clear cache after restore + if self.on_gpu: + torch.cuda.empty_cache() + def restore_state_if_checkpoint_exists(self, model): + did_restore = False + # do nothing if there's not dir or callback no_ckpt_callback = (self.checkpoint_callback is None) or (not self.checkpoint_callback) if no_ckpt_callback or not os.path.exists(self.checkpoint_callback.filepath): - return + return did_restore # restore trainer state and model if there is a weight for this experiment last_epoch = -1 @@ -71,6 +81,9 @@ class TrainerIOMixin(object): last_ckpt_path = os.path.join(self.checkpoint_callback.filepath, last_ckpt_name) self.restore(last_ckpt_path, self.on_gpu) print(f'model and trainer restored from checkpoint: {last_ckpt_path}') + did_restore = True + + return did_restore # -------------------- # HPC SIGNAL HANDLING @@ -198,6 +211,8 @@ class TrainerIOMixin(object): :param model: :return: """ + did_restore = False + # look for hpc weights folderpath = self.weights_save_path if os.path.exists(folderpath): @@ -207,6 +222,8 @@ class TrainerIOMixin(object): # if hpc weights exist restore model if len(hpc_weight_paths) > 0: self.hpc_load(folderpath, self.on_gpu) + did_restore = True + return did_restore def restore_training_state(self, checkpoint): """ diff --git a/tests/README.md b/tests/README.md index 032cf04f..045347f2 100644 --- a/tests/README.md +++ b/tests/README.md @@ -1,4 +1,7 @@ # PyTorch-Lightning Tests +Most PL tests train a full MNIST model under various trainer conditions (ddp, ddp2+amp, etc...). +This provides testing for most combinations of important settings. +The tests expect the model to perform to a reasonable degree of testing accuracy to pass. ## Running tests The automatic travis tests ONLY run CPU-based tests. Although these cover most of the use cases, @@ -26,22 +29,6 @@ The GPU machine must have: 2. [NVIDIA-apex](https://github.com/NVIDIA/apex#linux) installed. -### test_models.py -This file fits a tiny model on MNIST using these different set-ups. -1. CPU only. -2. Single GPU with DP. -3. Multiple (2) GPUs using DP. -3. Multiple (2) GPUs using DDP. -3. Multiple (2) GPUs using DP + apex (for 16-bit precision). -3. Multiple (2) GPUs using DDP + apex (for 16-bit precision). - -For each set up it also tests: -1. model saving. -2. model loading. -3. predicting with a loaded model. -4. simulated save from HPC signal. -5. simulated load from HPC signal. - ## Running Coverage Make sure to run coverage on a GPU machine with at least 2 GPUs and NVIDIA apex installed. diff --git a/tests/test_a_restore_models.py b/tests/test_a_restore_models.py new file mode 100644 index 00000000..5b747aec --- /dev/null +++ b/tests/test_a_restore_models.py @@ -0,0 +1,408 @@ +import os + +import pytest +import torch + +from pytorch_lightning import Trainer +from pytorch_lightning.callbacks import ModelCheckpoint +from pytorch_lightning.testing import LightningTestModel +from . import testing_utils + + +def test_running_test_pretrained_model_ddp(): + """Verify test() on pretrained model""" + if not testing_utils.can_run_gpu_test(): + return + + testing_utils.reset_seed() + testing_utils.set_random_master_port() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # exp file to get meta + logger = testing_utils.get_test_tube_logger(False) + + # exp file to get weights + checkpoint = testing_utils.init_checkpoint_callback(logger) + + trainer_options = dict( + show_progress_bar=False, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.2, + checkpoint_callback=checkpoint, + logger=logger, + gpus=[0, 1], + distributed_backend='ddp' + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + exp = logger.experiment + print(os.listdir(exp.get_data_path(exp.name, exp.version))) + + # correct result and ok accuracy + assert result == 1, 'training failed to complete' + pretrained_model = testing_utils.load_model(logger.experiment, + trainer.checkpoint_callback.filepath, + module_class=LightningTestModel) + + # run test set + new_trainer = Trainer(**trainer_options) + new_trainer.test(pretrained_model) + + for dataloader in model.test_dataloader(): + testing_utils.run_prediction(dataloader, pretrained_model) + + testing_utils.clear_save_dir() + + +def test_running_test_pretrained_model(): + testing_utils.reset_seed() + + """Verify test() on pretrained model""" + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + logger = testing_utils.get_test_tube_logger(False) + + # logger file to get weights + checkpoint = testing_utils.init_checkpoint_callback(logger) + + trainer_options = dict( + show_progress_bar=False, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.2, + checkpoint_callback=checkpoint, + logger=logger + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # correct result and ok accuracy + assert result == 1, 'training failed to complete' + pretrained_model = testing_utils.load_model( + logger.experiment, trainer.checkpoint_callback.filepath, module_class=LightningTestModel + ) + + new_trainer = Trainer(**trainer_options) + new_trainer.test(pretrained_model) + + # test we have good test accuracy + testing_utils.assert_ok_test_acc(new_trainer) + testing_utils.clear_save_dir() + + +def test_load_model_from_checkpoint(): + testing_utils.reset_seed() + + """Verify test() on pretrained model""" + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + trainer_options = dict( + show_progress_bar=False, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.2, + checkpoint_callback=True, + logger=False, + default_save_path=save_dir + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # correct result and ok accuracy + assert result == 1, 'training failed to complete' + pretrained_model = LightningTestModel.load_from_checkpoint( + os.path.join(trainer.checkpoint_callback.filepath, "_ckpt_epoch_1.ckpt") + ) + + # test that hparams loaded correctly + for k, v in vars(hparams).items(): + assert getattr(pretrained_model.hparams, k) == v + + new_trainer = Trainer(**trainer_options) + new_trainer.test(pretrained_model) + + # test we have good test accuracy + testing_utils.assert_ok_test_acc(new_trainer) + testing_utils.clear_save_dir() + + +def test_running_test_pretrained_model_dp(): + testing_utils.reset_seed() + + """Verify test() on pretrained model""" + if not testing_utils.can_run_gpu_test(): + return + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + logger = testing_utils.get_test_tube_logger(False) + + # logger file to get weights + checkpoint = testing_utils.init_checkpoint_callback(logger) + + trainer_options = dict( + show_progress_bar=True, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.2, + checkpoint_callback=checkpoint, + logger=logger, + gpus=[0, 1], + distributed_backend='dp' + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # correct result and ok accuracy + assert result == 1, 'training failed to complete' + pretrained_model = testing_utils.load_model(logger.experiment, + trainer.checkpoint_callback.filepath, + module_class=LightningTestModel) + + new_trainer = Trainer(**trainer_options) + new_trainer.test(pretrained_model) + + # test we have good test accuracy + testing_utils.assert_ok_test_acc(new_trainer) + testing_utils.clear_save_dir() + + +def test_dp_resume(): + """ + Make sure DP continues training correctly + :return: + """ + if not testing_utils.can_run_gpu_test(): + return + + testing_utils.reset_seed() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + trainer_options = dict( + show_progress_bar=True, + max_nb_epochs=2, + gpus=2, + distributed_backend='dp', + ) + + save_dir = testing_utils.init_save_dir() + + # get logger + logger = testing_utils.get_test_tube_logger(debug=False) + + # exp file to get weights + # logger file to get weights + checkpoint = testing_utils.init_checkpoint_callback(logger) + + # add these to the trainer options + trainer_options['logger'] = logger + trainer_options['checkpoint_callback'] = checkpoint + + # fit model + trainer = Trainer(**trainer_options) + trainer.is_slurm_managing_tasks = True + result = trainer.fit(model) + + # track epoch before saving + real_global_epoch = trainer.current_epoch + + # correct result and ok accuracy + assert result == 1, 'amp + dp model failed to complete' + + # --------------------------- + # HPC LOAD/SAVE + # --------------------------- + # save + trainer.hpc_save(save_dir, logger) + + # init new trainer + new_logger = testing_utils.get_test_tube_logger(version=logger.version) + trainer_options['logger'] = new_logger + trainer_options['checkpoint_callback'] = ModelCheckpoint(save_dir) + trainer_options['train_percent_check'] = 0.2 + trainer_options['val_percent_check'] = 0.2 + trainer_options['max_nb_epochs'] = 1 + new_trainer = Trainer(**trainer_options) + + # set the epoch start hook so we can predict before the model does the full training + def assert_good_acc(): + assert new_trainer.current_epoch == real_global_epoch and new_trainer.current_epoch > 0 + + # if model and state loaded correctly, predictions will be good even though we + # haven't trained with the new loaded model + dp_model = new_trainer.model + dp_model.eval() + + dataloader = trainer.get_train_dataloader() + testing_utils.run_prediction(dataloader, dp_model, dp=True) + + # new model + model = LightningTestModel(hparams) + model.on_sanity_check_start = assert_good_acc + + # fit new model which should load hpc weights + new_trainer.fit(model) + + # test freeze on gpu + model.freeze() + model.unfreeze() + + testing_utils.clear_save_dir() + + +def test_cpu_restore_training(): + """ + Verify continue training session on CPU + :return: + """ + testing_utils.reset_seed() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + test_logger_version = 10 + logger = testing_utils.get_test_tube_logger(False, version=test_logger_version) + + trainer_options = dict( + max_nb_epochs=2, + val_check_interval=0.50, + val_percent_check=0.2, + train_percent_check=0.2, + logger=logger, + checkpoint_callback=ModelCheckpoint(save_dir) + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + real_global_epoch = trainer.current_epoch + + # traning complete + assert result == 1, 'amp + ddp model failed to complete' + + # wipe-out trainer and model + # retrain with not much data... this simulates picking training back up after slurm + # we want to see if the weights come back correctly + new_logger = testing_utils.get_test_tube_logger(False, version=test_logger_version) + trainer_options = dict( + max_nb_epochs=2, + val_check_interval=0.50, + val_percent_check=0.2, + train_percent_check=0.2, + logger=new_logger, + checkpoint_callback=ModelCheckpoint(save_dir), + ) + trainer = Trainer(**trainer_options) + model = LightningTestModel(hparams) + + # set the epoch start hook so we can predict before the model does the full training + def assert_good_acc(): + assert trainer.current_epoch == real_global_epoch + assert trainer.current_epoch >= 0 + + # if model and state loaded correctly, predictions will be good even though we + # haven't trained with the new loaded model + trainer.model.eval() + for dataloader in trainer.get_val_dataloaders(): + testing_utils.run_prediction(dataloader, trainer.model) + + model.on_sanity_check_start = assert_good_acc + + # by calling fit again, we trigger training, loading weights from the cluster + # and our hook to predict using current model before any more weight updates + trainer.fit(model) + + testing_utils.clear_save_dir() + + +def test_model_saving_loading(): + """ + Tests use case where trainer saves the model, and user loads it from tags independently + :return: + """ + testing_utils.reset_seed() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + logger = testing_utils.get_test_tube_logger(False) + + trainer_options = dict( + max_nb_epochs=1, + logger=logger, + checkpoint_callback=ModelCheckpoint(save_dir) + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # traning complete + assert result == 1, 'amp + ddp model failed to complete' + + # make a prediction + for dataloader in model.test_dataloader(): + for batch in dataloader: + break + + x, y = batch + x = x.view(x.size(0), -1) + + # generate preds before saving model + model.eval() + pred_before_saving = model(x) + + # save model + new_weights_path = os.path.join(save_dir, 'save_test.ckpt') + trainer.save_checkpoint(new_weights_path) + + # load new model + tags_path = logger.experiment.get_data_path(logger.experiment.name, logger.experiment.version) + tags_path = os.path.join(tags_path, 'meta_tags.csv') + model_2 = LightningTestModel.load_from_metrics(weights_path=new_weights_path, + tags_csv=tags_path) + model_2.eval() + + # make prediction + # assert that both predictions are the same + new_pred = model_2(x) + assert torch.all(torch.eq(pred_before_saving, new_pred)).item() == 1 + + testing_utils.clear_save_dir() + + +if __name__ == '__main__': + pytest.main([__file__]) diff --git a/tests/test_cpu_models.py b/tests/test_cpu_models.py new file mode 100644 index 00000000..2c2be069 --- /dev/null +++ b/tests/test_cpu_models.py @@ -0,0 +1,320 @@ +import warnings + +import pytest +import torch + +from pytorch_lightning import Trainer +from pytorch_lightning.callbacks import ( + EarlyStopping, +) +from pytorch_lightning.testing import ( + LightningTestModel, + LightningTestModelBase, + LightningTestMixin, +) +from . import testing_utils + + +def test_early_stopping_cpu_model(): + """ + Test each of the trainer options + :return: + """ + testing_utils.reset_seed() + + stopping = EarlyStopping(monitor='val_loss') + trainer_options = dict( + early_stop_callback=stopping, + gradient_clip_val=1.0, + overfit_pct=0.20, + track_grad_norm=2, + print_nan_grads=True, + show_progress_bar=True, + logger=testing_utils.get_test_tube_logger(), + train_percent_check=0.1, + val_percent_check=0.1 + ) + + model, hparams = testing_utils.get_model() + testing_utils.run_gpu_model_test(trainer_options, model, hparams, on_gpu=False) + + # test freeze on cpu + model.freeze() + model.unfreeze() + + +def test_lbfgs_cpu_model(): + """ + Test each of the trainer options + :return: + """ + testing_utils.reset_seed() + + trainer_options = dict( + max_nb_epochs=1, + print_nan_grads=True, + show_progress_bar=False, + weights_summary='top', + train_percent_check=1.0, + val_percent_check=0.2 + ) + + model, hparams = testing_utils.get_model(use_test_model=True, lbfgs=True) + testing_utils.run_model_test_no_loggers(trainer_options, + model, hparams, on_gpu=False, min_acc=0.30) + + testing_utils.clear_save_dir() + + +def test_default_logger_callbacks_cpu_model(): + """ + Test each of the trainer options + :return: + """ + testing_utils.reset_seed() + + trainer_options = dict( + max_nb_epochs=1, + gradient_clip_val=1.0, + overfit_pct=0.20, + print_nan_grads=True, + show_progress_bar=False, + train_percent_check=0.01, + val_percent_check=0.01 + ) + + model, hparams = testing_utils.get_model() + testing_utils.run_model_test_no_loggers(trainer_options, model, hparams, on_gpu=False) + + # test freeze on cpu + model.freeze() + model.unfreeze() + + testing_utils.clear_save_dir() + + +def test_running_test_after_fitting(): + """Verify test() on fitted model""" + testing_utils.reset_seed() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + logger = testing_utils.get_test_tube_logger(False) + + # logger file to get weights + checkpoint = testing_utils.init_checkpoint_callback(logger) + + trainer_options = dict( + show_progress_bar=False, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.2, + test_percent_check=0.2, + checkpoint_callback=checkpoint, + logger=logger + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + assert result == 1, 'training failed to complete' + + trainer.test() + + # test we have good test accuracy + testing_utils.assert_ok_test_acc(trainer) + + testing_utils.clear_save_dir() + + +def test_running_test_without_val(): + testing_utils.reset_seed() + + """Verify test() works on a model with no val_loader""" + + class CurrentTestModel(LightningTestMixin, LightningTestModelBase): + pass + + hparams = testing_utils.get_hparams() + model = CurrentTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + logger = testing_utils.get_test_tube_logger(False) + + # logger file to get weights + checkpoint = testing_utils.init_checkpoint_callback(logger) + + trainer_options = dict( + show_progress_bar=False, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.2, + test_percent_check=0.2, + checkpoint_callback=checkpoint, + logger=logger + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + assert result == 1, 'training failed to complete' + + trainer.test() + + # test we have good test accuracy + testing_utils.assert_ok_test_acc(trainer) + + testing_utils.clear_save_dir() + + +def test_single_gpu_batch_parse(): + testing_utils.reset_seed() + + if not testing_utils.can_run_gpu_test(): + return + + trainer = Trainer() + + # batch is just a tensor + batch = torch.rand(2, 3) + batch = trainer.transfer_batch_to_gpu(batch, 0) + assert batch.device.index == 0 and batch.type() == 'torch.cuda.FloatTensor' + + # tensor list + batch = [torch.rand(2, 3), torch.rand(2, 3)] + batch = trainer.transfer_batch_to_gpu(batch, 0) + assert batch[0].device.index == 0 and batch[0].type() == 'torch.cuda.FloatTensor' + assert batch[1].device.index == 0 and batch[1].type() == 'torch.cuda.FloatTensor' + + # tensor list of lists + batch = [[torch.rand(2, 3), torch.rand(2, 3)]] + batch = trainer.transfer_batch_to_gpu(batch, 0) + assert batch[0][0].device.index == 0 and batch[0][0].type() == 'torch.cuda.FloatTensor' + assert batch[0][1].device.index == 0 and batch[0][1].type() == 'torch.cuda.FloatTensor' + + # tensor dict + batch = [{'a': torch.rand(2, 3), 'b': torch.rand(2, 3)}] + batch = trainer.transfer_batch_to_gpu(batch, 0) + assert batch[0]['a'].device.index == 0 and batch[0]['a'].type() == 'torch.cuda.FloatTensor' + assert batch[0]['b'].device.index == 0 and batch[0]['b'].type() == 'torch.cuda.FloatTensor' + + # tuple of tensor list and list of tensor dict + batch = ([torch.rand(2, 3) for _ in range(2)], + [{'a': torch.rand(2, 3), 'b': torch.rand(2, 3)} for _ in range(2)]) + batch = trainer.transfer_batch_to_gpu(batch, 0) + assert batch[0][0].device.index == 0 and batch[0][0].type() == 'torch.cuda.FloatTensor' + + assert batch[1][0]['a'].device.index == 0 + assert batch[1][0]['a'].type() == 'torch.cuda.FloatTensor' + + assert batch[1][0]['b'].device.index == 0 + assert batch[1][0]['b'].type() == 'torch.cuda.FloatTensor' + + +def test_simple_cpu(): + """ + Verify continue training session on CPU + :return: + """ + testing_utils.reset_seed() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + trainer_options = dict( + max_nb_epochs=1, + val_percent_check=0.1, + train_percent_check=0.1, + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # traning complete + assert result == 1, 'amp + ddp model failed to complete' + + testing_utils.clear_save_dir() + + +def test_cpu_model(): + """ + Make sure model trains on CPU + :return: + """ + testing_utils.reset_seed() + + trainer_options = dict( + show_progress_bar=False, + logger=testing_utils.get_test_tube_logger(), + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.4 + ) + + model, hparams = testing_utils.get_model() + + testing_utils.run_gpu_model_test(trainer_options, model, hparams, on_gpu=False) + + +def test_all_features_cpu_model(): + """ + Test each of the trainer options + :return: + """ + testing_utils.reset_seed() + + trainer_options = dict( + gradient_clip_val=1.0, + overfit_pct=0.20, + track_grad_norm=2, + print_nan_grads=True, + show_progress_bar=False, + logger=testing_utils.get_test_tube_logger(), + accumulate_grad_batches=2, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.4 + ) + + model, hparams = testing_utils.get_model() + testing_utils.run_gpu_model_test(trainer_options, model, hparams, on_gpu=False) + + +def test_single_gpu_model(): + """ + Make sure single GPU works (DP mode) + :return: + """ + testing_utils.reset_seed() + + if not torch.cuda.is_available(): + warnings.warn('test_single_gpu_model cannot run.' + ' Rerun on a GPU node to run this test') + return + model, hparams = testing_utils.get_model() + + trainer_options = dict( + show_progress_bar=False, + max_nb_epochs=1, + train_percent_check=0.1, + val_percent_check=0.1, + gpus=1 + ) + + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + +if __name__ == '__main__': + pytest.main([__file__]) diff --git a/tests/test_gpu_models.py b/tests/test_gpu_models.py new file mode 100644 index 00000000..118a3f2a --- /dev/null +++ b/tests/test_gpu_models.py @@ -0,0 +1,407 @@ +import os +import pytest +import torch + +from pytorch_lightning import Trainer +from pytorch_lightning.callbacks import ( + ModelCheckpoint, +) +from pytorch_lightning.root_module import memory +from pytorch_lightning.testing import ( + LightningTestModel, +) +from pytorch_lightning.trainer.dp_mixin import ( + parse_gpu_ids, + determine_root_gpu_device, +) +from pytorch_lightning.utilities.debugging import MisconfigurationException +from . import testing_utils + +PRETEND_N_OF_GPUS = 16 + + +def test_multi_gpu_model_ddp2(): + """ + Make sure DDP2 works + :return: + """ + if not testing_utils.can_run_gpu_test(): + return + + testing_utils.reset_seed() + testing_utils.set_random_master_port() + + model, hparams = testing_utils.get_model() + trainer_options = dict( + show_progress_bar=True, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.2, + gpus=2, + weights_summary=None, + distributed_backend='ddp2' + ) + + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + +def test_multi_gpu_model_ddp(): + """ + Make sure DDP works + :return: + """ + if not testing_utils.can_run_gpu_test(): + return + + testing_utils.reset_seed() + testing_utils.set_random_master_port() + + model, hparams = testing_utils.get_model() + trainer_options = dict( + show_progress_bar=False, + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.2, + gpus=[0, 1], + distributed_backend='ddp' + ) + + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + +def test_optimizer_return_options(): + testing_utils.reset_seed() + + trainer = Trainer() + model, hparams = testing_utils.get_model() + + # single optimizer + opt_a = torch.optim.Adam(model.parameters(), lr=0.002) + opt_b = torch.optim.SGD(model.parameters(), lr=0.002) + optim, lr_sched = trainer.init_optimizers(opt_a) + assert len(optim) == 1 and len(lr_sched) == 0 + + # opt tuple + opts = (opt_a, opt_b) + optim, lr_sched = trainer.init_optimizers(opts) + assert len(optim) == 2 and optim[0] == opts[0] and optim[1] == opts[1] + assert len(lr_sched) == 0 + + # opt list + opts = [opt_a, opt_b] + optim, lr_sched = trainer.init_optimizers(opts) + assert len(optim) == 2 and optim[0] == opts[0] and optim[1] == opts[1] + assert len(lr_sched) == 0 + + # opt tuple of lists + opts = ([opt_a], ['lr_scheduler']) + optim, lr_sched = trainer.init_optimizers(opts) + assert len(optim) == 1 and len(lr_sched) == 1 + assert optim[0] == opts[0][0] and lr_sched[0] == 'lr_scheduler' + + +def test_cpu_slurm_save_load(): + """ + Verify model save/load/checkpoint on CPU + :return: + """ + testing_utils.reset_seed() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + logger = testing_utils.get_test_tube_logger(False) + + version = logger.version + + trainer_options = dict( + max_nb_epochs=1, + logger=logger, + checkpoint_callback=ModelCheckpoint(save_dir) + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + real_global_step = trainer.global_step + + # traning complete + assert result == 1, 'amp + ddp model failed to complete' + + # predict with trained model before saving + # make a prediction + for dataloader in model.test_dataloader(): + for batch in dataloader: + break + + x, y = batch + x = x.view(x.size(0), -1) + + model.eval() + pred_before_saving = model(x) + + # test HPC saving + # simulate snapshot on slurm + saved_filepath = trainer.hpc_save(save_dir, logger) + assert os.path.exists(saved_filepath) + + # new logger file to get meta + logger = testing_utils.get_test_tube_logger(False, version=version) + + trainer_options = dict( + max_nb_epochs=1, + logger=logger, + checkpoint_callback=ModelCheckpoint(save_dir), + ) + trainer = Trainer(**trainer_options) + model = LightningTestModel(hparams) + + # set the epoch start hook so we can predict before the model does the full training + def assert_pred_same(): + assert trainer.global_step == real_global_step and trainer.global_step > 0 + + # predict with loaded model to make sure answers are the same + trainer.model.eval() + new_pred = trainer.model(x) + assert torch.all(torch.eq(pred_before_saving, new_pred)).item() == 1 + + model.on_epoch_start = assert_pred_same + + # by calling fit again, we trigger training, loading weights from the cluster + # and our hook to predict using current model before any more weight updates + trainer.fit(model) + + testing_utils.clear_save_dir() + + +def test_multi_gpu_none_backend(): + """ + Make sure when using multiple GPUs the user can't use + distributed_backend = None + :return: + """ + testing_utils.reset_seed() + + if not testing_utils.can_run_gpu_test(): + return + + model, hparams = testing_utils.get_model() + trainer_options = dict( + show_progress_bar=False, + max_nb_epochs=1, + train_percent_check=0.1, + val_percent_check=0.1, + gpus='-1' + ) + + with pytest.raises(MisconfigurationException): + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + +def test_multi_gpu_model_dp(): + """ + Make sure DP works + :return: + """ + testing_utils.reset_seed() + + if not testing_utils.can_run_gpu_test(): + return + + model, hparams = testing_utils.get_model() + trainer_options = dict( + show_progress_bar=False, + distributed_backend='dp', + max_nb_epochs=1, + train_percent_check=0.1, + val_percent_check=0.1, + gpus='-1' + ) + + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + # test memory helper functions + memory.get_gpu_memory_map() + + +def test_ddp_sampler_error(): + """ + Make sure DDP + AMP work + :return: + """ + if not testing_utils.can_run_gpu_test(): + return + + testing_utils.reset_seed() + testing_utils.set_random_master_port() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams, force_remove_distributed_sampler=True) + + logger = testing_utils.get_test_tube_logger(True) + + trainer = Trainer( + logger=logger, + show_progress_bar=False, + max_nb_epochs=1, + gpus=[0, 1], + distributed_backend='ddp', + use_amp=True + ) + + with pytest.warns(UserWarning): + trainer.get_dataloaders(model) + + testing_utils.clear_save_dir() + + +@pytest.fixture +def mocked_device_count(monkeypatch): + def device_count(): + return PRETEND_N_OF_GPUS + + monkeypatch.setattr(torch.cuda, 'device_count', device_count) + + +@pytest.fixture +def mocked_device_count_0(monkeypatch): + def device_count(): + return 0 + + monkeypatch.setattr(torch.cuda, 'device_count', device_count) + + +test_num_gpus_data = [ + pytest.param(None, 0, None, id="None - expect 0 gpu to use."), + pytest.param(0, 0, None, id="Oth gpu, expect 1 gpu to use."), + pytest.param(1, 1, None, id="1st gpu, expect 1 gpu to use."), + pytest.param(-1, PRETEND_N_OF_GPUS, "ddp", id="-1 - use all gpus"), + pytest.param('-1', PRETEND_N_OF_GPUS, "ddp", id="'-1' - use all gpus"), + pytest.param(3, 3, "ddp", id="3rd gpu - 1 gpu to use (backend:ddp)") +] + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize(["gpus", "expected_num_gpus", "distributed_backend"], test_num_gpus_data) +def test_trainer_gpu_parse(mocked_device_count, gpus, expected_num_gpus, distributed_backend): + assert Trainer(gpus=gpus, distributed_backend=distributed_backend).num_gpus == expected_num_gpus + + +test_num_gpus_data_0 = [ + pytest.param(None, 0, None, id="None - expect 0 gpu to use."), + pytest.param(None, 0, "ddp", id="None - expect 0 gpu to use."), +] + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize(["gpus", "expected_num_gpus", "distributed_backend"], test_num_gpus_data_0) +def test_trainer_num_gpu_0(mocked_device_count_0, gpus, expected_num_gpus, distributed_backend): + assert Trainer(gpus=gpus, distributed_backend=distributed_backend).num_gpus == expected_num_gpus + + +test_root_gpu_data = [ + pytest.param(None, None, "ddp", id="None is None"), + pytest.param(0, None, "ddp", id="O gpus, expect gpu root device to be None."), + pytest.param(1, 0, "ddp", id="1 gpu, expect gpu root device to be 0."), + pytest.param(-1, 0, "ddp", id="-1 - use all gpus, expect gpu root device to be 0."), + pytest.param('-1', 0, "ddp", id="'-1' - use all gpus, expect gpu root device to be 0."), + pytest.param(3, 0, "ddp", id="3 gpus, expect gpu root device to be 0.(backend:ddp)")] + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize(['gpus', 'expected_root_gpu', "distributed_backend"], test_root_gpu_data) +def test_root_gpu_property(mocked_device_count, gpus, expected_root_gpu, distributed_backend): + assert Trainer(gpus=gpus, distributed_backend=distributed_backend).root_gpu == expected_root_gpu + + +test_root_gpu_data_for_0_devices_passing = [ + pytest.param(None, None, None, id="None is None"), + pytest.param(None, None, "ddp", id="None is None"), + pytest.param(0, None, "ddp", id="None is None"), +] + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize([ + 'gpus', 'expected_root_gpu', "distributed_backend"], test_root_gpu_data_for_0_devices_passing) +def test_root_gpu_property_0_passing( + mocked_device_count_0, gpus, expected_root_gpu, distributed_backend): + assert Trainer(gpus=gpus, distributed_backend=distributed_backend).root_gpu == expected_root_gpu + + +# Asking for a gpu when non are available will result in a MisconfigurationException +test_root_gpu_data_for_0_devices_raising = [ + pytest.param(1, None, "ddp"), + pytest.param(3, None, "ddp"), + pytest.param(3, None, "ddp"), + pytest.param([1, 2], None, "ddp"), + pytest.param([0, 1], None, "ddp"), + pytest.param(-1, None, "ddp"), + pytest.param('-1', None, "ddp") +] + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize([ + 'gpus', 'expected_root_gpu', "distributed_backend"], test_root_gpu_data_for_0_devices_raising) +def test_root_gpu_property_0_raising( + mocked_device_count_0, gpus, expected_root_gpu, distributed_backend): + with pytest.raises(MisconfigurationException): + Trainer(gpus=gpus, distributed_backend=distributed_backend).root_gpu + + +test_determine_root_gpu_device_data = [ + pytest.param(None, None, id="No gpus, expect gpu root device to be None"), + pytest.param([0], 0, id="Oth gpu, expect gpu root device to be 0."), + pytest.param([1], 1, id="1st gpu, expect gpu root device to be 1."), + pytest.param([3], 3, id="3rd gpu, expect gpu root device to be 3."), + pytest.param([1, 2], 1, id="[1, 2] gpus, expect gpu root device to be 1."), +] + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize(['gpus', 'expected_root_gpu'], test_determine_root_gpu_device_data) +def test_determine_root_gpu_device(gpus, expected_root_gpu): + assert determine_root_gpu_device(gpus) == expected_root_gpu + + +test_parse_gpu_ids_data = [ + pytest.param(None, None), + pytest.param(0, None), + pytest.param(1, [0]), + pytest.param(-1, list(range(PRETEND_N_OF_GPUS)), id="-1 - use all gpus"), + pytest.param('-1', list(range(PRETEND_N_OF_GPUS)), id="'-1' - use all gpus"), + pytest.param(3, [0, 1, 2])] + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize(['gpus', 'expected_gpu_ids'], test_parse_gpu_ids_data) +def test_parse_gpu_ids(mocked_device_count, gpus, expected_gpu_ids): + assert parse_gpu_ids(gpus) == expected_gpu_ids + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize("gpus", [[1, 2, 19], -1, '-1']) +def test_parse_gpu_fail_on_non_existant_id(mocked_device_count_0, gpus): + with pytest.raises(MisconfigurationException): + parse_gpu_ids(gpus) + + +@pytest.mark.gpus_param_tests +def test_parse_gpu_fail_on_non_existant_id_2(mocked_device_count): + with pytest.raises(MisconfigurationException): + parse_gpu_ids([1, 2, 19]) + + +@pytest.mark.gpus_param_tests +@pytest.mark.parametrize("gpus", [-1, '-1']) +def test_parse_gpu_returns_None_when_no_devices_are_available(mocked_device_count_0, gpus): + with pytest.raises(MisconfigurationException): + parse_gpu_ids(gpus) + + +if __name__ == '__main__': + pytest.main([__file__]) diff --git a/tests/test_models.py b/tests/test_models.py deleted file mode 100644 index 577683d9..00000000 --- a/tests/test_models.py +++ /dev/null @@ -1,1843 +0,0 @@ -import os -import shutil -import warnings -from argparse import Namespace - -import numpy as np -import pytest -import torch - -from pl_examples import LightningTemplateModel -# sys.path += [os.path.abspath('..'), os.path.abspath('../..')] -from pytorch_lightning import Trainer -from pytorch_lightning.callbacks import ( - ModelCheckpoint, - EarlyStopping, -) -from pytorch_lightning.logging import TestTubeLogger -from pytorch_lightning.root_module import memory -from pytorch_lightning.testing import ( - LightningTestModel, - LightningTestModelBase, - LightningValidationStepMixin, - LightningValidationMultipleDataloadersMixin, - LightningTestMixin, - LightningTestMultipleDataloadersMixin, -) -from pytorch_lightning.trainer import trainer_io -from pytorch_lightning.trainer.dp_mixin import ( - parse_gpu_ids, - determine_root_gpu_device, -) -from pytorch_lightning.trainer.logging_mixin import TrainerLoggingMixin -from pytorch_lightning.utilities.debugging import MisconfigurationException - -# generate a list of random seeds for each test -RANDOM_FILE_PATHS = list(np.random.randint(12000, 19000, 1000)) -RANDOM_PORTS = list(np.random.randint(12000, 19000, 1000)) -ROOT_SEED = 1234 -torch.manual_seed(ROOT_SEED) -np.random.seed(ROOT_SEED) -RANDOM_SEEDS = list(np.random.randint(0, 10000, 1000)) -PRETEND_N_OF_GPUS = 16 - - -# ------------------------------------------------------------------------ -# TESTS -# ------------------------------------------------------------------------ - - -def test_multi_gpu_model_ddp2(): - """ - Make sure DDP2 works - :return: - """ - if not can_run_gpu_test(): - return - - reset_seed() - set_random_master_port() - - model, hparams = get_model() - trainer_options = dict( - show_progress_bar=True, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.2, - gpus=2, - weights_summary=None, - distributed_backend='ddp2' - ) - - run_gpu_model_test(trainer_options, model, hparams) - - -def test_early_stopping_cpu_model(): - """ - Test each of the trainer options - :return: - """ - reset_seed() - - stopping = EarlyStopping(monitor='val_loss') - trainer_options = dict( - early_stop_callback=stopping, - gradient_clip_val=1.0, - overfit_pct=0.20, - track_grad_norm=2, - print_nan_grads=True, - show_progress_bar=True, - logger=get_test_tube_logger(), - train_percent_check=0.1, - val_percent_check=0.1 - ) - - model, hparams = get_model() - run_gpu_model_test(trainer_options, model, hparams, on_gpu=False) - - # test freeze on cpu - model.freeze() - model.unfreeze() - - -def test_running_test_pretrained_model_ddp(): - """Verify test() on pretrained model""" - if not can_run_gpu_test(): - return - - reset_seed() - set_random_master_port() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - # exp file to get meta - logger = get_test_tube_logger(False) - - # exp file to get weights - checkpoint = init_checkpoint_callback(logger) - - trainer_options = dict( - show_progress_bar=False, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.2, - checkpoint_callback=checkpoint, - logger=logger, - gpus=[0, 1], - distributed_backend='ddp' - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - exp = logger.experiment - print(os.listdir(exp.get_data_path(exp.name, exp.version))) - - # correct result and ok accuracy - assert result == 1, 'training failed to complete' - pretrained_model = load_model(logger.experiment, trainer.checkpoint_callback.filepath, - module_class=LightningTestModel) - - # run test set - new_trainer = Trainer(**trainer_options) - new_trainer.test(pretrained_model) - - [run_prediction(dataloader, pretrained_model) for dataloader in model.test_dataloader()] - - clear_save_dir() - - -def test_lbfgs_cpu_model(): - """ - Test each of the trainer options - :return: - """ - reset_seed() - - trainer_options = dict( - max_nb_epochs=1, - print_nan_grads=True, - show_progress_bar=False, - weights_summary='top', - train_percent_check=1.0, - val_percent_check=0.2 - ) - - model, hparams = get_model(use_test_model=True, lbfgs=True) - run_model_test_no_loggers(trainer_options, model, hparams, on_gpu=False, min_acc=0.30) - - clear_save_dir() - - -def test_default_logger_callbacks_cpu_model(): - """ - Test each of the trainer options - :return: - """ - reset_seed() - - trainer_options = dict( - max_nb_epochs=1, - gradient_clip_val=1.0, - overfit_pct=0.20, - print_nan_grads=True, - show_progress_bar=False, - train_percent_check=0.01, - val_percent_check=0.01 - ) - - model, hparams = get_model() - run_model_test_no_loggers(trainer_options, model, hparams, on_gpu=False) - - # test freeze on cpu - model.freeze() - model.unfreeze() - - clear_save_dir() - - -def test_dp_resume(): - """ - Make sure DP continues training correctly - :return: - """ - if not can_run_gpu_test(): - return - - reset_seed() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - trainer_options = dict( - show_progress_bar=True, - max_nb_epochs=2, - gpus=2, - distributed_backend='dp', - ) - - save_dir = init_save_dir() - - # get logger - logger = get_test_tube_logger(debug=False) - - # exp file to get weights - # logger file to get weights - checkpoint = init_checkpoint_callback(logger) - - # add these to the trainer options - trainer_options['logger'] = logger - trainer_options['checkpoint_callback'] = checkpoint - - # fit model - trainer = Trainer(**trainer_options) - trainer.is_slurm_managing_tasks = True - result = trainer.fit(model) - - # track epoch before saving - real_global_epoch = trainer.current_epoch - - # correct result and ok accuracy - assert result == 1, 'amp + dp model failed to complete' - - # --------------------------- - # HPC LOAD/SAVE - # --------------------------- - # save - trainer.hpc_save(save_dir, logger) - - # init new trainer - new_logger = get_test_tube_logger(version=logger.version) - trainer_options['logger'] = new_logger - trainer_options['checkpoint_callback'] = ModelCheckpoint(save_dir) - trainer_options['train_percent_check'] = 0.2 - trainer_options['val_percent_check'] = 0.2 - trainer_options['max_nb_epochs'] = 1 - new_trainer = Trainer(**trainer_options) - - # set the epoch start hook so we can predict before the model does the full training - def assert_good_acc(): - assert new_trainer.current_epoch == real_global_epoch and new_trainer.current_epoch > 0 - - # if model and state loaded correctly, predictions will be good even though we - # haven't trained with the new loaded model - dp_model = new_trainer.model - dp_model.eval() - - dataloader = trainer.get_train_dataloader() - run_prediction(dataloader, dp_model, dp=True) - - # new model - model = LightningTestModel(hparams) - model.on_sanity_check_start = assert_good_acc - - # fit new model which should load hpc weights - new_trainer.fit(model) - - # test freeze on gpu - model.freeze() - model.unfreeze() - - clear_save_dir() - - -def test_running_test_after_fitting(): - """Verify test() on fitted model""" - reset_seed() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - # logger file to get weights - checkpoint = init_checkpoint_callback(logger) - - trainer_options = dict( - show_progress_bar=False, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.2, - test_percent_check=0.2, - checkpoint_callback=checkpoint, - logger=logger - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - assert result == 1, 'training failed to complete' - - trainer.test() - - # test we have good test accuracy - assert_ok_test_acc(trainer) - - clear_save_dir() - - -def test_running_test_without_val(): - reset_seed() - - """Verify test() works on a model with no val_loader""" - - class CurrentTestModel(LightningTestMixin, LightningTestModelBase): - pass - - hparams = get_hparams() - model = CurrentTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - # logger file to get weights - checkpoint = init_checkpoint_callback(logger) - - trainer_options = dict( - show_progress_bar=False, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.2, - test_percent_check=0.2, - checkpoint_callback=checkpoint, - logger=logger - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - assert result == 1, 'training failed to complete' - - trainer.test() - - # test we have good test accuracy - assert_ok_test_acc(trainer) - - clear_save_dir() - - -def test_running_test_pretrained_model(): - reset_seed() - - """Verify test() on pretrained model""" - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - # logger file to get weights - checkpoint = init_checkpoint_callback(logger) - - trainer_options = dict( - show_progress_bar=False, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.2, - checkpoint_callback=checkpoint, - logger=logger - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # correct result and ok accuracy - assert result == 1, 'training failed to complete' - pretrained_model = load_model( - logger.experiment, trainer.checkpoint_callback.filepath, module_class=LightningTestModel - ) - - new_trainer = Trainer(**trainer_options) - new_trainer.test(pretrained_model) - - # test we have good test accuracy - assert_ok_test_acc(new_trainer) - clear_save_dir() - - -def test_load_model_from_checkpoint(): - reset_seed() - - """Verify test() on pretrained model""" - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - trainer_options = dict( - show_progress_bar=False, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.2, - checkpoint_callback=True, - logger=False, - default_save_path=save_dir - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # correct result and ok accuracy - assert result == 1, 'training failed to complete' - pretrained_model = LightningTestModel.load_from_checkpoint( - os.path.join(trainer.checkpoint_callback.filepath, "_ckpt_epoch_1.ckpt") - ) - - # test that hparams loaded correctly - for k, v in vars(hparams).items(): - assert getattr(pretrained_model.hparams, k) == v - - new_trainer = Trainer(**trainer_options) - new_trainer.test(pretrained_model) - - # test we have good test accuracy - assert_ok_test_acc(new_trainer) - clear_save_dir() - - -def test_running_test_pretrained_model_dp(): - reset_seed() - - """Verify test() on pretrained model""" - if not can_run_gpu_test(): - return - - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - # logger file to get weights - checkpoint = init_checkpoint_callback(logger) - - trainer_options = dict( - show_progress_bar=True, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.2, - checkpoint_callback=checkpoint, - logger=logger, - gpus=[0, 1], - distributed_backend='dp' - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # correct result and ok accuracy - assert result == 1, 'training failed to complete' - pretrained_model = load_model(logger.experiment, trainer.checkpoint_callback.filepath, - module_class=LightningTestModel) - - new_trainer = Trainer(**trainer_options) - new_trainer.test(pretrained_model) - - # test we have good test accuracy - assert_ok_test_acc(new_trainer) - clear_save_dir() - - -def test_gradient_accumulation_scheduling(): - reset_seed() - - """ - Test grad accumulation by the freq of optimizer updates - """ - # test incorrect configs - with pytest.raises(IndexError): - assert Trainer(accumulate_grad_batches={0: 3, 1: 4, 4: 6}) - assert Trainer(accumulate_grad_batches={-2: 3}) - - with pytest.raises(TypeError): - assert Trainer(accumulate_grad_batches={}) - assert Trainer(accumulate_grad_batches=[[2, 3], [4, 6]]) - assert Trainer(accumulate_grad_batches={1: 2, 3.: 4}) - assert Trainer(accumulate_grad_batches={1: 2.5, 3: 5}) - - # test optimizer call freq matches scheduler - def optimizer_step(self, epoch_nb, batch_nb, optimizer, optimizer_i, second_order_closure=None): - # only test the first 12 batches in epoch - if batch_nb < 12: - if epoch_nb == 0: - # reset counter when starting epoch - if batch_nb == 0: - self.prev_called_batch_nb = 0 - - # use this opportunity to test once - assert self.trainer.accumulate_grad_batches == 1 - - assert batch_nb == self.prev_called_batch_nb - self.prev_called_batch_nb += 1 - - elif 1 <= epoch_nb <= 2: - # reset counter when starting epoch - if batch_nb == 1: - self.prev_called_batch_nb = 1 - - # use this opportunity to test once - assert self.trainer.accumulate_grad_batches == 2 - - assert batch_nb == self.prev_called_batch_nb - self.prev_called_batch_nb += 2 - - else: - if batch_nb == 3: - self.prev_called_batch_nb = 3 - - # use this opportunity to test once - assert self.trainer.accumulate_grad_batches == 4 - - assert batch_nb == self.prev_called_batch_nb - self.prev_called_batch_nb += 3 - - optimizer.step() - - # clear gradients - optimizer.zero_grad() - - hparams = get_hparams() - model = LightningTestModel(hparams) - schedule = {1: 2, 3: 4} - - trainer = Trainer(accumulate_grad_batches=schedule, - train_percent_check=0.1, - val_percent_check=0.1, - max_nb_epochs=4) - - # for the test - trainer.optimizer_step = optimizer_step - model.prev_called_batch_nb = 0 - - trainer.fit(model) - - -def test_multi_gpu_model_ddp(): - """ - Make sure DDP works - :return: - """ - if not can_run_gpu_test(): - return - - reset_seed() - set_random_master_port() - - model, hparams = get_model() - trainer_options = dict( - show_progress_bar=False, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.2, - gpus=[0, 1], - distributed_backend='ddp' - ) - - run_gpu_model_test(trainer_options, model, hparams) - - -def test_optimizer_return_options(): - reset_seed() - - trainer = Trainer() - model, hparams = get_model() - - # single optimizer - opt_a = torch.optim.Adam(model.parameters(), lr=0.002) - opt_b = torch.optim.SGD(model.parameters(), lr=0.002) - optim, lr_sched = trainer.init_optimizers(opt_a) - assert len(optim) == 1 and len(lr_sched) == 0 - - # opt tuple - opts = (opt_a, opt_b) - optim, lr_sched = trainer.init_optimizers(opts) - assert len(optim) == 2 and optim[0] == opts[0] and optim[1] == opts[1] - assert len(lr_sched) == 0 - - # opt list - opts = [opt_a, opt_b] - optim, lr_sched = trainer.init_optimizers(opts) - assert len(optim) == 2 and optim[0] == opts[0] and optim[1] == opts[1] - assert len(lr_sched) == 0 - - # opt tuple of lists - opts = ([opt_a], ['lr_scheduler']) - optim, lr_sched = trainer.init_optimizers(opts) - assert len(optim) == 1 and len(lr_sched) == 1 - assert optim[0] == opts[0][0] and lr_sched[0] == 'lr_scheduler' - - -def test_single_gpu_batch_parse(): - reset_seed() - - if not can_run_gpu_test(): - return - - trainer = Trainer() - - # batch is just a tensor - batch = torch.rand(2, 3) - batch = trainer.transfer_batch_to_gpu(batch, 0) - assert batch.device.index == 0 and batch.type() == 'torch.cuda.FloatTensor' - - # tensor list - batch = [torch.rand(2, 3), torch.rand(2, 3)] - batch = trainer.transfer_batch_to_gpu(batch, 0) - assert batch[0].device.index == 0 and batch[0].type() == 'torch.cuda.FloatTensor' - assert batch[1].device.index == 0 and batch[1].type() == 'torch.cuda.FloatTensor' - - # tensor list of lists - batch = [[torch.rand(2, 3), torch.rand(2, 3)]] - batch = trainer.transfer_batch_to_gpu(batch, 0) - assert batch[0][0].device.index == 0 and batch[0][0].type() == 'torch.cuda.FloatTensor' - assert batch[0][1].device.index == 0 and batch[0][1].type() == 'torch.cuda.FloatTensor' - - # tensor dict - batch = [{'a': torch.rand(2, 3), 'b': torch.rand(2, 3)}] - batch = trainer.transfer_batch_to_gpu(batch, 0) - assert batch[0]['a'].device.index == 0 and batch[0]['a'].type() == 'torch.cuda.FloatTensor' - assert batch[0]['b'].device.index == 0 and batch[0]['b'].type() == 'torch.cuda.FloatTensor' - - # tuple of tensor list and list of tensor dict - batch = ([torch.rand(2, 3) for _ in range(2)], - [{'a': torch.rand(2, 3), 'b': torch.rand(2, 3)} for _ in range(2)]) - batch = trainer.transfer_batch_to_gpu(batch, 0) - assert batch[0][0].device.index == 0 and batch[0][0].type() == 'torch.cuda.FloatTensor' - - assert batch[1][0]['a'].device.index == 0 - assert batch[1][0]['a'].type() == 'torch.cuda.FloatTensor' - - assert batch[1][0]['b'].device.index == 0 - assert batch[1][0]['b'].type() == 'torch.cuda.FloatTensor' - - -def test_no_val_module(): - """ - Tests use case where trainer saves the model, and user loads it from tags independently - :return: - """ - reset_seed() - - hparams = get_hparams() - - class CurrentTestModel(LightningTestModelBase): - pass - - model = CurrentTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - trainer_options = dict( - max_nb_epochs=1, - logger=logger, - checkpoint_callback=ModelCheckpoint(save_dir) - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # training complete - assert result == 1, 'amp + ddp model failed to complete' - - # save model - new_weights_path = os.path.join(save_dir, 'save_test.ckpt') - trainer.save_checkpoint(new_weights_path) - - # load new model - tags_path = logger.experiment.get_data_path(logger.experiment.name, logger.experiment.version) - tags_path = os.path.join(tags_path, 'meta_tags.csv') - model_2 = LightningTestModel.load_from_metrics(weights_path=new_weights_path, - tags_csv=tags_path) - model_2.eval() - - # make prediction - clear_save_dir() - - -def test_no_val_end_module(): - """ - Tests use case where trainer saves the model, and user loads it from tags independently - :return: - """ - reset_seed() - - class CurrentTestModel(LightningValidationStepMixin, LightningTestModelBase): - pass - - hparams = get_hparams() - model = CurrentTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - trainer_options = dict( - max_nb_epochs=1, - logger=logger, - checkpoint_callback=ModelCheckpoint(save_dir) - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # traning complete - assert result == 1, 'amp + ddp model failed to complete' - - # save model - new_weights_path = os.path.join(save_dir, 'save_test.ckpt') - trainer.save_checkpoint(new_weights_path) - - # load new model - tags_path = logger.experiment.get_data_path(logger.experiment.name, logger.experiment.version) - tags_path = os.path.join(tags_path, 'meta_tags.csv') - model_2 = LightningTestModel.load_from_metrics(weights_path=new_weights_path, - tags_csv=tags_path) - model_2.eval() - - # make prediction - clear_save_dir() - - -def test_simple_cpu(): - """ - Verify continue training session on CPU - :return: - """ - reset_seed() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - trainer_options = dict( - max_nb_epochs=1, - val_percent_check=0.1, - train_percent_check=0.1, - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # traning complete - assert result == 1, 'amp + ddp model failed to complete' - - clear_save_dir() - - -def test_amp_single_gpu(): - """ - Make sure DDP + AMP work - :return: - """ - reset_seed() - - if not torch.cuda.is_available(): - warnings.warn('test_amp_gpu_ddp cannot run.' - 'Rerun on a GPU node to run this test') - return - if not torch.cuda.device_count() > 1: - warnings.warn('test_amp_gpu_ddp cannot run.' - 'Rerun on a node with 2+ GPUs to run this test') - return - - hparams = get_hparams() - model = LightningTestModel(hparams) - - trainer_options = dict( - show_progress_bar=True, - max_nb_epochs=1, - gpus=1, - distributed_backend='ddp', - use_amp=True - ) - - run_gpu_model_test(trainer_options, model, hparams) - - -def test_no_amp_single_gpu(): - """ - Make sure DDP + AMP work - :return: - """ - reset_seed() - - if not torch.cuda.is_available(): - warnings.warn('test_amp_gpu_ddp cannot run.' - 'Rerun on a GPU node to run this test') - return - if not torch.cuda.device_count() > 1: - warnings.warn('test_amp_gpu_ddp cannot run.' - 'Rerun on a node with 2+ GPUs to run this test') - return - - hparams = get_hparams() - model = LightningTestModel(hparams) - - trainer_options = dict( - show_progress_bar=True, - max_nb_epochs=1, - gpus=1, - distributed_backend='dp', - use_amp=True - ) - - with pytest.raises((MisconfigurationException, ModuleNotFoundError)): - run_gpu_model_test(trainer_options, model, hparams) - - -def test_cpu_restore_training(): - """ - Verify continue training session on CPU - :return: - """ - reset_seed() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - test_logger_version = 10 - logger = get_test_tube_logger(False, version=test_logger_version) - - trainer_options = dict( - max_nb_epochs=2, - val_check_interval=0.50, - val_percent_check=0.2, - train_percent_check=0.2, - logger=logger, - checkpoint_callback=ModelCheckpoint(save_dir) - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - real_global_epoch = trainer.current_epoch - - # traning complete - assert result == 1, 'amp + ddp model failed to complete' - - # wipe-out trainer and model - # retrain with not much data... this simulates picking training back up after slurm - # we want to see if the weights come back correctly - new_logger = get_test_tube_logger(False, version=test_logger_version) - trainer_options = dict( - max_nb_epochs=2, - val_check_interval=0.50, - val_percent_check=0.2, - train_percent_check=0.2, - logger=new_logger, - checkpoint_callback=ModelCheckpoint(save_dir), - ) - trainer = Trainer(**trainer_options) - model = LightningTestModel(hparams) - - # set the epoch start hook so we can predict before the model does the full training - def assert_good_acc(): - assert trainer.current_epoch == real_global_epoch and trainer.current_epoch > 0 - - # if model and state loaded correctly, predictions will be good even though we - # haven't trained with the new loaded model - trainer.model.eval() - for dataloader in trainer.get_val_dataloaders(): - run_prediction(dataloader, trainer.model) - - model.on_sanity_check_start = assert_good_acc - - # by calling fit again, we trigger training, loading weights from the cluster - # and our hook to predict using current model before any more weight updates - trainer.fit(model) - - clear_save_dir() - - -def test_amp_gpu_ddp(): - """ - Make sure DDP + AMP work - :return: - """ - if not can_run_gpu_test(): - return - - reset_seed() - set_random_master_port() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - trainer_options = dict( - show_progress_bar=True, - max_nb_epochs=1, - gpus=2, - distributed_backend='ddp', - use_amp=True - ) - - run_gpu_model_test(trainer_options, model, hparams) - - -def test_cpu_slurm_save_load(): - """ - Verify model save/load/checkpoint on CPU - :return: - """ - reset_seed() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - version = logger.version - - trainer_options = dict( - max_nb_epochs=1, - logger=logger, - checkpoint_callback=ModelCheckpoint(save_dir) - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - real_global_step = trainer.global_step - - # traning complete - assert result == 1, 'amp + ddp model failed to complete' - - # predict with trained model before saving - # make a prediction - for dataloader in model.test_dataloader(): - for batch in dataloader: - break - - x, y = batch - x = x.view(x.size(0), -1) - - model.eval() - pred_before_saving = model(x) - - # test HPC saving - # simulate snapshot on slurm - saved_filepath = trainer.hpc_save(save_dir, logger) - assert os.path.exists(saved_filepath) - - # new logger file to get meta - logger = get_test_tube_logger(False, version=version) - - trainer_options = dict( - max_nb_epochs=1, - logger=logger, - checkpoint_callback=ModelCheckpoint(save_dir), - ) - trainer = Trainer(**trainer_options) - model = LightningTestModel(hparams) - - # set the epoch start hook so we can predict before the model does the full training - def assert_pred_same(): - assert trainer.global_step == real_global_step and trainer.global_step > 0 - - # predict with loaded model to make sure answers are the same - trainer.model.eval() - new_pred = trainer.model(x) - assert torch.all(torch.eq(pred_before_saving, new_pred)).item() == 1 - - model.on_epoch_start = assert_pred_same - - # by calling fit again, we trigger training, loading weights from the cluster - # and our hook to predict using current model before any more weight updates - trainer.fit(model) - - clear_save_dir() - - -def test_loading_meta_tags(): - reset_seed() - - from argparse import Namespace - hparams = get_hparams() - - # save tags - logger = get_test_tube_logger(False) - logger.log_hyperparams(Namespace(some_str='a_str', an_int=1, a_float=2.0)) - logger.log_hyperparams(hparams) - logger.save() - - # load tags - tags_path = logger.experiment.get_data_path( - logger.experiment.name, logger.experiment.version - ) + '/meta_tags.csv' - tags = trainer_io.load_hparams_from_tags_csv(tags_path) - - assert tags.batch_size == 32 and tags.hidden_dim == 1000 - - clear_save_dir() - - -def test_dp_output_reduce(): - mixin = TrainerLoggingMixin() - reset_seed() - - # test identity when we have a single gpu - out = torch.rand(3, 1) - assert mixin.reduce_distributed_output(out, nb_gpus=1) is out - - # average when we have multiples - assert mixin.reduce_distributed_output(out, nb_gpus=2) == out.mean() - - # when we have a dict of vals - out = { - 'a': out, - 'b': { - 'c': out - } - } - reduced = mixin.reduce_distributed_output(out, nb_gpus=3) - assert reduced['a'] == out['a'] - assert reduced['b']['c'] == out['b']['c'] - - -def test_model_saving_loading(): - """ - Tests use case where trainer saves the model, and user loads it from tags independently - :return: - """ - reset_seed() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - trainer_options = dict( - max_nb_epochs=1, - logger=logger, - checkpoint_callback=ModelCheckpoint(save_dir) - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # traning complete - assert result == 1, 'amp + ddp model failed to complete' - - # make a prediction - for dataloader in model.test_dataloader(): - for batch in dataloader: - break - - x, y = batch - x = x.view(x.size(0), -1) - - # generate preds before saving model - model.eval() - pred_before_saving = model(x) - - # save model - new_weights_path = os.path.join(save_dir, 'save_test.ckpt') - trainer.save_checkpoint(new_weights_path) - - # load new model - tags_path = logger.experiment.get_data_path(logger.experiment.name, logger.experiment.version) - tags_path = os.path.join(tags_path, 'meta_tags.csv') - model_2 = LightningTestModel.load_from_metrics(weights_path=new_weights_path, - tags_csv=tags_path) - model_2.eval() - - # make prediction - # assert that both predictions are the same - new_pred = model_2(x) - assert torch.all(torch.eq(pred_before_saving, new_pred)).item() == 1 - - clear_save_dir() - - -def test_model_freeze_unfreeze(): - reset_seed() - - hparams = get_hparams() - model = LightningTestModel(hparams) - - model.freeze() - model.unfreeze() - - -def test_amp_gpu_ddp_slurm_managed(): - """ - Make sure DDP + AMP work - :return: - """ - if not can_run_gpu_test(): - return - - reset_seed() - - # simulate setting slurm flags - set_random_master_port() - os.environ['SLURM_LOCALID'] = str(0) - - hparams = get_hparams() - model = LightningTestModel(hparams) - - trainer_options = dict( - show_progress_bar=True, - max_nb_epochs=1, - gpus=[0], - distributed_backend='ddp', - use_amp=True - ) - - save_dir = init_save_dir() - - # exp file to get meta - logger = get_test_tube_logger(False) - - # exp file to get weights - checkpoint = init_checkpoint_callback(logger) - - # add these to the trainer options - trainer_options['checkpoint_callback'] = checkpoint - trainer_options['logger'] = logger - - # fit model - trainer = Trainer(**trainer_options) - trainer.is_slurm_managing_tasks = True - result = trainer.fit(model) - - # correct result and ok accuracy - assert result == 1, 'amp + ddp model failed to complete' - - # test root model address - assert trainer.resolve_root_node_address('abc') == 'abc' - assert trainer.resolve_root_node_address('abc[23]') == 'abc23' - assert trainer.resolve_root_node_address('abc[23-24]') == 'abc23' - assert trainer.resolve_root_node_address('abc[23-24, 45-40, 40]') == 'abc23' - - # test model loading with a map_location - pretrained_model = load_model(logger.experiment, trainer.checkpoint_callback.filepath) - - # test model preds - [run_prediction(dataloader, pretrained_model) for dataloader in trainer.get_test_dataloaders()] - - if trainer.use_ddp: - # on hpc this would work fine... but need to hack it for the purpose of the test - trainer.model = pretrained_model - trainer.optimizers, trainer.lr_schedulers = pretrained_model.configure_optimizers() - - # test HPC loading / saving - trainer.hpc_save(save_dir, logger) - trainer.hpc_load(save_dir, on_gpu=True) - - # test freeze on gpu - model.freeze() - model.unfreeze() - - clear_save_dir() - - -def test_cpu_model_with_amp(): - """ - Make sure model trains on CPU - :return: - """ - reset_seed() - - trainer_options = dict( - show_progress_bar=False, - logger=get_test_tube_logger(), - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.4, - use_amp=True - ) - - model, hparams = get_model() - - with pytest.raises((MisconfigurationException, ModuleNotFoundError)): - run_gpu_model_test(trainer_options, model, hparams, on_gpu=False) - - -def test_cpu_model(): - """ - Make sure model trains on CPU - :return: - """ - reset_seed() - - trainer_options = dict( - show_progress_bar=False, - logger=get_test_tube_logger(), - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.4 - ) - - model, hparams = get_model() - - run_gpu_model_test(trainer_options, model, hparams, on_gpu=False) - - -def test_all_features_cpu_model(): - """ - Test each of the trainer options - :return: - """ - reset_seed() - - trainer_options = dict( - gradient_clip_val=1.0, - overfit_pct=0.20, - track_grad_norm=2, - print_nan_grads=True, - show_progress_bar=False, - logger=get_test_tube_logger(), - accumulate_grad_batches=2, - max_nb_epochs=1, - train_percent_check=0.4, - val_percent_check=0.4 - ) - - model, hparams = get_model() - run_gpu_model_test(trainer_options, model, hparams, on_gpu=False) - - -def test_single_gpu_model(): - """ - Make sure single GPU works (DP mode) - :return: - """ - reset_seed() - - if not torch.cuda.is_available(): - warnings.warn('test_single_gpu_model cannot run.' - ' Rerun on a GPU node to run this test') - return - model, hparams = get_model() - - trainer_options = dict( - show_progress_bar=False, - max_nb_epochs=1, - train_percent_check=0.1, - val_percent_check=0.1, - gpus=1 - ) - - run_gpu_model_test(trainer_options, model, hparams) - - -def test_multi_gpu_none_backend(): - """ - Make sure when using multiple GPUs the user can't use - distributed_backend = None - :return: - """ - reset_seed() - - if not can_run_gpu_test(): - return - - model, hparams = get_model() - trainer_options = dict( - show_progress_bar=False, - max_nb_epochs=1, - train_percent_check=0.1, - val_percent_check=0.1, - gpus='-1' - ) - - with pytest.raises(MisconfigurationException): - run_gpu_model_test(trainer_options, model, hparams) - - -def test_multi_gpu_model_dp(): - """ - Make sure DP works - :return: - """ - reset_seed() - - if not can_run_gpu_test(): - return - - model, hparams = get_model() - trainer_options = dict( - show_progress_bar=False, - distributed_backend='dp', - max_nb_epochs=1, - train_percent_check=0.1, - val_percent_check=0.1, - gpus='-1' - ) - - run_gpu_model_test(trainer_options, model, hparams) - - # test memory helper functions - memory.get_gpu_memory_map() - - -def test_amp_gpu_dp(): - """ - Make sure DP + AMP work - :return: - """ - reset_seed() - - if not can_run_gpu_test(): - return - - model, hparams = get_model() - trainer_options = dict( - max_nb_epochs=1, - gpus='0, 1', # test init with gpu string - distributed_backend='dp', - use_amp=True - ) - with pytest.raises(MisconfigurationException): - run_gpu_model_test(trainer_options, model, hparams) - - -def test_ddp_sampler_error(): - """ - Make sure DDP + AMP work - :return: - """ - if not can_run_gpu_test(): - return - - reset_seed() - set_random_master_port() - - hparams = get_hparams() - model = LightningTestModel(hparams, force_remove_distributed_sampler=True) - - logger = get_test_tube_logger(True) - - trainer = Trainer( - logger=logger, - show_progress_bar=False, - max_nb_epochs=1, - gpus=[0, 1], - distributed_backend='ddp', - use_amp=True - ) - - with pytest.warns(UserWarning): - trainer.get_dataloaders(model) - - clear_save_dir() - - -def test_multiple_val_dataloader(): - """ - Verify multiple val_dataloader - :return: - """ - reset_seed() - - class CurrentTestModel( - LightningValidationMultipleDataloadersMixin, - LightningTestModelBase - ): - pass - - hparams = get_hparams() - model = CurrentTestModel(hparams) - - # logger file to get meta - trainer_options = dict( - max_nb_epochs=1, - val_percent_check=0.1, - train_percent_check=1.0, - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # verify training completed - assert result == 1 - - # verify there are 2 val loaders - assert len(trainer.get_val_dataloaders()) == 2, \ - 'Multiple val_dataloaders not initiated properly' - - # make sure predictions are good for each val set - [run_prediction(dataloader, trainer.model) for dataloader in trainer.get_val_dataloaders()] - - -def test_multiple_test_dataloader(): - """ - Verify multiple test_dataloader - :return: - """ - reset_seed() - - class CurrentTestModel( - LightningTestMultipleDataloadersMixin, - LightningTestModelBase - ): - pass - - hparams = get_hparams() - model = CurrentTestModel(hparams) - - # logger file to get meta - trainer_options = dict( - max_nb_epochs=1, - val_percent_check=0.1, - train_percent_check=0.1, - ) - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # verify there are 2 val loaders - assert len(trainer.get_test_dataloaders()) == 2, \ - 'Multiple test_dataloaders not initiated properly' - - # make sure predictions are good for each test set - [run_prediction(dataloader, trainer.model) for dataloader in trainer.get_test_dataloaders()] - - # run the test method - trainer.test() - - -@pytest.fixture -def mocked_device_count(monkeypatch): - def device_count(): - return PRETEND_N_OF_GPUS - - monkeypatch.setattr(torch.cuda, 'device_count', device_count) - - -@pytest.fixture -def mocked_device_count_0(monkeypatch): - def device_count(): - return 0 - - monkeypatch.setattr(torch.cuda, 'device_count', device_count) - - -test_num_gpus_data = [ - pytest.param(None, 0, None, id="None - expect 0 gpu to use."), - pytest.param(0, 0, None, id="Oth gpu, expect 1 gpu to use."), - pytest.param(1, 1, None, id="1st gpu, expect 1 gpu to use."), - pytest.param(-1, PRETEND_N_OF_GPUS, "ddp", id="-1 - use all gpus"), - pytest.param('-1', PRETEND_N_OF_GPUS, "ddp", id="'-1' - use all gpus"), - pytest.param(3, 3, "ddp", id="3rd gpu - 1 gpu to use (backend:ddp)") -] - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize(["gpus", "expected_num_gpus", "distributed_backend"], test_num_gpus_data) -def test_trainer_gpu_parse(mocked_device_count, gpus, expected_num_gpus, distributed_backend): - assert Trainer(gpus=gpus, distributed_backend=distributed_backend).num_gpus == expected_num_gpus - - -test_num_gpus_data_0 = [ - pytest.param(None, 0, None, id="None - expect 0 gpu to use."), - pytest.param(None, 0, "ddp", id="None - expect 0 gpu to use."), -] - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize(["gpus", "expected_num_gpus", "distributed_backend"], test_num_gpus_data_0) -def test_trainer_num_gpu_0(mocked_device_count_0, gpus, expected_num_gpus, distributed_backend): - assert Trainer(gpus=gpus, distributed_backend=distributed_backend).num_gpus == expected_num_gpus - - -test_root_gpu_data = [ - pytest.param(None, None, "ddp", id="None is None"), - pytest.param(0, None, "ddp", id="O gpus, expect gpu root device to be None."), - pytest.param(1, 0, "ddp", id="1 gpu, expect gpu root device to be 0."), - pytest.param(-1, 0, "ddp", id="-1 - use all gpus, expect gpu root device to be 0."), - pytest.param('-1', 0, "ddp", id="'-1' - use all gpus, expect gpu root device to be 0."), - pytest.param(3, 0, "ddp", id="3 gpus, expect gpu root device to be 0.(backend:ddp)")] - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize(['gpus', 'expected_root_gpu', "distributed_backend"], test_root_gpu_data) -def test_root_gpu_property(mocked_device_count, gpus, expected_root_gpu, distributed_backend): - assert Trainer(gpus=gpus, distributed_backend=distributed_backend).root_gpu == expected_root_gpu - - -test_root_gpu_data_for_0_devices_passing = [ - pytest.param(None, None, None, id="None is None"), - pytest.param(None, None, "ddp", id="None is None"), - pytest.param(0, None, "ddp", id="None is None"), -] - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize([ - 'gpus', 'expected_root_gpu', "distributed_backend"], test_root_gpu_data_for_0_devices_passing) -def test_root_gpu_property_0_passing( - mocked_device_count_0, gpus, expected_root_gpu, distributed_backend): - assert Trainer(gpus=gpus, distributed_backend=distributed_backend).root_gpu == expected_root_gpu - - -# Asking for a gpu when non are available will result in a MisconfigurationException -test_root_gpu_data_for_0_devices_raising = [ - pytest.param(1, None, "ddp"), - pytest.param(3, None, "ddp"), - pytest.param(3, None, "ddp"), - pytest.param([1, 2], None, "ddp"), - pytest.param([0, 1], None, "ddp"), - pytest.param(-1, None, "ddp"), - pytest.param('-1', None, "ddp") -] - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize([ - 'gpus', 'expected_root_gpu', "distributed_backend"], test_root_gpu_data_for_0_devices_raising) -def test_root_gpu_property_0_raising( - mocked_device_count_0, gpus, expected_root_gpu, distributed_backend): - with pytest.raises(MisconfigurationException): - Trainer(gpus=gpus, distributed_backend=distributed_backend).root_gpu - - -test_determine_root_gpu_device_data = [ - pytest.param(None, None, id="No gpus, expect gpu root device to be None"), - pytest.param([0], 0, id="Oth gpu, expect gpu root device to be 0."), - pytest.param([1], 1, id="1st gpu, expect gpu root device to be 1."), - pytest.param([3], 3, id="3rd gpu, expect gpu root device to be 3."), - pytest.param([1, 2], 1, id="[1, 2] gpus, expect gpu root device to be 1."), -] - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize(['gpus', 'expected_root_gpu'], test_determine_root_gpu_device_data) -def test_determine_root_gpu_device(gpus, expected_root_gpu): - assert determine_root_gpu_device(gpus) == expected_root_gpu - - -test_parse_gpu_ids_data = [ - pytest.param(None, None), - pytest.param(0, None), - pytest.param(1, [0]), - pytest.param(-1, list(range(PRETEND_N_OF_GPUS)), id="-1 - use all gpus"), - pytest.param('-1', list(range(PRETEND_N_OF_GPUS)), id="'-1' - use all gpus"), - pytest.param(3, [0, 1, 2])] - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize(['gpus', 'expected_gpu_ids'], test_parse_gpu_ids_data) -def test_parse_gpu_ids(mocked_device_count, gpus, expected_gpu_ids): - assert parse_gpu_ids(gpus) == expected_gpu_ids - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize("gpus", [[1, 2, 19], -1, '-1']) -def test_parse_gpu_fail_on_non_existant_id(mocked_device_count_0, gpus): - with pytest.raises(MisconfigurationException): - parse_gpu_ids(gpus) - - -@pytest.mark.gpus_param_tests -def test_parse_gpu_fail_on_non_existant_id_2(mocked_device_count): - with pytest.raises(MisconfigurationException): - parse_gpu_ids([1, 2, 19]) - - -@pytest.mark.gpus_param_tests -@pytest.mark.parametrize("gpus", [-1, '-1']) -def test_parse_gpu_returns_None_when_no_devices_are_available(mocked_device_count_0, gpus): - with pytest.raises(MisconfigurationException): - parse_gpu_ids(gpus) - - -# ------------------------------------------------------------------------ -# UTILS -# ------------------------------------------------------------------------ -def run_model_test_no_loggers(trainer_options, model, hparams, on_gpu=True, min_acc=0.50): - save_dir = init_save_dir() - trainer_options['default_save_path'] = save_dir - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # correct result and ok accuracy - assert result == 1, 'amp + ddp model failed to complete' - - # test model loading - pretrained_model = load_model(trainer.logger.experiment, - trainer.checkpoint_callback.filepath) - - # test new model accuracy - for dataloader in model.test_dataloader(): - run_prediction(dataloader, pretrained_model, min_acc=min_acc) - - if trainer.use_ddp: - # on hpc this would work fine... but need to hack it for the purpose of the test - trainer.model = pretrained_model - trainer.optimizers, trainer.lr_schedulers = pretrained_model.configure_optimizers() - - clear_save_dir() - - -def run_gpu_model_test(trainer_options, model, hparams, on_gpu=True): - save_dir = init_save_dir() - - # logger file to get meta - logger = get_test_tube_logger(False) - - # logger file to get weights - checkpoint = init_checkpoint_callback(logger) - - # add these to the trainer options - trainer_options['checkpoint_callback'] = checkpoint - trainer_options['logger'] = logger - - # fit model - trainer = Trainer(**trainer_options) - result = trainer.fit(model) - - # correct result and ok accuracy - assert result == 1, 'amp + ddp model failed to complete' - - # test model loading - pretrained_model = load_model(logger.experiment, trainer.checkpoint_callback.filepath) - - # test new model accuracy - [run_prediction(dataloader, pretrained_model) for dataloader in model.test_dataloader()] - - if trainer.use_ddp or trainer.use_ddp2: - # on hpc this would work fine... but need to hack it for the purpose of the test - trainer.model = pretrained_model - trainer.optimizers, trainer.lr_schedulers = pretrained_model.configure_optimizers() - - # test HPC loading / saving - trainer.hpc_save(save_dir, logger) - trainer.hpc_load(save_dir, on_gpu=on_gpu) - - clear_save_dir() - - -def get_hparams(continue_training=False, hpc_exp_number=0): - root_dir = os.path.dirname(os.path.realpath(__file__)) - - args = { - 'drop_prob': 0.2, - 'batch_size': 32, - 'in_features': 28 * 28, - 'learning_rate': 0.001 * 8, - 'optimizer_name': 'adam', - 'data_root': os.path.join(root_dir, 'mnist'), - 'out_features': 10, - 'hidden_dim': 1000} - - if continue_training: - args['test_tube_do_checkpoint_load'] = True - args['hpc_exp_number'] = hpc_exp_number - - hparams = Namespace(**args) - return hparams - - -def get_model(use_test_model=False, lbfgs=False): - # set up model with these hyperparams - hparams = get_hparams() - if lbfgs: - setattr(hparams, 'optimizer_name', 'lbfgs') - setattr(hparams, 'learning_rate', 0.002) - - if use_test_model: - model = LightningTestModel(hparams) - else: - model = LightningTemplateModel(hparams) - - return model, hparams - - -def get_test_tube_logger(debug=True, version=None): - # set up logger object without actually saving logs - root_dir = os.path.dirname(os.path.realpath(__file__)) - save_dir = os.path.join(root_dir, 'save_dir') - logger = TestTubeLogger(save_dir, name='lightning_logs', debug=False, version=version) - return logger - - -def init_save_dir(): - root_dir = os.path.dirname(os.path.realpath(__file__)) - save_dir = os.path.join(root_dir, 'tests', 'save_dir') - - if os.path.exists(save_dir): - n = RANDOM_FILE_PATHS.pop() - shutil.move(save_dir, save_dir + f'_{n}') - - os.makedirs(save_dir, exist_ok=True) - - return save_dir - - -def clear_save_dir(): - root_dir = os.path.dirname(os.path.realpath(__file__)) - save_dir = os.path.join(root_dir, 'save_dir') - if os.path.exists(save_dir): - n = RANDOM_FILE_PATHS.pop() - shutil.move(save_dir, save_dir + f'_{n}') - - -def load_model(exp, root_weights_dir, module_class=LightningTemplateModel): - # load trained model - tags_path = exp.get_data_path(exp.name, exp.version) - tags_path = os.path.join(tags_path, 'meta_tags.csv') - - checkpoints = [x for x in os.listdir(root_weights_dir) if '.ckpt' in x] - weights_dir = os.path.join(root_weights_dir, checkpoints[0]) - - trained_model = module_class.load_from_metrics(weights_path=weights_dir, - tags_csv=tags_path) - - assert trained_model is not None, 'loading model failed' - - return trained_model - - -def run_prediction(dataloader, trained_model, dp=False, min_acc=0.50): - # run prediction on 1 batch - for batch in dataloader: - break - - x, y = batch - x = x.view(x.size(0), -1) - - if dp: - output = trained_model(batch, 0) - acc = output['val_acc'] - acc = torch.mean(acc).item() - - else: - y_hat = trained_model(x) - - # acc - labels_hat = torch.argmax(y_hat, dim=1) - acc = torch.sum(y == labels_hat).item() / (len(y) * 1.0) - acc = torch.tensor(acc) - acc = acc.item() - - assert acc > min_acc, f'this model is expected to get > {min_acc} in test set (it got {acc})' - - -def assert_ok_val_acc(trainer): - # this model should get 0.80+ acc - acc = trainer.training_tqdm_dict['val_acc'] - assert acc > 0.50, f'model failed to get expected 0.50 validation accuracy. Got: {acc}' - - -def assert_ok_test_acc(trainer): - # this model should get 0.80+ acc - acc = trainer.training_tqdm_dict['test_acc'] - assert acc > 0.50, f'model failed to get expected 0.50 validation accuracy. Got: {acc}' - - -def can_run_gpu_test(): - if not torch.cuda.is_available(): - warnings.warn('test_multi_gpu_model_ddp cannot run.' - ' Rerun on a GPU node to run this test') - return False - if not torch.cuda.device_count() > 1: - warnings.warn('test_multi_gpu_model_ddp cannot run.' - ' Rerun on a node with 2+ GPUs to run this test') - return False - return True - - -def reset_seed(): - SEED = RANDOM_SEEDS.pop() - torch.manual_seed(SEED) - np.random.seed(SEED) - - -def set_random_master_port(): - port = RANDOM_PORTS.pop() - os.environ['MASTER_PORT'] = str(port) - - -def init_checkpoint_callback(logger): - exp = logger.experiment - exp_path = exp.get_data_path(exp.name, exp.version) - ckpt_dir = os.path.join(exp_path, 'checkpoints') - checkpoint = ModelCheckpoint(ckpt_dir) - return checkpoint - - -if __name__ == '__main__': - pytest.main([__file__]) diff --git a/tests/test_trainer.py b/tests/test_trainer.py new file mode 100644 index 00000000..667143a8 --- /dev/null +++ b/tests/test_trainer.py @@ -0,0 +1,324 @@ +import os +import pytest +import torch + +from pytorch_lightning import Trainer +from pytorch_lightning.callbacks import ( + ModelCheckpoint, +) +from pytorch_lightning.testing import ( + LightningTestModel, + LightningTestModelBase, + LightningValidationStepMixin, + LightningValidationMultipleDataloadersMixin, + LightningTestMixin, + LightningTestMultipleDataloadersMixin, +) +from pytorch_lightning.trainer import trainer_io +from pytorch_lightning.trainer.logging_mixin import TrainerLoggingMixin +from . import testing_utils + + +def test_no_val_module(): + """ + Tests use case where trainer saves the model, and user loads it from tags independently + :return: + """ + testing_utils.reset_seed() + + hparams = testing_utils.get_hparams() + + class CurrentTestModel(LightningTestModelBase): + pass + + model = CurrentTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + logger = testing_utils.get_test_tube_logger(False) + + trainer_options = dict( + max_nb_epochs=1, + logger=logger, + checkpoint_callback=ModelCheckpoint(save_dir) + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # training complete + assert result == 1, 'amp + ddp model failed to complete' + + # save model + new_weights_path = os.path.join(save_dir, 'save_test.ckpt') + trainer.save_checkpoint(new_weights_path) + + # load new model + tags_path = logger.experiment.get_data_path(logger.experiment.name, logger.experiment.version) + tags_path = os.path.join(tags_path, 'meta_tags.csv') + model_2 = LightningTestModel.load_from_metrics(weights_path=new_weights_path, + tags_csv=tags_path) + model_2.eval() + + # make prediction + testing_utils.clear_save_dir() + + +def test_no_val_end_module(): + """ + Tests use case where trainer saves the model, and user loads it from tags independently + :return: + """ + testing_utils.reset_seed() + + class CurrentTestModel(LightningValidationStepMixin, LightningTestModelBase): + pass + + hparams = testing_utils.get_hparams() + model = CurrentTestModel(hparams) + + save_dir = testing_utils.init_save_dir() + + # logger file to get meta + logger = testing_utils.get_test_tube_logger(False) + + trainer_options = dict( + max_nb_epochs=1, + logger=logger, + checkpoint_callback=ModelCheckpoint(save_dir) + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # traning complete + assert result == 1, 'amp + ddp model failed to complete' + + # save model + new_weights_path = os.path.join(save_dir, 'save_test.ckpt') + trainer.save_checkpoint(new_weights_path) + + # load new model + tags_path = logger.experiment.get_data_path(logger.experiment.name, logger.experiment.version) + tags_path = os.path.join(tags_path, 'meta_tags.csv') + model_2 = LightningTestModel.load_from_metrics(weights_path=new_weights_path, + tags_csv=tags_path) + model_2.eval() + + # make prediction + testing_utils.clear_save_dir() + + +def test_gradient_accumulation_scheduling(): + testing_utils.reset_seed() + + """ + Test grad accumulation by the freq of optimizer updates + """ + # test incorrect configs + with pytest.raises(IndexError): + assert Trainer(accumulate_grad_batches={0: 3, 1: 4, 4: 6}) + assert Trainer(accumulate_grad_batches={-2: 3}) + + with pytest.raises(TypeError): + assert Trainer(accumulate_grad_batches={}) + assert Trainer(accumulate_grad_batches=[[2, 3], [4, 6]]) + assert Trainer(accumulate_grad_batches={1: 2, 3.: 4}) + assert Trainer(accumulate_grad_batches={1: 2.5, 3: 5}) + + # test optimizer call freq matches scheduler + def optimizer_step(self, epoch_nb, batch_nb, optimizer, optimizer_i, second_order_closure=None): + # only test the first 12 batches in epoch + if batch_nb < 12: + if epoch_nb == 0: + # reset counter when starting epoch + if batch_nb == 0: + self.prev_called_batch_nb = 0 + + # use this opportunity to test once + assert self.trainer.accumulate_grad_batches == 1 + + assert batch_nb == self.prev_called_batch_nb + self.prev_called_batch_nb += 1 + + elif 1 <= epoch_nb <= 2: + # reset counter when starting epoch + if batch_nb == 1: + self.prev_called_batch_nb = 1 + + # use this opportunity to test once + assert self.trainer.accumulate_grad_batches == 2 + + assert batch_nb == self.prev_called_batch_nb + self.prev_called_batch_nb += 2 + + else: + if batch_nb == 3: + self.prev_called_batch_nb = 3 + + # use this opportunity to test once + assert self.trainer.accumulate_grad_batches == 4 + + assert batch_nb == self.prev_called_batch_nb + self.prev_called_batch_nb += 3 + + optimizer.step() + + # clear gradients + optimizer.zero_grad() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + schedule = {1: 2, 3: 4} + + trainer = Trainer(accumulate_grad_batches=schedule, + train_percent_check=0.1, + val_percent_check=0.1, + max_nb_epochs=4) + + # for the test + trainer.optimizer_step = optimizer_step + model.prev_called_batch_nb = 0 + + trainer.fit(model) + + +def test_loading_meta_tags(): + testing_utils.reset_seed() + + from argparse import Namespace + hparams = testing_utils.get_hparams() + + # save tags + logger = testing_utils.get_test_tube_logger(False) + logger.log_hyperparams(Namespace(some_str='a_str', an_int=1, a_float=2.0)) + logger.log_hyperparams(hparams) + logger.save() + + # load tags + tags_path = logger.experiment.get_data_path( + logger.experiment.name, logger.experiment.version + ) + '/meta_tags.csv' + tags = trainer_io.load_hparams_from_tags_csv(tags_path) + + assert tags.batch_size == 32 and tags.hidden_dim == 1000 + + testing_utils.clear_save_dir() + + +def test_dp_output_reduce(): + mixin = TrainerLoggingMixin() + testing_utils.reset_seed() + + # test identity when we have a single gpu + out = torch.rand(3, 1) + assert mixin.reduce_distributed_output(out, nb_gpus=1) is out + + # average when we have multiples + assert mixin.reduce_distributed_output(out, nb_gpus=2) == out.mean() + + # when we have a dict of vals + out = { + 'a': out, + 'b': { + 'c': out + } + } + reduced = mixin.reduce_distributed_output(out, nb_gpus=3) + assert reduced['a'] == out['a'] + assert reduced['b']['c'] == out['b']['c'] + + +def test_model_freeze_unfreeze(): + testing_utils.reset_seed() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + model.freeze() + model.unfreeze() + + +def test_multiple_val_dataloader(): + """ + Verify multiple val_dataloader + :return: + """ + testing_utils.reset_seed() + + class CurrentTestModel( + LightningValidationMultipleDataloadersMixin, + LightningTestModelBase + ): + pass + + hparams = testing_utils.get_hparams() + model = CurrentTestModel(hparams) + + # logger file to get meta + trainer_options = dict( + max_nb_epochs=1, + val_percent_check=0.1, + train_percent_check=1.0, + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # verify training completed + assert result == 1 + + # verify there are 2 val loaders + assert len(trainer.get_val_dataloaders()) == 2, \ + 'Multiple val_dataloaders not initiated properly' + + # make sure predictions are good for each val set + for dataloader in trainer.get_val_dataloaders(): + testing_utils.run_prediction(dataloader, trainer.model) + + +def test_multiple_test_dataloader(): + """ + Verify multiple test_dataloader + :return: + """ + testing_utils.reset_seed() + + class CurrentTestModel( + LightningTestMultipleDataloadersMixin, + LightningTestModelBase + ): + pass + + hparams = testing_utils.get_hparams() + model = CurrentTestModel(hparams) + + # logger file to get meta + trainer_options = dict( + max_nb_epochs=1, + val_percent_check=0.1, + train_percent_check=0.1, + ) + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # verify there are 2 val loaders + assert len(trainer.get_test_dataloaders()) == 2, \ + 'Multiple test_dataloaders not initiated properly' + + # make sure predictions are good for each test set + for dataloader in trainer.get_test_dataloaders(): + testing_utils.run_prediction(dataloader, trainer.model) + + # run the test method + trainer.test() + + +if __name__ == '__main__': + pytest.main([__file__]) diff --git a/tests/test_logging.py b/tests/test_y_logging.py similarity index 91% rename from tests/test_logging.py rename to tests/test_y_logging.py index 9db65750..39b7edfc 100644 --- a/tests/test_logging.py +++ b/tests/test_y_logging.py @@ -5,7 +5,7 @@ import torch from pytorch_lightning import Trainer from pytorch_lightning.testing import LightningTestModel -from .test_models import get_hparams, get_test_tube_logger, init_save_dir, clear_save_dir +from . import testing_utils RANDOM_FILE_PATHS = list(np.random.randint(12000, 19000, 1000)) ROOT_SEED = 1234 @@ -19,12 +19,12 @@ def test_testtube_logger(): verify that basic functionality of test tube logger works """ reset_seed() - hparams = get_hparams() + hparams = testing_utils.get_hparams() model = LightningTestModel(hparams) - save_dir = init_save_dir() + save_dir = testing_utils.init_save_dir() - logger = get_test_tube_logger(False) + logger = testing_utils.get_test_tube_logger(False) trainer_options = dict( max_nb_epochs=1, @@ -37,7 +37,7 @@ def test_testtube_logger(): assert result == 1, "Training failed" - clear_save_dir() + testing_utils.clear_save_dir() def test_testtube_pickle(): @@ -46,12 +46,12 @@ def test_testtube_pickle(): """ reset_seed() - hparams = get_hparams() + hparams = testing_utils.get_hparams() model = LightningTestModel(hparams) - save_dir = init_save_dir() + save_dir = testing_utils.init_save_dir() - logger = get_test_tube_logger(False) + logger = testing_utils.get_test_tube_logger(False) logger.log_hyperparams(hparams) logger.save() @@ -66,7 +66,7 @@ def test_testtube_pickle(): trainer2 = pickle.loads(pkl_bytes) trainer2.logger.log_metrics({"acc": 1.0}) - clear_save_dir() + testing_utils.clear_save_dir() # def test_mlflow_logger(): diff --git a/tests/test_z_amp.py b/tests/test_z_amp.py new file mode 100644 index 00000000..faae0707 --- /dev/null +++ b/tests/test_z_amp.py @@ -0,0 +1,208 @@ +import os +import warnings + +import pytest +import torch + +from pytorch_lightning import Trainer +from pytorch_lightning.testing import ( + LightningTestModel, +) +from pytorch_lightning.utilities.debugging import MisconfigurationException +from . import testing_utils + + +def test_amp_single_gpu(): + """ + Make sure DDP + AMP work + :return: + """ + testing_utils.reset_seed() + + if not testing_utils.can_run_gpu_test(): + return + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + trainer_options = dict( + show_progress_bar=True, + max_nb_epochs=1, + gpus=1, + distributed_backend='ddp', + use_amp=True + ) + + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + +def test_no_amp_single_gpu(): + """ + Make sure DDP + AMP work + :return: + """ + testing_utils.reset_seed() + + if not testing_utils.can_run_gpu_test(): + return + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + trainer_options = dict( + show_progress_bar=True, + max_nb_epochs=1, + gpus=1, + distributed_backend='dp', + use_amp=True + ) + + with pytest.raises((MisconfigurationException, ModuleNotFoundError)): + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + +def test_amp_gpu_ddp(): + """ + Make sure DDP + AMP work + :return: + """ + if not testing_utils.can_run_gpu_test(): + return + + testing_utils.reset_seed() + testing_utils.set_random_master_port() + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + trainer_options = dict( + show_progress_bar=True, + max_nb_epochs=1, + gpus=2, + distributed_backend='ddp', + use_amp=True + ) + + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + +def test_amp_gpu_ddp_slurm_managed(): + """ + Make sure DDP + AMP work + :return: + """ + if not testing_utils.can_run_gpu_test(): + return + + testing_utils.reset_seed() + + # simulate setting slurm flags + testing_utils.set_random_master_port() + os.environ['SLURM_LOCALID'] = str(0) + + hparams = testing_utils.get_hparams() + model = LightningTestModel(hparams) + + trainer_options = dict( + show_progress_bar=True, + max_nb_epochs=1, + gpus=[0], + distributed_backend='ddp', + use_amp=True + ) + + save_dir = testing_utils.init_save_dir() + + # exp file to get meta + logger = testing_utils.get_test_tube_logger(False) + + # exp file to get weights + checkpoint = testing_utils.init_checkpoint_callback(logger) + + # add these to the trainer options + trainer_options['checkpoint_callback'] = checkpoint + trainer_options['logger'] = logger + + # fit model + trainer = Trainer(**trainer_options) + trainer.is_slurm_managing_tasks = True + result = trainer.fit(model) + + # correct result and ok accuracy + assert result == 1, 'amp + ddp model failed to complete' + + # test root model address + assert trainer.resolve_root_node_address('abc') == 'abc' + assert trainer.resolve_root_node_address('abc[23]') == 'abc23' + assert trainer.resolve_root_node_address('abc[23-24]') == 'abc23' + assert trainer.resolve_root_node_address('abc[23-24, 45-40, 40]') == 'abc23' + + # test model loading with a map_location + pretrained_model = testing_utils.load_model(logger.experiment, + trainer.checkpoint_callback.filepath) + + # test model preds + for dataloader in trainer.get_test_dataloaders(): + testing_utils.run_prediction(dataloader, pretrained_model) + + if trainer.use_ddp: + # on hpc this would work fine... but need to hack it for the purpose of the test + trainer.model = pretrained_model + trainer.optimizers, trainer.lr_schedulers = pretrained_model.configure_optimizers() + + # test HPC loading / saving + trainer.hpc_save(save_dir, logger) + trainer.hpc_load(save_dir, on_gpu=True) + + # test freeze on gpu + model.freeze() + model.unfreeze() + + testing_utils.clear_save_dir() + + +def test_cpu_model_with_amp(): + """ + Make sure model trains on CPU + :return: + """ + testing_utils.reset_seed() + + trainer_options = dict( + show_progress_bar=False, + logger=testing_utils.get_test_tube_logger(), + max_nb_epochs=1, + train_percent_check=0.4, + val_percent_check=0.4, + use_amp=True + ) + + model, hparams = testing_utils.get_model() + + with pytest.raises((MisconfigurationException, ModuleNotFoundError)): + testing_utils.run_gpu_model_test(trainer_options, model, hparams, on_gpu=False) + + +def test_amp_gpu_dp(): + """ + Make sure DP + AMP work + :return: + """ + testing_utils.reset_seed() + + if not testing_utils.can_run_gpu_test(): + return + + model, hparams = testing_utils.get_model() + trainer_options = dict( + max_nb_epochs=1, + gpus='0, 1', # test init with gpu string + distributed_backend='dp', + use_amp=True + ) + with pytest.raises(MisconfigurationException): + testing_utils.run_gpu_model_test(trainer_options, model, hparams) + + +if __name__ == '__main__': + pytest.main([__file__]) diff --git a/tests/testing_utils.py b/tests/testing_utils.py new file mode 100644 index 00000000..86d705f9 --- /dev/null +++ b/tests/testing_utils.py @@ -0,0 +1,239 @@ +import os +import shutil +import warnings +from argparse import Namespace + +import numpy as np +import torch + +from pl_examples import LightningTemplateModel +from pytorch_lightning import Trainer +from pytorch_lightning.callbacks import ( + ModelCheckpoint, +) +from pytorch_lightning.logging import TestTubeLogger +from pytorch_lightning.testing import ( + LightningTestModel, +) + +# generate a list of random seeds for each test +RANDOM_FILE_PATHS = list(np.random.randint(12000, 19000, 1000)) +RANDOM_PORTS = list(np.random.randint(12000, 19000, 1000)) +ROOT_SEED = 1234 +torch.manual_seed(ROOT_SEED) +np.random.seed(ROOT_SEED) +RANDOM_SEEDS = list(np.random.randint(0, 10000, 1000)) + + +def run_model_test_no_loggers(trainer_options, model, hparams, on_gpu=True, min_acc=0.50): + save_dir = init_save_dir() + trainer_options['default_save_path'] = save_dir + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # correct result and ok accuracy + assert result == 1, 'amp + ddp model failed to complete' + + # test model loading + pretrained_model = load_model(trainer.logger.experiment, + trainer.checkpoint_callback.filepath) + + # test new model accuracy + for dataloader in model.test_dataloader(): + run_prediction(dataloader, pretrained_model, min_acc=min_acc) + + if trainer.use_ddp: + # on hpc this would work fine... but need to hack it for the purpose of the test + trainer.model = pretrained_model + trainer.optimizers, trainer.lr_schedulers = pretrained_model.configure_optimizers() + + clear_save_dir() + + +def run_gpu_model_test(trainer_options, model, hparams, on_gpu=True): + save_dir = init_save_dir() + + # logger file to get meta + logger = get_test_tube_logger(False) + + # logger file to get weights + checkpoint = init_checkpoint_callback(logger) + + # add these to the trainer options + trainer_options['checkpoint_callback'] = checkpoint + trainer_options['logger'] = logger + + # fit model + trainer = Trainer(**trainer_options) + result = trainer.fit(model) + + # correct result and ok accuracy + assert result == 1, 'amp + ddp model failed to complete' + + # test model loading + pretrained_model = load_model(logger.experiment, trainer.checkpoint_callback.filepath) + + # test new model accuracy + [run_prediction(dataloader, pretrained_model) for dataloader in model.test_dataloader()] + + if trainer.use_ddp or trainer.use_ddp2: + # on hpc this would work fine... but need to hack it for the purpose of the test + trainer.model = pretrained_model + trainer.optimizers, trainer.lr_schedulers = pretrained_model.configure_optimizers() + + # test HPC loading / saving + trainer.hpc_save(save_dir, logger) + trainer.hpc_load(save_dir, on_gpu=on_gpu) + + clear_save_dir() + + +def get_hparams(continue_training=False, hpc_exp_number=0): + root_dir = os.path.dirname(os.path.realpath(__file__)) + + args = { + 'drop_prob': 0.2, + 'batch_size': 32, + 'in_features': 28 * 28, + 'learning_rate': 0.001 * 8, + 'optimizer_name': 'adam', + 'data_root': os.path.join(root_dir, 'mnist'), + 'out_features': 10, + 'hidden_dim': 1000} + + if continue_training: + args['test_tube_do_checkpoint_load'] = True + args['hpc_exp_number'] = hpc_exp_number + + hparams = Namespace(**args) + return hparams + + +def get_model(use_test_model=False, lbfgs=False): + # set up model with these hyperparams + hparams = get_hparams() + if lbfgs: + setattr(hparams, 'optimizer_name', 'lbfgs') + setattr(hparams, 'learning_rate', 0.002) + + if use_test_model: + model = LightningTestModel(hparams) + else: + model = LightningTemplateModel(hparams) + + return model, hparams + + +def get_test_tube_logger(debug=True, version=None): + # set up logger object without actually saving logs + root_dir = os.path.dirname(os.path.realpath(__file__)) + save_dir = os.path.join(root_dir, 'save_dir') + logger = TestTubeLogger(save_dir, name='lightning_logs', debug=False, version=version) + return logger + + +def init_save_dir(): + root_dir = os.path.dirname(os.path.realpath(__file__)) + save_dir = os.path.join(root_dir, 'tests', 'save_dir') + + if os.path.exists(save_dir): + n = RANDOM_FILE_PATHS.pop() + shutil.move(save_dir, save_dir + f'_{n}') + + os.makedirs(save_dir, exist_ok=True) + + return save_dir + + +def clear_save_dir(): + root_dir = os.path.dirname(os.path.realpath(__file__)) + save_dir = os.path.join(root_dir, 'save_dir') + if os.path.exists(save_dir): + n = RANDOM_FILE_PATHS.pop() + shutil.move(save_dir, save_dir + f'_{n}') + + +def load_model(exp, root_weights_dir, module_class=LightningTemplateModel): + # load trained model + tags_path = exp.get_data_path(exp.name, exp.version) + tags_path = os.path.join(tags_path, 'meta_tags.csv') + + checkpoints = [x for x in os.listdir(root_weights_dir) if '.ckpt' in x] + weights_dir = os.path.join(root_weights_dir, checkpoints[0]) + + trained_model = module_class.load_from_metrics(weights_path=weights_dir, + tags_csv=tags_path) + + assert trained_model is not None, 'loading model failed' + + return trained_model + + +def run_prediction(dataloader, trained_model, dp=False, min_acc=0.50): + # run prediction on 1 batch + for batch in dataloader: + break + + x, y = batch + x = x.view(x.size(0), -1) + + if dp: + output = trained_model(batch, 0) + acc = output['val_acc'] + acc = torch.mean(acc).item() + + else: + y_hat = trained_model(x) + + # acc + labels_hat = torch.argmax(y_hat, dim=1) + acc = torch.sum(y == labels_hat).item() / (len(y) * 1.0) + acc = torch.tensor(acc) + acc = acc.item() + + assert acc > min_acc, f'this model is expected to get > {min_acc} in test set (it got {acc})' + + +def assert_ok_val_acc(trainer): + # this model should get 0.80+ acc + acc = trainer.training_tqdm_dict['val_acc'] + assert acc > 0.50, f'model failed to get expected 0.50 validation accuracy. Got: {acc}' + + +def assert_ok_test_acc(trainer): + # this model should get 0.80+ acc + acc = trainer.training_tqdm_dict['test_acc'] + assert acc > 0.50, f'model failed to get expected 0.50 validation accuracy. Got: {acc}' + + +def can_run_gpu_test(): + if not torch.cuda.is_available(): + warnings.warn('test_multi_gpu_model_ddp cannot run.' + ' Rerun on a GPU node to run this test') + return False + if not torch.cuda.device_count() > 1: + warnings.warn('test_multi_gpu_model_ddp cannot run.' + ' Rerun on a node with 2+ GPUs to run this test') + return False + return True + + +def reset_seed(): + SEED = RANDOM_SEEDS.pop() + torch.manual_seed(SEED) + np.random.seed(SEED) + + +def set_random_master_port(): + port = RANDOM_PORTS.pop() + os.environ['MASTER_PORT'] = str(port) + + +def init_checkpoint_callback(logger): + exp = logger.experiment + exp_path = exp.get_data_path(exp.name, exp.version) + ckpt_dir = os.path.join(exp_path, 'checkpoints') + checkpoint = ModelCheckpoint(ckpt_dir) + return checkpoint